mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-10 04:56:24 +02:00
Merge pull request #7141 from hrydgard/neon-vfpu-2
The old neon-vfpu branch, rebased on master
This commit is contained in:
commit
a893c213de
15 files changed
+2403
-97
No files matched your search
@@ -1022,6 +1022,7 @@ if(ARM)
|
||||
Core/MIPS/ARM/ArmCompLoadStore.cpp
|
||||
Core/MIPS/ARM/ArmCompVFPU.cpp
|
||||
Core/MIPS/ARM/ArmCompVFPUNEON.cpp
|
||||
Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp
|
||||
Core/MIPS/ARM/ArmCompReplace.cpp
|
||||
Core/MIPS/ARM/ArmJit.cpp
|
||||
Core/MIPS/ARM/ArmJit.h
|
||||
|
||||
+14
-2
@@ -317,6 +317,18 @@
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
|
||||
</ClCompile>
|
||||
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
|
||||
</ClCompile>
|
||||
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEONUtil.cpp">
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
|
||||
</ClCompile>
|
||||
<ClCompile Include="MIPS\ARM\ArmRegCacheFPU.cpp">
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
|
||||
@@ -563,7 +575,7 @@
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
|
||||
</ClInclude>
|
||||
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
|
||||
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEONUtil.h">
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
|
||||
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
|
||||
@@ -658,4 +670,4 @@
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
</Project>
|
||||
@@ -393,6 +393,12 @@
|
||||
<ClCompile Include="MIPS\ARM\ArmCompVFPU.cpp">
|
||||
<Filter>MIPS\ARM</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
|
||||
<Filter>MIPS\ARM</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEONUtil.cpp">
|
||||
<Filter>MIPS\ARM</Filter>
|
||||
</ClCompile>
|
||||
<ClCompile Include="..\ext\disarm.cpp">
|
||||
<Filter>Ext</Filter>
|
||||
</ClCompile>
|
||||
@@ -969,7 +975,7 @@
|
||||
<ClInclude Include="MIPS\JitCommon\JitState.h">
|
||||
<Filter>MIPS\JitCommon</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
|
||||
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEONUtil.h">
|
||||
<Filter>MIPS\ARM</Filter>
|
||||
</ClInclude>
|
||||
<ClInclude Include="Debugger\DisassemblyManager.h">
|
||||
@@ -1027,4 +1033,4 @@
|
||||
<None Include="..\android\jni\Android.mk" />
|
||||
<None Include="GameLogNotes.txt" />
|
||||
</ItemGroup>
|
||||
</Project>
|
||||
</Project>
|
||||
@@ -357,9 +357,12 @@ void Jit::Comp_mxc1(MIPSOpcode op)
|
||||
switch ((op >> 21) & 0x1f)
|
||||
{
|
||||
case 0: // R(rt) = FI(fs); break; //mfc1
|
||||
fpr.MapReg(fs);
|
||||
gpr.MapReg(rt, MAP_DIRTY | MAP_NOINIT);
|
||||
VMOV(gpr.R(rt), fpr.R(fs));
|
||||
if (fpr.IsMapped(fs)) {
|
||||
VMOV(gpr.R(rt), fpr.R(fs));
|
||||
} else {
|
||||
LDR(gpr.R(rt), CTXREG, fpr.GetMipsRegOffset(fs));
|
||||
}
|
||||
return;
|
||||
|
||||
case 2: //cfc1
|
||||
@@ -393,11 +396,11 @@ void Jit::Comp_mxc1(MIPSOpcode op)
|
||||
|
||||
case 4: //FI(fs) = R(rt); break; //mtc1
|
||||
if (rt == MIPS_REG_ZERO) {
|
||||
fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT);
|
||||
fpr.MapReg(fs, MAP_NOINIT);
|
||||
MOVI2F(fpr.R(fs), 0.0f, R0);
|
||||
} else {
|
||||
gpr.MapReg(rt);
|
||||
fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT);
|
||||
fpr.MapReg(fs, MAP_NOINIT);
|
||||
VMOV(fpr.R(fs), gpr.R(rt));
|
||||
}
|
||||
return;
|
||||
|
||||
@@ -324,8 +324,8 @@ namespace MIPSComp
|
||||
|
||||
void Jit::Comp_SVQ(MIPSOpcode op)
|
||||
{
|
||||
CONDITIONAL_DISABLE;
|
||||
NEON_IF_AVAILABLE(CompNEON_SVQ);
|
||||
CONDITIONAL_DISABLE;
|
||||
|
||||
int imm = (signed short)(op&0xFFFC);
|
||||
int vt = (((op >> 16) & 0x1f)) | ((op&1) << 5);
|
||||
@@ -506,6 +506,7 @@ namespace MIPSComp
|
||||
|
||||
void Jit::Comp_VIdt(MIPSOpcode op) {
|
||||
NEON_IF_AVAILABLE(CompNEON_VIdt);
|
||||
|
||||
CONDITIONAL_DISABLE;
|
||||
if (js.HasUnknownPrefix()) {
|
||||
DISABLE;
|
||||
@@ -1574,7 +1575,7 @@ namespace MIPSComp
|
||||
VMUL(fpr.V(temp3), fpr.V(sregs[0]), fpr.V(tregs[1]));
|
||||
VMLS(fpr.V(temp3), fpr.V(sregs[1]), fpr.V(tregs[0]));
|
||||
|
||||
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT);
|
||||
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT);
|
||||
VMOV(fpr.V(dregs[0]), S0);
|
||||
VMOV(fpr.V(dregs[1]), S1);
|
||||
VMOV(fpr.V(dregs[2]), fpr.V(temp3));
|
||||
@@ -1609,7 +1610,7 @@ namespace MIPSComp
|
||||
VMLS(fpr.V(temp4), fpr.V(sregs[2]), fpr.V(tregs[2]));
|
||||
VMLA(fpr.V(temp4), fpr.V(sregs[3]), fpr.V(tregs[3]));
|
||||
|
||||
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT);
|
||||
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT);
|
||||
VMOV(fpr.V(dregs[0]), S0);
|
||||
VMOV(fpr.V(dregs[1]), S1);
|
||||
VMOV(fpr.V(dregs[2]), fpr.V(temp3));
|
||||
@@ -2118,5 +2119,4 @@ namespace MIPSComp
|
||||
void Jit::Comp_Vbfy(MIPSOpcode op) {
|
||||
DISABLE;
|
||||
}
|
||||
|
||||
}
|
||||
+1286
-29
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,419 @@
|
||||
// Copyright (c) 2013- PPSSPP Project.
|
||||
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU General Public License as published by
|
||||
// the Free Software Foundation, version 2.0 or later versions.
|
||||
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU General Public License 2.0 for more details.
|
||||
|
||||
// A copy of the GPL 2.0 should have been included with the program.
|
||||
// If not, see http://www.gnu.org/licenses/
|
||||
|
||||
// Official git repository and contact information can be found at
|
||||
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
|
||||
|
||||
// NEON VFPU
|
||||
// This is where we will create an alternate implementation of the VFPU emulation
|
||||
// that uses NEON Q registers to cache pairs/tris/quads, and so on.
|
||||
// Will require major extensions to the reg cache and other things.
|
||||
|
||||
// ARM NEON can only do pairs and quads, not tris and scalars.
|
||||
// We can do scalars, though, for many operations if all the operands
|
||||
// are below Q8 (D16, S32) using regular VFP instructions but really not sure
|
||||
// if it's worth it.
|
||||
|
||||
#include <cmath>
|
||||
|
||||
#include "base/logging.h"
|
||||
#include "math/math_util.h"
|
||||
|
||||
#include "Common/CPUDetect.h"
|
||||
#include "Core/MemMap.h"
|
||||
#include "Core/MIPS/MIPS.h"
|
||||
#include "Core/MIPS/MIPSAnalyst.h"
|
||||
#include "Core/MIPS/MIPSCodeUtils.h"
|
||||
#include "Core/MIPS/MIPSVFPUUtils.h"
|
||||
#include "Core/Config.h"
|
||||
#include "Core/Reporting.h"
|
||||
|
||||
#include "Core/MIPS/ARM/ArmJit.h"
|
||||
#include "Core/MIPS/ARM/ArmRegCache.h"
|
||||
#include "Core/MIPS/ARM/ArmCompVFPUNEONUtil.h"
|
||||
|
||||
// TODO: Somehow #ifdef away on ARMv5eabi, without breaking the linker.
|
||||
// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
|
||||
|
||||
#define CONDITIONAL_DISABLE ;
|
||||
#define DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
|
||||
|
||||
|
||||
#define _RS MIPS_GET_RS(op)
|
||||
#define _RT MIPS_GET_RT(op)
|
||||
#define _RD MIPS_GET_RD(op)
|
||||
#define _FS MIPS_GET_FS(op)
|
||||
#define _FT MIPS_GET_FT(op)
|
||||
#define _FD MIPS_GET_FD(op)
|
||||
#define _SA MIPS_GET_SA(op)
|
||||
#define _POS ((op>> 6) & 0x1F)
|
||||
#define _SIZE ((op>>11) & 0x1F)
|
||||
#define _IMM16 (signed short)(op & 0xFFFF)
|
||||
#define _IMM26 (op & 0x03FFFFFF)
|
||||
|
||||
namespace MIPSComp {
|
||||
|
||||
static const float minus_one = -1.0f;
|
||||
static const float one = 1.0f;
|
||||
static const float zero = 0.0f;
|
||||
|
||||
// On NEON, we map triples to Q registers and singles to D registers.
|
||||
// Sometimes, as when doing dot products, it matters what's in that unused reg. This zeroes it.
|
||||
void Jit::NEONMaskToSize(ARMReg vs, VectorSize sz) {
|
||||
// TODO
|
||||
}
|
||||
|
||||
ARMReg Jit::NEONMapPrefixST(int mipsReg, VectorSize sz, u32 prefix, int mapFlags) {
|
||||
static const float constantArray[8] = { 0.f, 1.f, 2.f, 0.5f, 3.f, 1.f / 3.f, 0.25f, 1.f / 6.f };
|
||||
static const float constantArrayNegated[8] = { -0.f, -1.f, -2.f, -0.5f, -3.f, -1.f / 3.f, -0.25f, -1.f / 6.f };
|
||||
|
||||
// Applying prefixes in SIMD fashion will actually be a lot easier than the old style.
|
||||
if (prefix == 0xE4) {
|
||||
return fpr.QMapReg(mipsReg, sz, mapFlags);
|
||||
}
|
||||
|
||||
int n = GetNumVectorElements(sz);
|
||||
|
||||
int regnum[4] = { -1, -1, -1, -1 };
|
||||
int abs[4] = { 0 };
|
||||
int negate[4] = { 0 };
|
||||
int constants[4] = { 0 };
|
||||
int constNum[4] = { 0 };
|
||||
|
||||
int full_mask = (1 << n) - 1;
|
||||
|
||||
int abs_mask = (prefix >> 8) & full_mask;
|
||||
int negate_mask = (prefix >> 16) & full_mask;
|
||||
int constants_mask = (prefix >> 12) & full_mask;
|
||||
|
||||
// Decode prefix to keep the rest readable
|
||||
int permuteMask = 0;
|
||||
for (int i = 0; i < n; i++) {
|
||||
permuteMask |= 3 << (i * 2);
|
||||
regnum[i] = (prefix >> (i * 2)) & 3;
|
||||
abs[i] = (prefix >> (8 + i)) & 1;
|
||||
negate[i] = (prefix >> (16 + i)) & 1;
|
||||
constants[i] = (prefix >> (12 + i)) & 1;
|
||||
|
||||
if (constants[i]) {
|
||||
constNum[i] = regnum[i] + (abs[i] << 2);
|
||||
abs[i] = 0;
|
||||
}
|
||||
}
|
||||
abs_mask &= ~constants_mask;
|
||||
|
||||
bool anyPermute = (prefix & permuteMask) != (0xE4 & permuteMask);
|
||||
|
||||
if (constants_mask == full_mask) {
|
||||
// It's all constants! Don't even bother mapping the input register,
|
||||
// just allocate a temp one.
|
||||
// If a single, this can sometimes be done cheaper. But meh.
|
||||
ARMReg ar = fpr.QAllocTemp(sz);
|
||||
for (int i = 0; i < n; i++) {
|
||||
if ((i & 1) == 0) {
|
||||
if (constNum[i] == constNum[i + 1]) {
|
||||
// Replace two loads with a single immediate when easily possible.
|
||||
ARMReg dest = i & 2 ? D_1(ar) : D_0(ar);
|
||||
switch (constNum[i]) {
|
||||
case 0:
|
||||
case 1:
|
||||
{
|
||||
float c = constantArray[constNum[i]];
|
||||
VMOV_immf(dest, negate[i] ? -c : c);
|
||||
}
|
||||
break;
|
||||
// TODO: There are a few more that are doable.
|
||||
default:
|
||||
goto skip;
|
||||
}
|
||||
|
||||
i++;
|
||||
continue;
|
||||
skip:
|
||||
;
|
||||
}
|
||||
}
|
||||
MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]);
|
||||
VLD1_lane(F_32, ar, R0, i, true);
|
||||
}
|
||||
return ar;
|
||||
}
|
||||
|
||||
// 1. Permute.
|
||||
// 2. Abs
|
||||
// If any constants:
|
||||
// 3. Replace values with constants
|
||||
// 4. Negate
|
||||
|
||||
ARMReg inputAR = fpr.QMapReg(mipsReg, sz, mapFlags);
|
||||
ARMReg ar = fpr.QAllocTemp(sz);
|
||||
|
||||
if (!anyPermute) {
|
||||
VMOV(ar, inputAR);
|
||||
// No permutations!
|
||||
} else {
|
||||
bool allSame = false;
|
||||
for (int i = 1; i < n; i++) {
|
||||
if (regnum[0] == regnum[i])
|
||||
allSame = true;
|
||||
}
|
||||
|
||||
if (allSame) {
|
||||
// Easy, someone is duplicating one value onto all the reg parts.
|
||||
// If this is happening and QMapReg must load, we can combine these two actions
|
||||
// into a VLD1_lane. TODO
|
||||
VDUP(F_32, ar, inputAR, regnum[0]);
|
||||
} else {
|
||||
// Do some special cases
|
||||
if (regnum[0] == 1 && regnum[1] == 0) {
|
||||
INFO_LOG(HLE, "PREFIXST: Bottom swap!");
|
||||
VREV64(I_32, ar, inputAR);
|
||||
regnum[0] = 0;
|
||||
regnum[1] = 1;
|
||||
}
|
||||
|
||||
// TODO: Make a generic fallback using another temp register
|
||||
|
||||
bool match = true;
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (regnum[i] != i)
|
||||
match = false;
|
||||
}
|
||||
|
||||
// TODO: Cannot do this permutation yet!
|
||||
if (!match) {
|
||||
ERROR_LOG(HLE, "PREFIXST: Unsupported permute! %i %i %i %i / %i", regnum[0], regnum[1], regnum[2], regnum[3], n);
|
||||
VMOV(ar, inputAR);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ABS
|
||||
// Two methods: If all lanes are "absoluted", it's easy.
|
||||
if (abs_mask == full_mask) {
|
||||
// TODO: elide the above VMOV (in !anyPermute) when possible
|
||||
VABS(F_32, ar, ar);
|
||||
} else if (abs_mask != 0) {
|
||||
// Partial ABS!
|
||||
if (abs_mask == 3) {
|
||||
VABS(F_32, D_0(ar), D_0(ar));
|
||||
} else {
|
||||
// Horrifying fallback: Mov to Q0, abs, move back.
|
||||
// TODO: Optimize for lower quads where we don't need to move.
|
||||
VMOV(MatchSize(Q0, ar), ar);
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (abs_mask & (1 << i)) {
|
||||
VABS((ARMReg)(S0 + i), (ARMReg)(S0 + i));
|
||||
}
|
||||
}
|
||||
VMOV(ar, MatchSize(Q0, ar));
|
||||
INFO_LOG(HLE, "PREFIXST: Partial ABS %i/%i! Slow fallback generated.", abs_mask, full_mask);
|
||||
}
|
||||
}
|
||||
|
||||
if (negate_mask == full_mask) {
|
||||
// TODO: elide the above VMOV when possible
|
||||
VNEG(F_32, ar, ar);
|
||||
} else if (negate_mask != 0) {
|
||||
// Partial negate! I guess we build sign bits in another register
|
||||
// and simply XOR.
|
||||
if (negate_mask == 3) {
|
||||
VNEG(F_32, D_0(ar), D_0(ar));
|
||||
} else {
|
||||
// Horrifying fallback: Mov to Q0, negate, move back.
|
||||
// TODO: Optimize for lower quads where we don't need to move.
|
||||
VMOV(MatchSize(Q0, ar), ar);
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (negate_mask & (1 << i)) {
|
||||
VNEG((ARMReg)(S0 + i), (ARMReg)(S0 + i));
|
||||
}
|
||||
}
|
||||
VMOV(ar, MatchSize(Q0, ar));
|
||||
INFO_LOG(HLE, "PREFIXST: Partial Negate %i/%i! Slow fallback generated.", negate_mask, full_mask);
|
||||
}
|
||||
}
|
||||
|
||||
// Insert constants where requested, and check negate!
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (constants[i]) {
|
||||
MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]);
|
||||
VLD1_lane(F_32, ar, R0, i, true);
|
||||
}
|
||||
}
|
||||
|
||||
return ar;
|
||||
}
|
||||
|
||||
Jit::DestARMReg Jit::NEONMapPrefixD(int vreg, VectorSize sz, int mapFlags) {
|
||||
// Inverted from the actual bits, easier to reason about 1 == write
|
||||
int writeMask = (~(js.prefixD >> 8)) & 0xF;
|
||||
int n = GetNumVectorElements(sz);
|
||||
int full_mask = (1 << n) - 1;
|
||||
|
||||
DestARMReg dest;
|
||||
dest.sz = sz;
|
||||
if ((writeMask & full_mask) == full_mask) {
|
||||
// No need to apply a write mask.
|
||||
// Let's not make things complicated.
|
||||
dest.rd = fpr.QMapReg(vreg, sz, mapFlags);
|
||||
dest.backingRd = dest.rd;
|
||||
} else {
|
||||
// Allocate a temporary register.
|
||||
ELOG("PREFIXD: Write mask allocated! %i/%i", writeMask, full_mask);
|
||||
dest.rd = fpr.QAllocTemp(sz);
|
||||
dest.backingRd = fpr.QMapReg(vreg, sz, mapFlags & ~MAP_NOINIT); // Force initialization of the backing reg.
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
void Jit::NEONApplyPrefixD(DestARMReg dest) {
|
||||
// Apply clamps to dest.rd
|
||||
int n = GetNumVectorElements(dest.sz);
|
||||
|
||||
int sat1_mask = 0;
|
||||
int sat3_mask = 0;
|
||||
int full_mask = (1 << n) - 1;
|
||||
for (int i = 0; i < n; i++) {
|
||||
int sat = (js.prefixD >> (i * 2)) & 3;
|
||||
if (sat == 1)
|
||||
sat1_mask |= 1 << i;
|
||||
if (sat == 3)
|
||||
sat3_mask |= 1 << i;
|
||||
}
|
||||
|
||||
if (sat1_mask && sat3_mask) {
|
||||
// Why would anyone do this?
|
||||
ELOG("PREFIXD: Can't have both sat[0-1] and sat[-1-1] at the same time yet");
|
||||
}
|
||||
|
||||
if (sat1_mask) {
|
||||
if (sat1_mask != full_mask) {
|
||||
ELOG("PREFIXD: Can't have partial sat1 mask yet (%i vs %i)", sat1_mask, full_mask);
|
||||
}
|
||||
if (IsD(dest.rd)) {
|
||||
VMOV_immf(D0, 0.0);
|
||||
VMOV_immf(D1, 1.0);
|
||||
VMAX(F_32, dest.rd, dest.rd, D0);
|
||||
VMIN(F_32, dest.rd, dest.rd, D1);
|
||||
} else {
|
||||
VMOV_immf(Q0, 1.0);
|
||||
VMIN(F_32, dest.rd, dest.rd, Q0);
|
||||
VMOV_immf(Q0, 0.0);
|
||||
VMAX(F_32, dest.rd, dest.rd, Q0);
|
||||
}
|
||||
}
|
||||
|
||||
if (sat3_mask && sat1_mask != full_mask) {
|
||||
if (sat3_mask != full_mask) {
|
||||
ELOG("PREFIXD: Can't have partial sat3 mask yet (%i vs %i)", sat3_mask, full_mask);
|
||||
}
|
||||
if (IsD(dest.rd)) {
|
||||
VMOV_immf(D0, 0.0);
|
||||
VMOV_immf(D1, 1.0);
|
||||
VMAX(F_32, dest.rd, dest.rd, D0);
|
||||
VMIN(F_32, dest.rd, dest.rd, D1);
|
||||
} else {
|
||||
VMOV_immf(Q0, 1.0);
|
||||
VMIN(F_32, dest.rd, dest.rd, Q0);
|
||||
VMOV_immf(Q0, -1.0);
|
||||
VMAX(F_32, dest.rd, dest.rd, Q0);
|
||||
}
|
||||
}
|
||||
|
||||
// Check for actual mask operation (unrelated to the "masks" above).
|
||||
if (dest.backingRd != dest.rd) {
|
||||
// This means that we need to apply the write mask, from rd to backingRd.
|
||||
// What a pain. We can at least shortcut easy cases like half the register.
|
||||
// And we can generate the masks easily with some of the crazy vector imm modes. (bits2bytes for example).
|
||||
// So no need to load them from RAM.
|
||||
int writeMask = (~(js.prefixD >> 8)) & 0xF;
|
||||
|
||||
if (writeMask == 3) {
|
||||
ILOG("Doing writemask = 3");
|
||||
VMOV(D_0(dest.rd), D_0(dest.backingRd));
|
||||
} else {
|
||||
// TODO
|
||||
ELOG("PREFIXD: Arbitrary write masks not supported (%i / %i)", writeMask, full_mask);
|
||||
VMOV(dest.backingRd, dest.rd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Jit::MappedRegs Jit::NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes) {
|
||||
MappedRegs regs;
|
||||
if (applyPrefixes) {
|
||||
regs.vs = NEONMapPrefixS(_VS, ssize, 0);
|
||||
regs.vt = NEONMapPrefixT(_VT, tsize, 0);
|
||||
} else {
|
||||
regs.vs = fpr.QMapReg(_VS, ssize, 0);
|
||||
regs.vt = fpr.QMapReg(_VT, ssize, 0);
|
||||
}
|
||||
|
||||
regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0 || GetVectorOverlap(_VD, dsize, _VT, ssize);
|
||||
if (applyPrefixes) {
|
||||
regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
|
||||
} else {
|
||||
regs.vd.rd = fpr.QMapReg(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
|
||||
regs.vd.backingRd = regs.vd.rd;
|
||||
regs.vd.sz = dsize;
|
||||
}
|
||||
return regs;
|
||||
}
|
||||
|
||||
Jit::MappedRegs Jit::NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes) {
|
||||
MappedRegs regs;
|
||||
if (applyPrefixes) {
|
||||
regs.vs = NEONMapPrefixS(_VS, ssize, 0);
|
||||
regs.vt = NEONMapPrefixT(_VT, tsize, 0);
|
||||
} else {
|
||||
regs.vs = fpr.QMapReg(_VS, ssize, 0);
|
||||
regs.vt = fpr.QMapReg(_VT, ssize, 0);
|
||||
}
|
||||
regs.vd.rd = INVALID_REG;
|
||||
regs.vd.sz = V_Invalid;
|
||||
return regs;
|
||||
}
|
||||
|
||||
Jit::MappedRegs Jit::NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes) {
|
||||
MappedRegs regs;
|
||||
regs.vs = NEONMapPrefixS(_VS, ssize, 0);
|
||||
regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0;
|
||||
regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
|
||||
return regs;
|
||||
}
|
||||
|
||||
// Requires quad registers.
|
||||
void Jit::NEONTranspose4x4(ARMReg cols[4]) {
|
||||
// 0123 _\ 0426
|
||||
// 4567 / 1537
|
||||
VTRN(F_32, cols[0], cols[1]);
|
||||
|
||||
// 89ab _\ 8cae
|
||||
// cdef / 9dbf
|
||||
VTRN(F_32, cols[2], cols[3]);
|
||||
|
||||
// 04[26] 048c
|
||||
// 15 37 -> 1537
|
||||
// [8c]ae 26ae
|
||||
// 9d bf 9dbf
|
||||
VSWP(D_1(cols[0]), D_0(cols[2]));
|
||||
|
||||
// 04 8c 048c
|
||||
// 15[37] -> 159d
|
||||
// 26 ae 26ae
|
||||
// [9d]bf 37bf
|
||||
VSWP(D_1(cols[1]), D_0(cols[3]));
|
||||
}
|
||||
|
||||
} // namespace MIPSComp
|
||||
@@ -0,0 +1,20 @@
|
||||
#pragma once
|
||||
|
||||
#include "Core/MIPS/ARM/ArmJit.h"
|
||||
#include "Core/MIPS/ARM/ArmRegCache.h"
|
||||
|
||||
namespace MIPSComp {
|
||||
|
||||
|
||||
inline ARMReg MatchSize(ARMReg x, ARMReg target) {
|
||||
if (IsQ(target) && IsQ(x))
|
||||
return x;
|
||||
if (IsD(target) && IsD(x))
|
||||
return x;
|
||||
if (IsD(target) && IsQ(x))
|
||||
return D_0(x);
|
||||
// if (IsQ(target) && IsD(x))
|
||||
return (ARMReg)(D0 + (x - Q0) * 2);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() {
|
||||
useNEONVFPU = false;
|
||||
}
|
||||
|
||||
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips)
|
||||
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips)
|
||||
{
|
||||
logBlocks = 0;
|
||||
dontLogBlocks = 0;
|
||||
@@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b)
|
||||
while (js.compiling)
|
||||
{
|
||||
gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages
|
||||
fpr.SetCompilerPC(js.compilerPC);
|
||||
MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC);
|
||||
js.downcountAmount += MIPSGetInstructionCycleEstimate(inst);
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "Common/CPUDetect.h"
|
||||
#include "Core/MIPS/JitCommon/JitState.h"
|
||||
#include "Core/MIPS/JitCommon/JitBlockCache.h"
|
||||
#include "Core/MIPS/ARM/ArmRegCache.h"
|
||||
@@ -240,6 +241,43 @@ private:
|
||||
}
|
||||
void GetVectorRegsPrefixD(u8 *regs, VectorSize sz, int vectorReg);
|
||||
|
||||
|
||||
// For NEON mappings, it will be easier to deal directly in ARM registers.
|
||||
|
||||
ARMReg NEONMapPrefixST(int vfpuReg, VectorSize sz, u32 prefix, int mapFlags);
|
||||
ARMReg NEONMapPrefixS(int vfpuReg, VectorSize sz, int mapFlags) {
|
||||
return NEONMapPrefixST(vfpuReg, sz, js.prefixS, mapFlags);
|
||||
}
|
||||
ARMReg NEONMapPrefixT(int vfpuReg, VectorSize sz, int mapFlags) {
|
||||
return NEONMapPrefixST(vfpuReg, sz, js.prefixT, mapFlags);
|
||||
}
|
||||
|
||||
struct DestARMReg {
|
||||
ARMReg rd;
|
||||
ARMReg backingRd;
|
||||
VectorSize sz;
|
||||
|
||||
operator ARMReg() const { return rd; }
|
||||
};
|
||||
|
||||
struct MappedRegs {
|
||||
ARMReg vs;
|
||||
ARMReg vt;
|
||||
DestARMReg vd;
|
||||
bool overlap;
|
||||
};
|
||||
|
||||
MappedRegs NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true);
|
||||
MappedRegs NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true);
|
||||
MappedRegs NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes = true);
|
||||
|
||||
DestARMReg NEONMapPrefixD(int vfpuReg, VectorSize sz, int mapFlags);
|
||||
void NEONApplyPrefixD(DestARMReg dest);
|
||||
|
||||
// NEON utils
|
||||
void NEONMaskToSize(ARMReg vs, VectorSize sz);
|
||||
void NEONTranspose4x4(ARMReg cols[4]);
|
||||
|
||||
// Utils
|
||||
void SetR0ToEffectiveAddress(MIPSGPReg rs, s16 offset);
|
||||
void SetCCAndR0ForSafeAddress(MIPSGPReg rs, s16 offset, ARMReg tempReg, bool reverse = false);
|
||||
|
||||
@@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) {
|
||||
return mr[mipsReg].loc == ML_ARMREG_AS_PTR;
|
||||
}
|
||||
|
||||
bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) {
|
||||
return mr[mipsReg].loc == ML_ARMREG;
|
||||
}
|
||||
|
||||
void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) {
|
||||
// If we can do it with a simple Operand2, let's do that.
|
||||
Operand2 op2;
|
||||
|
||||
@@ -103,6 +103,7 @@ public:
|
||||
ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0);
|
||||
ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty.
|
||||
|
||||
bool IsMapped(MIPSGPReg reg);
|
||||
bool IsMappedAsPointer(MIPSGPReg reg);
|
||||
|
||||
void MapInIn(MIPSGPReg rd, MIPSGPReg rs);
|
||||
|
||||
@@ -16,14 +16,17 @@
|
||||
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
|
||||
|
||||
#include <cstring>
|
||||
|
||||
#include "base/logging.h"
|
||||
#include "Common/CPUDetect.h"
|
||||
#include "Core/MIPS/MIPS.h"
|
||||
#include "Core/MIPS/ARM/ArmRegCacheFPU.h"
|
||||
#include "Core/MIPS/ARM/ArmJit.h"
|
||||
#include "Core/MIPS/MIPSTables.h"
|
||||
|
||||
using namespace ArmGen;
|
||||
|
||||
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) {
|
||||
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) {
|
||||
if (cpu_info.bNEON) {
|
||||
numARMFpuReg_ = 32;
|
||||
} else {
|
||||
@@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init
|
||||
}
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::Init(ARMXEmitter *emitter) {
|
||||
emit_ = emitter;
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) {
|
||||
if (!initialReady) {
|
||||
SetupInitialRegs();
|
||||
@@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() {
|
||||
mrInitial[i].spillLock = false;
|
||||
mrInitial[i].tempLock = false;
|
||||
}
|
||||
for (int i = 0; i < MAX_ARMQUADS; i++) {
|
||||
qr[i].isDirty = false;
|
||||
qr[i].mipsVec = -1;
|
||||
qr[i].sz = V_Invalid;
|
||||
qr[i].spillLock = false;
|
||||
qr[i].isTemp = false;
|
||||
memset(qr[i].vregs, 0xff, 4);
|
||||
}
|
||||
}
|
||||
|
||||
static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) {
|
||||
// We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things.
|
||||
static const ARMReg allocationOrder[] = {
|
||||
S2, S3,
|
||||
@@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
S12, S13, S14, S15
|
||||
};
|
||||
|
||||
// With NEON, we have many more.
|
||||
// In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU.
|
||||
// VFPU will use NEON to do SIMD and it will be awkward to mix with FPU.
|
||||
// VFP mapping
|
||||
// VFPU registers and regular FP registers are mapped interchangably on top of the standard
|
||||
// 16 FPU registers.
|
||||
|
||||
// NEON mapping
|
||||
// We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers.
|
||||
// VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON
|
||||
// (or D16-D31 as doubles, but not relevant).
|
||||
// Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads.
|
||||
|
||||
// We should attempt to map scalars to low Q registers and wider things to high registers,
|
||||
// as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be
|
||||
// able to use regular VFP instructions too.
|
||||
@@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
S4, S5, S6, S7, // Q1
|
||||
S8, S9, S10, S11, // Q2
|
||||
S12, S13, S14, S15, // Q3
|
||||
S16, S17, S18, S19, // Q4
|
||||
S20, S21, S22, S23, // Q5
|
||||
S24, S25, S26, S27, // Q6
|
||||
S28, S29, S30, S31, // Q7
|
||||
// Q8-Q15 free for NEON tricks
|
||||
// Q4-Q15 free for VFPU
|
||||
};
|
||||
|
||||
if (cpu_info.bNEON) {
|
||||
if (jo_->useNEONVFPU) {
|
||||
count = sizeof(allocationOrderNEON) / sizeof(const int);
|
||||
return allocationOrderNEON;
|
||||
} else {
|
||||
@@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
}
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::IsMapped(MIPSReg r) {
|
||||
return mr[r].loc == ML_ARMREG;
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) {
|
||||
// INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags);
|
||||
if (jo_->useNEONVFPU && mipsReg >= 32) {
|
||||
ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC);
|
||||
return S0;
|
||||
}
|
||||
|
||||
pendingFlush = true;
|
||||
// Let's see if it's already mapped. If so we just need to update the dirty flag.
|
||||
// We don't need to check for ML_NOINIT because we assume that anyone who maps
|
||||
@@ -157,7 +176,7 @@ allocate:
|
||||
}
|
||||
|
||||
// Uh oh, we have all them spilllocked....
|
||||
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc);
|
||||
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC);
|
||||
return INVALID_REG;
|
||||
}
|
||||
|
||||
@@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::FlushArmReg(ARMReg r) {
|
||||
int reg = r - S0;
|
||||
if (ar[reg].mipsReg == -1) {
|
||||
// Nothing to do, reg not mapped.
|
||||
return;
|
||||
}
|
||||
if (ar[reg].mipsReg != -1) {
|
||||
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) {
|
||||
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
|
||||
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
|
||||
if (r >= S0 && r <= S31) {
|
||||
int reg = r - S0;
|
||||
if (ar[reg].mipsReg == -1) {
|
||||
// Nothing to do, reg not mapped.
|
||||
return;
|
||||
}
|
||||
// IMMs won't be in an ARM reg.
|
||||
mr[ar[reg].mipsReg].loc = ML_MEM;
|
||||
mr[ar[reg].mipsReg].reg = INVALID_REG;
|
||||
} else {
|
||||
ERROR_LOG(JIT, "Dirty but no mipsreg?");
|
||||
if (ar[reg].mipsReg != -1) {
|
||||
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG)
|
||||
{
|
||||
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
|
||||
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
|
||||
}
|
||||
// IMMs won't be in an ARM reg.
|
||||
mr[ar[reg].mipsReg].loc = ML_MEM;
|
||||
mr[ar[reg].mipsReg].reg = INVALID_REG;
|
||||
} else {
|
||||
ERROR_LOG(JIT, "Dirty but no mipsreg?");
|
||||
}
|
||||
ar[reg].isDirty = false;
|
||||
ar[reg].mipsReg = -1;
|
||||
} else if (r >= D0 && r <= D31) {
|
||||
// TODO: Convert to S regs and flush them individually.
|
||||
} else if (r >= Q0 && r <= Q15) {
|
||||
int quad = r - Q0;
|
||||
QFlush(r);
|
||||
}
|
||||
ar[reg].isDirty = false;
|
||||
ar[reg].mipsReg = -1;
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::FlushV(MIPSReg r) {
|
||||
FlushR(r + 32);
|
||||
}
|
||||
|
||||
/*
|
||||
void ArmRegCacheFPU::FlushQWithV(MIPSReg r) {
|
||||
// Look for it in all the quads. If it's in any, flush that quad clean.
|
||||
int flushCount = 0;
|
||||
for (int i = 0; i < MAX_ARMQUADS; i++) {
|
||||
if (qr[i].sz == V_Invalid)
|
||||
continue;
|
||||
|
||||
int n = qr[i].sz;
|
||||
bool flushThis = false;
|
||||
for (int j = 0; j < n; j++) {
|
||||
if (qr[i].vregs[j] == r) {
|
||||
flushThis = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (flushThis) {
|
||||
QFlush(i);
|
||||
flushCount++;
|
||||
}
|
||||
}
|
||||
|
||||
if (flushCount > 1) {
|
||||
WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r);
|
||||
}
|
||||
}
|
||||
*/
|
||||
|
||||
void ArmRegCacheFPU::FlushR(MIPSReg r) {
|
||||
switch (mr[r].loc) {
|
||||
case ML_IMM:
|
||||
@@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) {
|
||||
if (mr[r].reg == (int)INVALID_REG) {
|
||||
ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg");
|
||||
}
|
||||
if (ar[mr[r].reg].isDirty) {
|
||||
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
|
||||
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
|
||||
ar[mr[r].reg].isDirty = false;
|
||||
|
||||
if (mr[r].reg >= Q0 && mr[r].reg <= Q15) {
|
||||
// This should happen rarely, but occasionally we need to flush a single stray
|
||||
// mipsreg that's been part of a quad.
|
||||
int quad = mr[r].reg - Q0;
|
||||
if (qr[quad].isDirty) {
|
||||
WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1);
|
||||
emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true);
|
||||
}
|
||||
} else {
|
||||
if (ar[mr[r].reg].isDirty) {
|
||||
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
|
||||
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
|
||||
ar[mr[r].reg].isDirty = false;
|
||||
}
|
||||
ar[mr[r].reg].mipsReg = -1;
|
||||
}
|
||||
ar[mr[r].reg].mipsReg = -1;
|
||||
break;
|
||||
|
||||
case ML_MEM:
|
||||
@@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() {
|
||||
return 16;
|
||||
}
|
||||
|
||||
// Scalar only. Need a similar one for sequential Q vectors.
|
||||
int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) {
|
||||
int c = 1;
|
||||
int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg);
|
||||
@@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() {
|
||||
DiscardR(i);
|
||||
}
|
||||
|
||||
// Flush quads!
|
||||
// These could also use sequential detection.
|
||||
for (int i = 4; i < MAX_ARMQUADS; i++) {
|
||||
QFlush(i);
|
||||
}
|
||||
|
||||
// Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are
|
||||
// sequential. This is necessary because we store VFPU registers in a staggered order to get
|
||||
// columns sequential (most VFPU math in nearly all games is in columns, not rows).
|
||||
@@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const {
|
||||
}
|
||||
|
||||
int ArmRegCacheFPU::GetTempR() {
|
||||
if (jo_->useNEONVFPU) {
|
||||
ERROR_LOG(JIT, "VFP temps not allowed in NEON mode");
|
||||
return 0;
|
||||
}
|
||||
pendingFlush = true;
|
||||
for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) {
|
||||
if (mr[r].loc == ML_MEM && !mr[r].tempLock) {
|
||||
@@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) {
|
||||
|
||||
// This is actually pretty slow with all the 160 regs...
|
||||
void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() {
|
||||
for (int i = 0; i < NUM_MIPSFPUREG; i++)
|
||||
for (int i = 0; i < NUM_MIPSFPUREG; i++) {
|
||||
mr[i].spillLock = false;
|
||||
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i)
|
||||
}
|
||||
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) {
|
||||
DiscardR(i);
|
||||
}
|
||||
for (int i = 0; i < MAX_ARMQUADS; i++) {
|
||||
qr[i].spillLock = false;
|
||||
if (qr[i].isTemp) {
|
||||
qr[i].isTemp = false;
|
||||
qr[i].sz = V_Invalid;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::R(int mipsReg) {
|
||||
@@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) {
|
||||
return (ARMReg)(mr[mipsReg].reg + S0);
|
||||
} else {
|
||||
if (mipsReg < 32) {
|
||||
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_));
|
||||
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
|
||||
} else if (mipsReg < 32 + 128) {
|
||||
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
|
||||
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
|
||||
} else {
|
||||
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
|
||||
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
|
||||
}
|
||||
return INVALID_REG; // BAAAD
|
||||
}
|
||||
}
|
||||
|
||||
inline ARMReg QuadAsD(int quad) {
|
||||
return (ARMReg)(D0 + quad * 2);
|
||||
}
|
||||
|
||||
inline ARMReg QuadAsQ(int quad) {
|
||||
return (ARMReg)(Q0 + quad);
|
||||
}
|
||||
|
||||
bool MappableQ(int quad) {
|
||||
return quad >= 4;
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) {
|
||||
ERROR_LOG(JIT, "QLoad4x4 not implemented");
|
||||
// TODO
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::QFlush(int quad) {
|
||||
if (!MappableQ(quad)) {
|
||||
ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad);
|
||||
return;
|
||||
}
|
||||
|
||||
if (qr[quad].isDirty && !qr[quad].isTemp) {
|
||||
INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz));
|
||||
|
||||
ARMReg q = QuadAsQ(quad);
|
||||
// Unlike reads, when writing to the register file we need to be careful to write the correct
|
||||
// number of floats.
|
||||
|
||||
switch (qr[quad].sz) {
|
||||
case V_Single:
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
// WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
break;
|
||||
case V_Pair:
|
||||
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) {
|
||||
// Can combine, it's a column!
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
// WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 1, true);
|
||||
}
|
||||
break;
|
||||
case V_Triple:
|
||||
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
|
||||
emit_->VST1_lane(F_32, q, R0, 2, true);
|
||||
} else {
|
||||
// WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 2, true);
|
||||
}
|
||||
break;
|
||||
case V_Quad:
|
||||
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
// WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 2, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 3, true);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz);
|
||||
break;
|
||||
}
|
||||
|
||||
qr[quad].isDirty = false;
|
||||
|
||||
int n = GetNumVectorElements(qr[quad].sz);
|
||||
for (int i = 0; i < n; i++) {
|
||||
int vr = qr[quad].vregs[i];
|
||||
if (vr < 0 || vr > 128) {
|
||||
ERROR_LOG(JIT, "Bad vr %i", vr);
|
||||
}
|
||||
FPURegMIPS &m = mr[32 + vr];
|
||||
m.loc = ML_MEM;
|
||||
m.lane = -1;
|
||||
m.reg = -1;
|
||||
}
|
||||
|
||||
} else {
|
||||
if (qr[quad].isTemp) {
|
||||
WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp);
|
||||
}
|
||||
}
|
||||
|
||||
qr[quad].isTemp = false;
|
||||
qr[quad].mipsVec = -1;
|
||||
qr[quad].sz = V_Invalid;
|
||||
memset(qr[quad].vregs, 0xFF, 4);
|
||||
}
|
||||
|
||||
int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) {
|
||||
// Search for a free quad. A quad is free if the first register in it is free.
|
||||
int quad = -1;
|
||||
|
||||
for (int i = 0; i < count; i++) {
|
||||
int q = (i + start) & 15;
|
||||
|
||||
if (!MappableQ(q))
|
||||
continue;
|
||||
|
||||
// Don't steal temp quads!
|
||||
if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) {
|
||||
// INFO_LOG(JIT, "Free quad: %i", q);
|
||||
// Oh yeah! Free quad!
|
||||
return q;
|
||||
}
|
||||
}
|
||||
|
||||
// Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some
|
||||
// sort of age.
|
||||
int bestQuad = -1;
|
||||
int bestScore = -1;
|
||||
for (int i = 0; i < count; i++) {
|
||||
int q = (i + start) & 15;
|
||||
|
||||
if (!MappableQ(q))
|
||||
continue;
|
||||
if (qr[q].spillLock)
|
||||
continue;
|
||||
if (qr[q].isTemp)
|
||||
continue;
|
||||
|
||||
int score = 0;
|
||||
if (!qr[q].isDirty) {
|
||||
score += 5;
|
||||
}
|
||||
|
||||
if (score > bestScore) {
|
||||
bestQuad = q;
|
||||
bestScore = score;
|
||||
}
|
||||
}
|
||||
|
||||
if (bestQuad == -1) {
|
||||
ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!");
|
||||
return -1;
|
||||
} else {
|
||||
INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason");
|
||||
QFlush(bestQuad);
|
||||
return bestQuad;
|
||||
}
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) {
|
||||
int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps
|
||||
if (q < 0) {
|
||||
ERROR_LOG(JIT, "Failed to allocate temp quad");
|
||||
q = 0;
|
||||
}
|
||||
qr[q].spillLock = true;
|
||||
qr[q].isTemp = true;
|
||||
qr[q].sz = sz;
|
||||
qr[q].isDirty = false; // doesn't matter
|
||||
|
||||
INFO_LOG(JIT, "Allocated temp quad %i", q);
|
||||
|
||||
if (sz == V_Single || sz == V_Pair) {
|
||||
return D_0(ARMReg(Q0 + q));
|
||||
} else {
|
||||
return ARMReg(Q0 + q);
|
||||
}
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::Consecutive(int v1, int v2) const {
|
||||
return (voffset[v1] + 1) == voffset[v2];
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const {
|
||||
return Consecutive(v1, v2) && Consecutive(v2, v3);
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const {
|
||||
return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4);
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) {
|
||||
u8 vregs[4];
|
||||
if (flags & MAP_MTX_TRANSPOSED) {
|
||||
GetMatrixRows(matrix, mz, vregs);
|
||||
} else {
|
||||
GetMatrixColumns(matrix, mz, vregs);
|
||||
}
|
||||
|
||||
// TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load.
|
||||
int n = GetMatrixSide(mz);
|
||||
VectorSize vsz = GetVectorSize(mz);
|
||||
for (int i = 0; i < n; i++) {
|
||||
regs[i] = QMapReg(vregs[i], vsz, flags);
|
||||
}
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) {
|
||||
qTime_++;
|
||||
|
||||
int n = GetNumVectorElements(sz);
|
||||
u8 vregs[4];
|
||||
GetVectorRegs(vregs, sz, vreg);
|
||||
|
||||
// Range of registers to consider
|
||||
int start = 0;
|
||||
int count = 16;
|
||||
|
||||
if (flags & MAP_PREFER_HIGH) {
|
||||
start = 8;
|
||||
} else if (flags & MAP_PREFER_LOW) {
|
||||
start = 4;
|
||||
} else if (flags & MAP_FORCE_LOW) {
|
||||
start = 4;
|
||||
count = 4;
|
||||
} else if (flags & MAP_FORCE_HIGH) {
|
||||
start = 8;
|
||||
count = 8;
|
||||
}
|
||||
|
||||
// Let's check if they are all mapped in a quad somewhere.
|
||||
// At the same time, check for the quad already being mapped.
|
||||
// Later we can check for possible transposes as well.
|
||||
|
||||
// First just loop over all registers. If it's here and not in range, or overlapped, kick.
|
||||
std::vector<int> quadsToFlush;
|
||||
for (int i = 0; i < 16; i++) {
|
||||
int q = (i + start) & 15;
|
||||
if (!MappableQ(q))
|
||||
continue;
|
||||
|
||||
// Skip unmapped quads.
|
||||
if (qr[q].sz == V_Invalid)
|
||||
continue;
|
||||
|
||||
// Check if completely there already. If so, set spill-lock, transfer dirty flag and exit.
|
||||
if (vreg == qr[q].mipsVec && sz == qr[q].sz) {
|
||||
if (i < count) {
|
||||
INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz);
|
||||
qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY);
|
||||
qr[q].spillLock = true;
|
||||
|
||||
// Sanity check vregs
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (vregs[i] != qr[q].vregs[i]) {
|
||||
ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]);
|
||||
}
|
||||
}
|
||||
|
||||
return (ARMReg)(Q0 + q);
|
||||
} else {
|
||||
INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count);
|
||||
quadsToFlush.push_back(q);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for any overlap. Overlap == flush.
|
||||
int origN = GetNumVectorElements(qr[q].sz);
|
||||
for (int a = 0; a < n; a++) {
|
||||
for (int b = 0; b < origN; b++) {
|
||||
if (vregs[a] == qr[q].vregs[b]) {
|
||||
quadsToFlush.push_back(q);
|
||||
goto doubleBreak;
|
||||
}
|
||||
}
|
||||
}
|
||||
doubleBreak:
|
||||
;
|
||||
}
|
||||
|
||||
// We didn't find the extra register, but we got a list of regs to flush. Flush 'em.
|
||||
// Here we can check for opportunities to do a "transpose-flush" of row vectors, etc.
|
||||
if (!quadsToFlush.empty()) {
|
||||
ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size());
|
||||
}
|
||||
for (size_t i = 0; i < quadsToFlush.size(); i++) {
|
||||
QFlush(quadsToFlush[i]);
|
||||
}
|
||||
|
||||
// Find where we want to map it, obeying the constraints we gave.
|
||||
int quad = QGetFreeQuad(start, count, "mapping");
|
||||
|
||||
// If parts of our register are elsewhere, and we are dirty, we need to flush them
|
||||
// before we reload in a new location.
|
||||
// This may be problematic if inputs overlap irregularly with output, say:
|
||||
// vdot S700, R000, C000
|
||||
// It might still work by accident...
|
||||
if (flags & MAP_DIRTY) {
|
||||
for (int i = 0; i < n; i++) {
|
||||
FlushV(vregs[i]);
|
||||
}
|
||||
}
|
||||
|
||||
qr[quad].sz = sz;
|
||||
qr[quad].mipsVec = vreg;
|
||||
|
||||
if (!(flags & MAP_NOINIT)) {
|
||||
// Okay, now we will try to load the whole thing in one go. This is possible
|
||||
// if it's a row and easy if it's a single.
|
||||
// Rows are rare, columns are common - but thanks to our register reordering,
|
||||
// columns are actually in-order in memory.
|
||||
switch (sz) {
|
||||
case V_Single:
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
break;
|
||||
case V_Pair:
|
||||
if (Consecutive(vregs[0], vregs[1])) {
|
||||
// Can combine, it's a column!
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
|
||||
}
|
||||
break;
|
||||
case V_Triple:
|
||||
if (Consecutive(vregs[0], vregs[1], vregs[2])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
|
||||
} else {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
|
||||
}
|
||||
break;
|
||||
case V_Quad:
|
||||
if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
;
|
||||
}
|
||||
}
|
||||
|
||||
// OK, let's fill out the arrays to confirm that we have grabbed these registers.
|
||||
for (int i = 0; i < n; i++) {
|
||||
int mipsReg = 32 + vregs[i];
|
||||
mr[mipsReg].loc = ML_ARMREG;
|
||||
mr[mipsReg].reg = QuadAsQ(quad);
|
||||
mr[mipsReg].lane = i;
|
||||
qr[quad].vregs[i] = vregs[i];
|
||||
}
|
||||
qr[quad].isDirty = (flags & MAP_DIRTY) != 0;
|
||||
qr[quad].spillLock = true;
|
||||
|
||||
INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty);
|
||||
if (sz == V_Single || sz == V_Pair) {
|
||||
return D_0(QuadAsQ(quad));
|
||||
} else {
|
||||
return QuadAsQ(quad);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -33,28 +33,56 @@ enum {
|
||||
TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS,
|
||||
};
|
||||
|
||||
enum {
|
||||
MAP_READ = 0,
|
||||
MAP_MTX_TRANSPOSED = 16,
|
||||
MAP_PREFER_LOW = 16,
|
||||
MAP_PREFER_HIGH = 32,
|
||||
|
||||
// Force is not yet correctly implemented, if the reg is already mapped it will not move
|
||||
MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15)
|
||||
MAP_FORCE_HIGH = 128, // Only map Q8-Q15
|
||||
};
|
||||
|
||||
struct FPURegARM {
|
||||
int mipsReg; // if -1, no mipsreg attached.
|
||||
bool isDirty; // Should the register be written back?
|
||||
};
|
||||
|
||||
struct FPURegQuad {
|
||||
int mipsVec;
|
||||
VectorSize sz;
|
||||
u8 vregs[4];
|
||||
bool isDirty;
|
||||
bool spillLock;
|
||||
bool isTemp;
|
||||
};
|
||||
|
||||
struct FPURegMIPS {
|
||||
// Where is this MIPS register?
|
||||
RegMIPSLoc loc;
|
||||
// Data (only one of these is used, depending on loc. Could make a union).
|
||||
int reg;
|
||||
int lane;
|
||||
|
||||
bool spillLock; // if true, this register cannot be spilled.
|
||||
bool tempLock;
|
||||
// If loc == ML_MEM, it's back in its location in the CPU context struct.
|
||||
};
|
||||
|
||||
namespace MIPSComp {
|
||||
struct ArmJitOptions;
|
||||
struct JitState;
|
||||
}
|
||||
|
||||
class ArmRegCacheFPU
|
||||
{
|
||||
public:
|
||||
ArmRegCacheFPU(MIPSState *mips);
|
||||
ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo);
|
||||
~ArmRegCacheFPU() {}
|
||||
|
||||
void Init(ARMXEmitter *emitter);
|
||||
|
||||
void Start(MIPSAnalyst::AnalysisResults &stats);
|
||||
|
||||
// Protect the arm register containing a MIPS register from spilling, to ensure that
|
||||
@@ -80,9 +108,11 @@ public:
|
||||
void MapDirty(MIPSReg rd);
|
||||
void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true);
|
||||
void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true);
|
||||
bool IsMapped(MIPSReg r);
|
||||
void FlushArmReg(ARMReg r);
|
||||
void FlushR(MIPSReg r);
|
||||
void DiscardR(MIPSReg r);
|
||||
ARMReg R(int preg); // Returns a cached register
|
||||
|
||||
// VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode!
|
||||
void MapRegV(int vreg, int flags = 0);
|
||||
@@ -90,18 +120,41 @@ public:
|
||||
void MapInInV(int rt, int rs);
|
||||
void MapDirtyInV(int rd, int rs, bool avoidLoad = true);
|
||||
void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true);
|
||||
bool IsTempX(ARMReg r) const;
|
||||
|
||||
MIPSReg GetTempR();
|
||||
bool IsTempX(ARMReg r) const;
|
||||
MIPSReg GetTempV() { return GetTempR() - 32; }
|
||||
// VFPU registers as single VFP registers.
|
||||
ARMReg V(int vreg) { return R(vreg + 32); }
|
||||
|
||||
int FlushGetSequential(int a, int maxArmReg);
|
||||
void FlushAll();
|
||||
|
||||
ARMReg R(int preg); // Returns a cached register
|
||||
// This one is allowed at any point.
|
||||
void FlushV(MIPSReg r);
|
||||
|
||||
// VFPU registers mapped to match NEON quads (and doubles, for pairs and singles)
|
||||
// Here we return the ARM register directly instead of providing a "V" accessor
|
||||
// and so on. Might switch to this model for the other regallocs later.
|
||||
|
||||
// Quad mapping does NOT look into the ar array. Instead we use the qr array to keep
|
||||
// track of what's in each quad.
|
||||
|
||||
// Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types.
|
||||
// Need to explicitly allow spilling to get spilling.
|
||||
ARMReg QMapReg(int vreg, VectorSize sz, int flags);
|
||||
|
||||
// TODO
|
||||
// Maps a matrix as a set of columns (yes, even transposed ones, always columns
|
||||
// as those are faster to load/flush). When possible it will map into consecutive
|
||||
// quad registers, enabling blazing-fast full-matrix loads, transposed or not.
|
||||
void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags);
|
||||
|
||||
ARMReg QAllocTemp(VectorSize sz);
|
||||
|
||||
// VFPU registers as single VFP registers
|
||||
ARMReg V(int vreg) { return R(vreg + 32); }
|
||||
void QAllowSpill(int quad);
|
||||
void QFlush(int quad);
|
||||
void QLoad4x4(MIPSGPReg regPtr, int vquads[4]);
|
||||
//void FlushQWithV(MIPSReg r);
|
||||
|
||||
// NOTE: These require you to release spill locks manually!
|
||||
void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags);
|
||||
@@ -112,31 +165,43 @@ public:
|
||||
|
||||
void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; }
|
||||
|
||||
// For better log output only.
|
||||
void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; }
|
||||
|
||||
int GetMipsRegOffset(MIPSReg r);
|
||||
|
||||
private:
|
||||
bool Consecutive(int v1, int v2) const;
|
||||
bool Consecutive(int v1, int v2, int v3) const;
|
||||
bool Consecutive(int v1, int v2, int v3, int v4) const;
|
||||
|
||||
MIPSReg GetTempR();
|
||||
const ARMReg *GetMIPSAllocationOrder(int &count);
|
||||
int GetMipsRegOffsetV(MIPSReg r) {
|
||||
return GetMipsRegOffset(r + 32);
|
||||
}
|
||||
// This one WILL get a free quad as long as you haven't spill-locked them all.
|
||||
int QGetFreeQuad(int start, int count, const char *reason);
|
||||
int GetNumARMFPURegs();
|
||||
|
||||
private:
|
||||
void SetupInitialRegs();
|
||||
|
||||
MIPSState *mips_;
|
||||
ARMXEmitter *emit_;
|
||||
u32 compilerPC_;
|
||||
MIPSComp::JitState *js_;
|
||||
MIPSComp::ArmJitOptions *jo_;
|
||||
|
||||
int numARMFpuReg_;
|
||||
int qTime_;
|
||||
|
||||
enum {
|
||||
MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON
|
||||
// With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers
|
||||
// are individually mappable though.
|
||||
MAX_ARMFPUREG = 32,
|
||||
MAX_ARMQUADS = 16,
|
||||
NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS,
|
||||
};
|
||||
|
||||
FPURegARM ar[MAX_ARMFPUREG];
|
||||
FPURegMIPS mr[NUM_MIPSFPUREG];
|
||||
FPURegQuad qr[MAX_ARMQUADS];
|
||||
FPURegMIPS *vr;
|
||||
|
||||
bool pendingFlush;
|
||||
|
||||
@@ -64,6 +64,7 @@ ARCH_FILES := \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmAsm.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmJit.cpp \
|
||||
@@ -84,6 +85,7 @@ ARCH_FILES := \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmAsm.cpp \
|
||||
$(SRC)/Core/MIPS/ARM/ArmJit.cpp \
|
||||
|
||||
Reference in new issue
Block a user