diff --git a/CMakeLists.txt b/CMakeLists.txt
index bb71c0c6cf..e520e72672 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -1022,6 +1022,7 @@ if(ARM)
Core/MIPS/ARM/ArmCompLoadStore.cpp
Core/MIPS/ARM/ArmCompVFPU.cpp
Core/MIPS/ARM/ArmCompVFPUNEON.cpp
+ Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp
Core/MIPS/ARM/ArmCompReplace.cpp
Core/MIPS/ARM/ArmJit.cpp
Core/MIPS/ARM/ArmJit.h
diff --git a/Core/Core.vcxproj b/Core/Core.vcxproj
index ad0cc66e4a..6725309679 100644
--- a/Core/Core.vcxproj
+++ b/Core/Core.vcxproj
@@ -317,6 +317,18 @@
true
true
+
+ true
+ true
+ true
+ true
+
+
+ true
+ true
+ true
+ true
+
true
true
@@ -563,7 +575,7 @@
true
true
-
+
true
true
true
@@ -658,4 +670,4 @@
-
\ No newline at end of file
+
diff --git a/Core/Core.vcxproj.filters b/Core/Core.vcxproj.filters
index 3144b4caae..915ae6c43d 100644
--- a/Core/Core.vcxproj.filters
+++ b/Core/Core.vcxproj.filters
@@ -393,6 +393,12 @@
MIPS\ARM
+
+ MIPS\ARM
+
+
+ MIPS\ARM
+
Ext
@@ -969,7 +975,7 @@
MIPS\JitCommon
-
+
MIPS\ARM
@@ -1027,4 +1033,4 @@
-
\ No newline at end of file
+
diff --git a/Core/MIPS/ARM/ArmCompFPU.cpp b/Core/MIPS/ARM/ArmCompFPU.cpp
index 37986404bb..e0a022477d 100644
--- a/Core/MIPS/ARM/ArmCompFPU.cpp
+++ b/Core/MIPS/ARM/ArmCompFPU.cpp
@@ -357,9 +357,12 @@ void Jit::Comp_mxc1(MIPSOpcode op)
switch ((op >> 21) & 0x1f)
{
case 0: // R(rt) = FI(fs); break; //mfc1
- fpr.MapReg(fs);
gpr.MapReg(rt, MAP_DIRTY | MAP_NOINIT);
- VMOV(gpr.R(rt), fpr.R(fs));
+ if (fpr.IsMapped(fs)) {
+ VMOV(gpr.R(rt), fpr.R(fs));
+ } else {
+ LDR(gpr.R(rt), CTXREG, fpr.GetMipsRegOffset(fs));
+ }
return;
case 2: //cfc1
@@ -393,11 +396,11 @@ void Jit::Comp_mxc1(MIPSOpcode op)
case 4: //FI(fs) = R(rt); break; //mtc1
if (rt == MIPS_REG_ZERO) {
- fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT);
+ fpr.MapReg(fs, MAP_NOINIT);
MOVI2F(fpr.R(fs), 0.0f, R0);
} else {
gpr.MapReg(rt);
- fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT);
+ fpr.MapReg(fs, MAP_NOINIT);
VMOV(fpr.R(fs), gpr.R(rt));
}
return;
diff --git a/Core/MIPS/ARM/ArmCompVFPU.cpp b/Core/MIPS/ARM/ArmCompVFPU.cpp
index 3993b64f48..a05aee73a9 100644
--- a/Core/MIPS/ARM/ArmCompVFPU.cpp
+++ b/Core/MIPS/ARM/ArmCompVFPU.cpp
@@ -324,8 +324,8 @@ namespace MIPSComp
void Jit::Comp_SVQ(MIPSOpcode op)
{
- CONDITIONAL_DISABLE;
NEON_IF_AVAILABLE(CompNEON_SVQ);
+ CONDITIONAL_DISABLE;
int imm = (signed short)(op&0xFFFC);
int vt = (((op >> 16) & 0x1f)) | ((op&1) << 5);
@@ -506,6 +506,7 @@ namespace MIPSComp
void Jit::Comp_VIdt(MIPSOpcode op) {
NEON_IF_AVAILABLE(CompNEON_VIdt);
+
CONDITIONAL_DISABLE;
if (js.HasUnknownPrefix()) {
DISABLE;
@@ -1574,7 +1575,7 @@ namespace MIPSComp
VMUL(fpr.V(temp3), fpr.V(sregs[0]), fpr.V(tregs[1]));
VMLS(fpr.V(temp3), fpr.V(sregs[1]), fpr.V(tregs[0]));
- fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT);
+ fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT);
VMOV(fpr.V(dregs[0]), S0);
VMOV(fpr.V(dregs[1]), S1);
VMOV(fpr.V(dregs[2]), fpr.V(temp3));
@@ -1609,7 +1610,7 @@ namespace MIPSComp
VMLS(fpr.V(temp4), fpr.V(sregs[2]), fpr.V(tregs[2]));
VMLA(fpr.V(temp4), fpr.V(sregs[3]), fpr.V(tregs[3]));
- fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT);
+ fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT);
VMOV(fpr.V(dregs[0]), S0);
VMOV(fpr.V(dregs[1]), S1);
VMOV(fpr.V(dregs[2]), fpr.V(temp3));
@@ -2118,5 +2119,4 @@ namespace MIPSComp
void Jit::Comp_Vbfy(MIPSOpcode op) {
DISABLE;
}
-
}
diff --git a/Core/MIPS/ARM/ArmCompVFPUNEON.cpp b/Core/MIPS/ARM/ArmCompVFPUNEON.cpp
index a9d17595e7..0ec5ef2061 100644
--- a/Core/MIPS/ARM/ArmCompVFPUNEON.cpp
+++ b/Core/MIPS/ARM/ArmCompVFPUNEON.cpp
@@ -20,7 +20,16 @@
// that uses NEON Q registers to cache pairs/tris/quads, and so on.
// Will require major extensions to the reg cache and other things.
+// ARM NEON can only do pairs and quads, not tris and scalars.
+// We can do scalars, though, for many operations if all the operands
+// are below Q8 (D16, S32) using regular VFP instructions but really not sure
+// if it's worth it.
+
+
+
#include
+
+#include "base/logging.h"
#include "math/math_util.h"
#include "Common/CPUDetect.h"
@@ -28,49 +37,703 @@
#include "Core/MIPS/MIPS.h"
#include "Core/MIPS/MIPSAnalyst.h"
#include "Core/MIPS/MIPSCodeUtils.h"
+#include "Core/MIPS/MIPSVFPUUtils.h"
#include "Core/Config.h"
#include "Core/Reporting.h"
#include "Core/MIPS/ARM/ArmJit.h"
#include "Core/MIPS/ARM/ArmRegCache.h"
+#include "Core/MIPS/ARM/ArmCompVFPUNEONUtil.h"
// TODO: Somehow #ifdef away on ARMv5eabi, without breaking the linker.
+// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
#define CONDITIONAL_DISABLE ;
#define DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
+#define DISABLE_UNKNOWN_PREFIX { WLOG("DISABLE: Unknown Prefix in %s", __FUNCTION__); fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
+
+
+#define _RS MIPS_GET_RS(op)
+#define _RT MIPS_GET_RT(op)
+#define _RD MIPS_GET_RD(op)
+#define _FS MIPS_GET_FS(op)
+#define _FT MIPS_GET_FT(op)
+#define _FD MIPS_GET_FD(op)
+#define _SA MIPS_GET_SA(op)
+#define _POS ((op>> 6) & 0x1F)
+#define _SIZE ((op>>11) & 0x1F)
+#define _IMM16 (signed short)(op & 0xFFFF)
+#define _IMM26 (op & 0x03FFFFFF)
+
namespace MIPSComp {
+static const float minus_one = -1.0f;
+static const float one = 1.0f;
+static const float zero = 0.0f;
+
+
+void Jit::CompNEON_VecDo3(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ MappedRegs r = NEONMapDirtyInIn(op, sz, sz, sz);
+ ARMReg temp = MatchSize(Q0, r.vs);
+ // TODO: Special case for scalar
+ switch (op >> 26) {
+ case 24: //VFPU0
+ switch ((op >> 23) & 7) {
+ case 0: VADD(F_32, r.vd, r.vs, r.vt); break; // vadd
+ case 1: VSUB(F_32, r.vd, r.vs, r.vt); break; // vsub
+ case 7: // vdiv // vdiv THERE IS NO NEON SIMD VDIV :( There's a fast reciprocal iterator thing though.
+ {
+ // Implement by falling back to VFP
+ VMOV(D0, D_0(r.vs));
+ VMOV(D1, D_0(r.vt));
+ VDIV(S0, S0, S2);
+ if (sz >= V_Pair)
+ VDIV(S1, S1, S3);
+ VMOV(D_0(r.vd), D0);
+ if (sz >= V_Triple) {
+ VMOV(D0, D_1(r.vs));
+ VMOV(D1, D_1(r.vt));
+ VDIV(S0, S0, S2);
+ if (sz == V_Quad)
+ VDIV(S1, S1, S3);
+ VMOV(D_1(r.vd), D0);
+ }
+ }
+ break;
+ default:
+ DISABLE;
+ }
+ break;
+ case 25: //VFPU1
+ switch ((op >> 23) & 7) {
+ case 0: VMUL(F_32, r.vd, r.vs, r.vt); break; // vmul
+ default:
+ DISABLE;
+ }
+ break;
+ case 27: //VFPU3
+ switch ((op >> 23) & 7) {
+ case 2: VMIN(F_32, r.vd, r.vs, r.vt); break; // vmin
+ case 3: VMAX(F_32, r.vd, r.vs, r.vt); break; // vmax
+ case 6: // vsge
+ VMOV_immf(temp, 1.0f);
+ VCGE(F_32, r.vd, r.vs, r.vt);
+ VAND(r.vd, r.vd, temp);
+ break;
+ case 7: // vslt
+ VMOV_immf(temp, 1.0f);
+ VCLT(F_32, r.vd, r.vs, r.vt);
+ VAND(r.vd, r.vd, temp);
+ break;
+ }
+ break;
+
+ default:
+ DISABLE;
+ }
+
+ NEONApplyPrefixD(r.vd);
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
+}
+
+
+// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
+
void Jit::CompNEON_SV(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+
+ // Remember to use single lane stores here and not VLDR/VSTR - switching usage
+ // between NEON and VFPU can be expensive on some chips.
+
+ // Here's a common idiom we should optimize:
+ // lv.s S200, 0(s4)
+ // lv.s S201, 4(s4)
+ // lv.s S202, 8(s4)
+ // vone.s S203
+ // vtfm4.q C000, E600, C200
+ // Would be great if we could somehow combine the lv.s into one vector instead of mapping three
+ // separate quads.
+
+ s32 offset = (signed short)(op & 0xFFFC);
+ int vt = ((op >> 16) & 0x1f) | ((op & 3) << 5);
+ MIPSGPReg rs = _RS;
+
+ bool doCheck = false;
+ switch (op >> 26)
+ {
+ case 50: //lv.s // VI(vt) = Memory::Read_U32(addr);
+ {
+ if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && (offset & 3) == 0 && offset < 0x400 && offset > -0x400) {
+ INFO_LOG(HLE, "LV.S fastmode!");
+ // TODO: Also look forward and combine multiple loads.
+ gpr.MapRegAsPointer(rs);
+ ARMReg ar = fpr.QMapReg(vt, V_Single, MAP_NOINIT | MAP_DIRTY);
+ if (offset) {
+ ADDI2R(R0, gpr.RPtr(rs), offset, R1);
+ VLD1_lane(F_32, ar, R0, 0, true);
+ } else {
+ VLD1_lane(F_32, ar, gpr.RPtr(rs), 0, true);
+ }
+ break;
+ }
+ INFO_LOG(HLE, "LV.S slowmode!");
+
+ // CC might be set by slow path below, so load regs first.
+ ARMReg ar = fpr.QMapReg(vt, V_Single, MAP_DIRTY | MAP_NOINIT);
+ if (gpr.IsImm(rs)) {
+ u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF;
+ gpr.SetRegImm(R0, addr + (u32)Memory::base);
+ } else {
+ gpr.MapReg(rs);
+ if (g_Config.bFastMemory) {
+ SetR0ToEffectiveAddress(rs, offset);
+ } else {
+ SetCCAndR0ForSafeAddress(rs, offset, R1);
+ doCheck = true;
+ }
+ ADD(R0, R0, MEMBASEREG);
+ }
+ FixupBranch skip;
+ if (doCheck) {
+ skip = B_CC(CC_EQ);
+ }
+ VLD1_lane(F_32, ar, R0, 0, true);
+ if (doCheck) {
+ SetJumpTarget(skip);
+ SetCC(CC_AL);
+ }
+ }
+ break;
+
+ case 58: //sv.s // Memory::Write_U32(VI(vt), addr);
+ {
+ if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && (offset & 3) == 0 && offset < 0x400 && offset > -0x400) {
+ INFO_LOG(HLE, "SV.S fastmode!");
+ // TODO: Also look forward and combine multiple stores.
+ gpr.MapRegAsPointer(rs);
+ ARMReg ar = fpr.QMapReg(vt, V_Single, 0);
+ if (offset) {
+ ADDI2R(R0, gpr.RPtr(rs), offset, R1);
+ VST1_lane(F_32, ar, R0, 0, true);
+ } else {
+ VST1_lane(F_32, ar, gpr.RPtr(rs), 0, true);
+ }
+ break;
+ }
+
+ INFO_LOG(HLE, "SV.S slowmode!");
+ // CC might be set by slow path below, so load regs first.
+ ARMReg ar = fpr.QMapReg(vt, V_Single, 0);
+ if (gpr.IsImm(rs)) {
+ u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF;
+ gpr.SetRegImm(R0, addr + (u32)Memory::base);
+ } else {
+ gpr.MapReg(rs);
+ if (g_Config.bFastMemory) {
+ SetR0ToEffectiveAddress(rs, offset);
+ } else {
+ SetCCAndR0ForSafeAddress(rs, offset, R1);
+ doCheck = true;
+ }
+ ADD(R0, R0, MEMBASEREG);
+ }
+ FixupBranch skip;
+ if (doCheck) {
+ skip = B_CC(CC_EQ);
+ }
+ VST1_lane(F_32, ar, R0, 0, true);
+ if (doCheck) {
+ SetJumpTarget(skip);
+ SetCC(CC_AL);
+ }
+ }
+ break;
+ }
+ fpr.ReleaseSpillLocksAndDiscardTemps();
+}
+
+inline int MIPS_GET_VQVT(u32 op) {
+ return (((op >> 16) & 0x1f)) | ((op & 1) << 5);
}
void Jit::CompNEON_SVQ(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+
+ int offset = (signed short)(op & 0xFFFC);
+ int vt = MIPS_GET_VQVT(op.encoding);
+ MIPSGPReg rs = _RS;
+ bool doCheck = false;
+ switch (op >> 26)
+ {
+ case 54: //lv.q
+ {
+ // Check for four-in-a-row
+ const u32 ops[4] = {
+ op.encoding,
+ Memory::Read_Instruction(js.compilerPC + 4).encoding,
+ Memory::Read_Instruction(js.compilerPC + 8).encoding,
+ Memory::Read_Instruction(js.compilerPC + 12).encoding
+ };
+ if (g_Config.bFastMemory && (ops[1] >> 26) == 54 && (ops[2] >> 26) == 54 && (ops[3] >> 26) == 54) {
+ int offsets[4] = {offset, (s16)(ops[1] & 0xFFFC), (s16)(ops[2] & 0xFFFC), (s16)(ops[3] & 0xFFFC)};
+ int rss[4] = {MIPS_GET_RS(op), MIPS_GET_RS(ops[1]), MIPS_GET_RS(ops[2]), MIPS_GET_RS(ops[3])};
+ if (offsets[1] == offset + 16 && offsets[2] == offsets[1] + 16 && offsets[3] == offsets[2] + 16 &&
+ rss[0] == rss[1] && rss[1] == rss[2] && rss[2] == rss[3]) {
+ int vts[4] = {MIPS_GET_VQVT(op.encoding), MIPS_GET_VQVT(ops[1]), MIPS_GET_VQVT(ops[2]), MIPS_GET_VQVT(ops[3])};
+ // Also check the destination registers!
+
+ // Detected four consecutive ones!
+ // gpr.MapRegAsPointer(rs);
+ // fpr.QLoad4x4(vts[4], rs, offset);
+ INFO_LOG(JIT, "Matrix load detected! TODO: optimize");
+ // break;
+ }
+ }
+
+ if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && offset < 0x400-16 && offset > -0x400-16) {
+ gpr.MapRegAsPointer(rs);
+ ARMReg ar = fpr.QMapReg(vt, V_Quad, MAP_DIRTY | MAP_NOINIT);
+ if (offset) {
+ ADDI2R(R0, gpr.RPtr(rs), offset, R1);
+ VLD1(F_32, ar, R0, 2, ALIGN_128);
+ } else {
+ VLD1(F_32, ar, gpr.RPtr(rs), 2, ALIGN_128);
+ }
+ break;
+ }
+
+ // CC might be set by slow path below, so load regs first.
+ ARMReg ar = fpr.QMapReg(vt, V_Quad, MAP_DIRTY | MAP_NOINIT);
+ if (gpr.IsImm(rs)) {
+ u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF;
+ gpr.SetRegImm(R0, addr + (u32)Memory::base);
+ } else {
+ gpr.MapReg(rs);
+ if (g_Config.bFastMemory) {
+ SetR0ToEffectiveAddress(rs, offset);
+ } else {
+ SetCCAndR0ForSafeAddress(rs, offset, R1);
+ doCheck = true;
+ }
+ ADD(R0, R0, MEMBASEREG);
+ }
+
+ FixupBranch skip;
+ if (doCheck) {
+ skip = B_CC(CC_EQ);
+ }
+
+ VLD1(F_32, ar, R0, 2, ALIGN_128);
+
+ if (doCheck) {
+ SetJumpTarget(skip);
+ SetCC(CC_AL);
+ }
+ }
+ break;
+
+ case 62: //sv.q
+ {
+ const u32 ops[4] = {
+ op.encoding,
+ Memory::Read_Instruction(js.compilerPC + 4).encoding,
+ Memory::Read_Instruction(js.compilerPC + 8).encoding,
+ Memory::Read_Instruction(js.compilerPC + 12).encoding
+ };
+ if (g_Config.bFastMemory && (ops[1] >> 26) == 54 && (ops[2] >> 26) == 54 && (ops[3] >> 26) == 54) {
+ int offsets[4] = { offset, (s16)(ops[1] & 0xFFFC), (s16)(ops[2] & 0xFFFC), (s16)(ops[3] & 0xFFFC) };
+ int rss[4] = { MIPS_GET_RS(op), MIPS_GET_RS(ops[1]), MIPS_GET_RS(ops[2]), MIPS_GET_RS(ops[3]) };
+ if (offsets[1] == offset + 16 && offsets[2] == offsets[1] + 16 && offsets[3] == offsets[2] + 16 &&
+ rss[0] == rss[1] && rss[1] == rss[2] && rss[2] == rss[3]) {
+ int vts[4] = { MIPS_GET_VQVT(op.encoding), MIPS_GET_VQVT(ops[1]), MIPS_GET_VQVT(ops[2]), MIPS_GET_VQVT(ops[3]) };
+ // Also check the destination registers!
+
+ // Detected four consecutive ones!
+ // gpr.MapRegAsPointer(rs);
+ // fpr.QLoad4x4(vts[4], rs, offset);
+ INFO_LOG(JIT, "Matrix store detected! TODO: optimize");
+ // break;
+ }
+ }
+
+ if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && offset < 0x400-16 && offset > -0x400-16) {
+ gpr.MapRegAsPointer(rs);
+ ARMReg ar = fpr.QMapReg(vt, V_Quad, 0);
+ if (offset) {
+ ADDI2R(R0, gpr.RPtr(rs), offset, R1);
+ VST1(F_32, ar, R0, 2, ALIGN_128);
+ } else {
+ VST1(F_32, ar, gpr.RPtr(rs), 2, ALIGN_128);
+ }
+ break;
+ }
+
+ // CC might be set by slow path below, so load regs first.
+ u8 vregs[4];
+ ARMReg ar = fpr.QMapReg(vt, V_Quad, 0);
+
+ if (gpr.IsImm(rs)) {
+ u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF;
+ gpr.SetRegImm(R0, addr + (u32)Memory::base);
+ } else {
+ gpr.MapReg(rs);
+ if (g_Config.bFastMemory) {
+ SetR0ToEffectiveAddress(rs, offset);
+ } else {
+ SetCCAndR0ForSafeAddress(rs, offset, R1);
+ doCheck = true;
+ }
+ ADD(R0, R0, MEMBASEREG);
+ }
+
+ FixupBranch skip;
+ if (doCheck) {
+ skip = B_CC(CC_EQ);
+ }
+
+ VST1(F_32, ar, R0, 2, ALIGN_128);
+
+ if (doCheck) {
+ SetJumpTarget(skip);
+ SetCC(CC_AL);
+ }
+ }
+ break;
+
+ default:
+ DISABLE;
+ break;
+ }
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_VVectorInit(MIPSOpcode op) {
- DISABLE;
-}
+ CONDITIONAL_DISABLE;
+ // WARNING: No prefix support!
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+ VectorSize sz = GetVecSize(op);
+ DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_NOINIT | MAP_DIRTY);
-void Jit::CompNEON_VMatrixInit(MIPSOpcode op) {
- DISABLE;
+ switch ((op >> 16) & 0xF) {
+ case 6: // vzero
+ VEOR(vd.rd, vd.rd, vd.rd);
+ break;
+ case 7: // vone
+ VMOV_immf(vd.rd, 1.0f);
+ break;
+ default:
+ DISABLE;
+ break;
+ }
+ NEONApplyPrefixD(vd);
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_VDot(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ MappedRegs r = NEONMapDirtyInIn(op, V_Single, sz, sz);
+
+ switch (sz) {
+ case V_Pair:
+ VMUL(F_32, r.vd, r.vs, r.vt);
+ VPADD(F_32, r.vd, r.vd, r.vd);
+ break;
+ case V_Triple:
+ VMUL(F_32, Q0, r.vs, r.vt);
+ VPADD(F_32, D0, D0, D0);
+ VADD(F_32, r.vd, D0, D1);
+ break;
+ case V_Quad:
+ VMUL(F_32, D0, D_0(r.vs), D_0(r.vt));
+ VMLA(F_32, D0, D_1(r.vs), D_1(r.vt));
+ VPADD(F_32, r.vd, D0, D0);
+ break;
+ case V_Single:
+ case V_Invalid:
+ ;
+ }
+
+ NEONApplyPrefixD(r.vd);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
-void Jit::CompNEON_VecDo3(MIPSOpcode op) {
+
+void Jit::CompNEON_VHdp(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
DISABLE;
+
+ // Similar to VDot but the last component is only s instead of s * t.
+ // A bit tricky on NEON...
+}
+
+void Jit::CompNEON_VScl(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ MappedRegs r = NEONMapDirtyInIn(op, sz, sz, V_Single);
+
+ ARMReg temp = MatchSize(Q0, r.vt);
+
+ // TODO: VMUL_scalar directly when possible
+ VMOV_neon(temp, r.vt);
+ VMUL_scalar(F_32, r.vd, r.vs, DScalar(Q0, 0));
+
+ NEONApplyPrefixD(r.vd);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_VV2Op(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ // Pre-processing: Eliminate silly no-op VMOVs, common in Wipeout Pure
+ if (((op >> 16) & 0x1f) == 0 && _VS == _VD && js.HasNoPrefix()) {
+ return;
+ }
+
+ // Must bail before we start mapping registers.
+ switch ((op >> 16) & 0x1f) {
+ case 0: // d[i] = s[i]; break; //vmov
+ case 1: // d[i] = fabsf(s[i]); break; //vabs
+ case 2: // d[i] = -s[i]; break; //vneg
+ case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
+ break;
+
+ default:
+ DISABLE;
+ break;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ MappedRegs r = NEONMapDirtyIn(op, sz, sz);
+
+ ARMReg temp = MatchSize(Q0, r.vs);
+
+ switch ((op >> 16) & 0x1f) {
+ case 0: // d[i] = s[i]; break; //vmov
+ // Probably for swizzle.
+ VMOV_neon(r.vd, r.vs);
+ break;
+ case 1: // d[i] = fabsf(s[i]); break; //vabs
+ VABS(F_32, r.vd, r.vs);
+ break;
+ case 2: // d[i] = -s[i]; break; //vneg
+ VNEG(F_32, r.vd, r.vs);
+ break;
+
+ case 4: // if (s[i] < 0) d[i] = 0; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat0
+ if (IsD(r.vd)) {
+ VMOV_immf(D0, 0.0f);
+ VMOV_immf(D1, 1.0f);
+ VMAX(F_32, r.vd, r.vs, D0);
+ VMIN(F_32, r.vd, r.vd, D1);
+ } else {
+ VMOV_immf(Q0, 1.0f);
+ VMIN(F_32, r.vd, r.vs, Q0);
+ VMOV_immf(Q0, 0.0f);
+ VMAX(F_32, r.vd, r.vd, Q0);
+ }
+ break;
+ case 5: // if (s[i] < -1.0f) d[i] = -1.0f; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat1
+ if (IsD(r.vd)) {
+ VMOV_immf(D0, -1.0f);
+ VMOV_immf(D1, 1.0f);
+ VMAX(F_32, r.vd, r.vs, D0);
+ VMIN(F_32, r.vd, r.vd, D1);
+ } else {
+ VMOV_immf(Q0, 1.0f);
+ VMIN(F_32, r.vd, r.vs, Q0);
+ VMOV_immf(Q0, -1.0f);
+ VMAX(F_32, r.vd, r.vd, Q0);
+ }
+ break;
+
+ case 16: // d[i] = 1.0f / s[i]; break; //vrcp
+ // Can just fallback to VFP and use VDIV.
+ DISABLE;
+ {
+ ARMReg temp2 = fpr.QAllocTemp(sz);
+ // Needs iterations on NEON. And two temps - which is a problem if vs == vd! Argh!
+ VRECPE(F_32, temp, r.vs);
+ VRECPS(temp2, r.vs, temp);
+ VMUL(F_32, temp2, temp2, temp);
+ VRECPS(temp2, r.vs, temp);
+ VMUL(F_32, temp2, temp2, temp);
+ }
+ // http://stackoverflow.com/questions/6759897/how-to-divide-in-neon-intrinsics-by-a-float-number
+ // reciprocal = vrecpeq_f32(b);
+ // reciprocal = vmulq_f32(vrecpsq_f32(b, reciprocal), reciprocal);
+ // reciprocal = vmulq_f32(vrecpsq_f32(b, reciprocal), reciprocal);
+ DISABLE;
+ break;
+
+ case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
+ DISABLE;
+ // Needs iterations on NEON
+ {
+ if (true) {
+ // Not-very-accurate estimate
+ VRSQRTE(F_32, r.vd, r.vs);
+ } else {
+ ARMReg temp2 = fpr.QAllocTemp(sz);
+ // TODO: It's likely that some games will require one or two Newton-Raphson
+ // iterations to refine the estimate.
+ VRSQRTE(F_32, temp, r.vs);
+ VRSQRTS(temp2, r.vs, temp);
+ VMUL(F_32, r.vd, temp2, temp);
+ //VRSQRTS(temp2, r.vs, temp);
+ // VMUL(F_32, r.vd, temp2, temp);
+ }
+ }
+ break;
+ case 18: // d[i] = sinf((float)M_PI_2 * s[i]); break; //vsin
+ DISABLE;
+ break;
+ case 19: // d[i] = cosf((float)M_PI_2 * s[i]); break; //vcos
+ DISABLE;
+ break;
+ case 20: // d[i] = powf(2.0f, s[i]); break; //vexp2
+ DISABLE;
+ break;
+ case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2
+ DISABLE;
+ break;
+ case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
+ // Let's just defer to VFP for now. Better than calling the interpreter for sure.
+ VMOV_neon(MatchSize(Q0, r.vs), r.vs);
+ for (int i = 0; i < n; i++) {
+ VSQRT((ARMReg)(S0 + i), (ARMReg)(S0 + i));
+ }
+ VMOV_neon(MatchSize(Q0, r.vd), r.vd);
+ break;
+ case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin
+ DISABLE;
+ break;
+ case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
+ // Needs iterations on NEON. Just do the same as vrcp and negate.
+ DISABLE;
+ break;
+ case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
+ DISABLE;
+ break;
+ case 28: // d[i] = 1.0f / expf(s[i] * (float)M_LOG2E); break; // vrexp2
+ DISABLE;
+ break;
+ default:
+ DISABLE;
+ break;
+ }
+
+ NEONApplyPrefixD(r.vd);
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Mftv(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ int imm = op & 0xFF;
+ MIPSGPReg rt = _RT;
+ switch ((op >> 21) & 0x1f) {
+ case 3: //mfv / mfvc
+ // rt = 0, imm = 255 appears to be used as a CPU interlock by some games.
+ if (rt != 0) {
+ if (imm < 128) { //R(rt) = VI(imm);
+ ARMReg r = fpr.QMapReg(imm, V_Single, MAP_READ);
+ gpr.MapReg(rt, MAP_NOINIT | MAP_DIRTY);
+ // TODO: Gotta be a faster way
+ VMOV_neon(MatchSize(Q0, r), r);
+ VMOV(gpr.R(rt), S0);
+ } else if (imm < 128 + VFPU_CTRL_MAX) { //mtvc
+ // In case we have a saved prefix.
+ FlushPrefixV();
+ if (imm - 128 == VFPU_CTRL_CC) {
+ gpr.MapDirtyIn(rt, MIPS_REG_VFPUCC);
+ MOV(gpr.R(rt), gpr.R(MIPS_REG_VFPUCC));
+ } else {
+ gpr.MapReg(rt, MAP_NOINIT | MAP_DIRTY);
+ LDR(gpr.R(rt), CTXREG, offsetof(MIPSState, vfpuCtrl) + 4 * (imm - 128));
+ }
+ } else {
+ //ERROR - maybe need to make this value too an "interlock" value?
+ ERROR_LOG(CPU, "mfv - invalid register %i", imm);
+ }
+ }
+ break;
+
+ case 7: // mtv
+ if (imm < 128) {
+ // TODO: It's pretty common that this is preceded by mfc1, that is, a value is being
+ // moved from the regular floating point registers. It would probably be faster to do
+ // the copy directly in the FPRs instead of going through the GPRs.
+
+ ARMReg r = fpr.QMapReg(imm, V_Single, MAP_DIRTY | MAP_NOINIT);
+ if (gpr.IsMapped(rt)) {
+ VMOV(S0, gpr.R(rt));
+ VMOV_neon(r, MatchSize(Q0, r));
+ } else {
+ ADDI2R(R0, CTXREG, gpr.GetMipsRegOffset(rt), R1);
+ VLD1_lane(F_32, r, R0, 0, true);
+ }
+ } else if (imm < 128 + VFPU_CTRL_MAX) { //mtvc //currentMIPS->vfpuCtrl[imm - 128] = R(rt);
+ if (imm - 128 == VFPU_CTRL_CC) {
+ gpr.MapDirtyIn(MIPS_REG_VFPUCC, rt);
+ MOV(gpr.R(MIPS_REG_VFPUCC), rt);
+ } else {
+ gpr.MapReg(rt);
+ STR(gpr.R(rt), CTXREG, offsetof(MIPSState, vfpuCtrl) + 4 * (imm - 128));
+ }
+ //gpr.BindToRegister(rt, true, false);
+ //MOV(32, M(¤tMIPS->vfpuCtrl[imm - 128]), gpr.R(rt));
+
+ // TODO: Optimization if rt is Imm?
+ // Set these BEFORE disable!
+ if (imm - 128 == VFPU_CTRL_SPREFIX) {
+ js.prefixSFlag = JitState::PREFIX_UNKNOWN;
+ } else if (imm - 128 == VFPU_CTRL_TPREFIX) {
+ js.prefixTFlag = JitState::PREFIX_UNKNOWN;
+ } else if (imm - 128 == VFPU_CTRL_DPREFIX) {
+ js.prefixDFlag = JitState::PREFIX_UNKNOWN;
+ }
+ } else {
+ //ERROR
+ _dbg_assert_msg_(CPU,0,"mtv - invalid register");
+ }
+ break;
+
+ default:
+ DISABLE;
+ }
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vmfvc(MIPSOpcode op) {
@@ -78,31 +741,223 @@ void Jit::CompNEON_Vmfvc(MIPSOpcode op) {
}
void Jit::CompNEON_Vmtvc(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+
+ int vs = _VS;
+ int imm = op & 0xFF;
+ if (imm >= 128 && imm < 128 + VFPU_CTRL_MAX) {
+ ARMReg r = fpr.QMapReg(vs, V_Single, 0);
+ ADDI2R(R0, CTXREG, offsetof(MIPSState, vfpuCtrl[0]) + (imm - 128) * 4, R1);
+ VST1_lane(F_32, r, R0, 0, true);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
+
+ if (imm - 128 == VFPU_CTRL_SPREFIX) {
+ js.prefixSFlag = JitState::PREFIX_UNKNOWN;
+ } else if (imm - 128 == VFPU_CTRL_TPREFIX) {
+ js.prefixTFlag = JitState::PREFIX_UNKNOWN;
+ } else if (imm - 128 == VFPU_CTRL_DPREFIX) {
+ js.prefixDFlag = JitState::PREFIX_UNKNOWN;
+ }
+ }
+}
+
+void Jit::CompNEON_VMatrixInit(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+
+ MatrixSize msz = GetMtxSize(op);
+ int n = GetMatrixSide(msz);
+
+ ARMReg cols[4];
+ fpr.QMapMatrix(cols, _VD, msz, MAP_NOINIT | MAP_DIRTY);
+
+ switch ((op >> 16) & 0xF) {
+ case 3: // vmidt
+ // There has to be a better way to synthesize: 1.0, 0.0, 0.0, 1.0 in a quad
+ VEOR(D0, D0, D0);
+ VMOV_immf(D1, 1.0f);
+ VTRN(F_32, D0, D1);
+ VREV64(I_32, D0, D0);
+ switch (msz) {
+ case M_2x2:
+ VMOV_neon(cols[0], D0);
+ VMOV_neon(cols[1], D1);
+ break;
+ case M_3x3:
+ VMOV_neon(D_0(cols[0]), D0);
+ VMOV_imm(I_8, D_1(cols[0]), VIMMxxxxxxxx, 0);
+ VMOV_neon(D_0(cols[1]), D1);
+ VMOV_imm(I_8, D_1(cols[1]), VIMMxxxxxxxx, 0);
+ VMOV_imm(I_8, D_0(cols[2]), VIMMxxxxxxxx, 0);
+ VMOV_neon(D_1(cols[2]), D0);
+ break;
+ case M_4x4:
+ VMOV_neon(D_0(cols[0]), D0);
+ VMOV_imm(I_8, D_1(cols[0]), VIMMxxxxxxxx, 0);
+ VMOV_neon(D_0(cols[1]), D1);
+ VMOV_imm(I_8, D_1(cols[1]), VIMMxxxxxxxx, 0);
+ VMOV_imm(I_8, D_0(cols[2]), VIMMxxxxxxxx, 0);
+ VMOV_neon(D_1(cols[2]), D0);
+ VMOV_imm(I_8, D_0(cols[3]), VIMMxxxxxxxx, 0);
+ VMOV_neon(D_1(cols[3]), D1);
+
+ // NEONTranspose4x4(cols);
+ break;
+ default:
+ _assert_msg_(JIT, 0, "Bad matrix size");
+ break;
+ }
+ break;
+ case 6: // vmzero
+ for (int i = 0; i < n; i++) {
+ VEOR(cols[i], cols[i], cols[i]);
+ }
+ break;
+ case 7: // vmone
+ for (int i = 0; i < n; i++) {
+ VMOV_immf(cols[i], 1.0f);
+ }
+ break;
+ }
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vmmov(MIPSOpcode op) {
- DISABLE;
-}
+ CONDITIONAL_DISABLE;
+ if (_VS == _VD) {
+ // A lot of these no-op matrix moves in Wipeout... Just drop the instruction entirely.
+ return;
+ }
-void Jit::CompNEON_VScl(MIPSOpcode op) {
- DISABLE;
+ MatrixSize msz = GetMtxSize(op);
+
+ MatrixOverlapType overlap = GetMatrixOverlap(_VD, _VS, msz);
+ if (overlap != OVERLAP_NONE) {
+ // Too complicated to bother handling in the JIT.
+ // TODO: Special case for in-place (and other) transpose, etc.
+ DISABLE;
+ }
+
+ ARMReg s_cols[4], d_cols[4];
+ fpr.QMapMatrix(s_cols, _VS, msz, 0);
+ fpr.QMapMatrix(d_cols, _VD, msz, MAP_DIRTY | MAP_NOINIT);
+
+ int n = GetMatrixSide(msz);
+ for (int i = 0; i < n; i++) {
+ VMOV_neon(d_cols[i], s_cols[i]);
+ }
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vmmul(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+
+ MatrixSize msz = GetMtxSize(op);
+ int n = GetMatrixSide(msz);
+
+ bool overlap = GetMatrixOverlap(_VD, _VS, msz) || GetMatrixOverlap(_VD, _VT, msz);
+ if (overlap) {
+ // Later. Fortunately, the VFPU also seems to prohibit overlap for matrix mul.
+ ILOG("Matrix overlap, ignoring.");
+ DISABLE;
+ }
+
+ // Having problems with 2x2s for some reason.
+ if (msz == M_2x2) {
+ DISABLE;
+ }
+
+ ARMReg s_cols[4], t_cols[4], d_cols[4];
+
+ // For some reason, vmmul is encoded with the first matrix (S) transposed from the real meaning.
+ fpr.QMapMatrix(t_cols, _VT, msz, MAP_FORCE_LOW); // Need to see if we can avoid having to force it low in some sane way. Will need crazy prediction logic for loads otherwise.
+ fpr.QMapMatrix(s_cols, Xpose(_VS), msz, MAP_PREFER_HIGH);
+ fpr.QMapMatrix(d_cols, _VD, msz, MAP_PREFER_HIGH | MAP_NOINIT | MAP_DIRTY);
+
+ // TODO: Getting there but still getting wrong results.
+ for (int i = 0; i < n; i++) {
+ for (int j = 0; j < n; j++) {
+ if (i == 0) {
+ VMUL_scalar(F_32, d_cols[j], s_cols[i], XScalar(t_cols[j], i));
+ } else {
+ VMLA_scalar(F_32, d_cols[j], s_cols[i], XScalar(t_cols[j], i));
+ }
+ }
+ }
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vmscl(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+
+ MatrixSize msz = GetMtxSize(op);
+
+ bool overlap = GetMatrixOverlap(_VD, _VS, msz);
+ if (overlap) {
+ DISABLE;
+ }
+
+ int n = GetMatrixSide(msz);
+
+ ARMReg s_cols[4], t, d_cols[4];
+ fpr.QMapMatrix(s_cols, _VS, msz, 0);
+ fpr.QMapMatrix(d_cols, _VD, msz, MAP_NOINIT | MAP_DIRTY);
+
+ t = fpr.QMapReg(_VT, V_Single, 0);
+ VMOV_neon(D0, t);
+ for (int i = 0; i < n; i++) {
+ VMUL_scalar(F_32, d_cols[i], s_cols[i], DScalar(D0, 0));
+ }
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vtfm(MIPSOpcode op) {
- DISABLE;
-}
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE;
+ }
-void Jit::CompNEON_VHdp(MIPSOpcode op) {
- DISABLE;
+ if (_VT == _VD) {
+ DISABLE;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ MatrixSize msz = GetMtxSize(op);
+ int n = GetNumVectorElements(sz);
+ int ins = (op >> 23) & 7;
+
+ bool homogenous = false;
+ if (n == ins) {
+ n++;
+ sz = (VectorSize)((int)(sz)+1);
+ msz = (MatrixSize)((int)(msz)+1);
+ homogenous = true;
+ }
+ // Otherwise, n should already be ins + 1.
+ else if (n != ins + 1) {
+ DISABLE;
+ }
+
+ ARMReg s_cols[4], t, d;
+ t = fpr.QMapReg(_VT, sz, MAP_FORCE_LOW);
+ fpr.QMapMatrix(s_cols, Xpose(_VS), msz, MAP_PREFER_HIGH);
+ d = fpr.QMapReg(_VD, sz, MAP_DIRTY | MAP_NOINIT | MAP_PREFER_HIGH);
+
+ VMUL_scalar(F_32, d, s_cols[0], XScalar(t, 0));
+ for (int i = 1; i < n; i++) {
+ if (homogenous && i == n - 1) {
+ VADD(F_32, d, d, s_cols[i]);
+ } else {
+ VMLA_scalar(F_32, d, s_cols[i], XScalar(t, i));
+ }
+ }
+
+ // VTFM does not have prefix support.
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_VCrs(MIPSOpcode op) {
@@ -126,55 +981,457 @@ void Jit::CompNEON_Vf2i(MIPSOpcode op) {
}
void Jit::CompNEON_Vi2f(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE;
+ }
+
DISABLE;
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ int imm = (op >> 16) & 0x1f;
+ const float mult = 1.0f / (float)(1UL << imm);
+
+ MappedRegs regs = NEONMapDirtyIn(op, sz, sz);
+
+ MOVI2F_neon(MatchSize(Q0, regs.vd), mult, R0);
+
+ VCVT(F_32, regs.vd, regs.vs);
+ VMUL(F_32, regs.vd, regs.vd, Q0);
+
+ NEONApplyPrefixD(regs.vd);
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vh2f(MIPSOpcode op) {
- DISABLE;
+ if (!cpu_info.bHalf) {
+ // No hardware support for half-to-float, fallback to interpreter
+ // TODO: Translate the fast SSE solution to standard integer/VFP stuff
+ // for the weaker CPUs.
+ DISABLE;
+ }
+
+ VectorSize sz = GetVecSize(op);
+
+ VectorSize outsize = V_Pair;
+ switch (sz) {
+ case V_Single:
+ outsize = V_Pair;
+ break;
+ case V_Pair:
+ outsize = V_Quad;
+ break;
+ default:
+ ERROR_LOG(JIT, "Vh2f: Must be pair or quad");
+ break;
+ }
+
+ ARMReg vs = NEONMapPrefixS(_VS, sz, 0);
+ DestARMReg vd = NEONMapPrefixD(_VD, outsize, MAP_DIRTY | (vd == vs ? 0 : MAP_NOINIT));
+
+ VCVTF32F16(vd.rd, vs);
+
+ NEONApplyPrefixD(vd);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vcst(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ int conNum = (op >> 16) & 0x1f;
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+ DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_DIRTY | MAP_NOINIT);
+ gpr.SetRegImm(R0, (u32)(void *)&cst_constants[conNum]);
+ VLD1_all_lanes(F_32, vd, R0, true);
+ NEONApplyPrefixD(vd); // TODO: Could bake this into the constant we load.
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vhoriz(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+ VectorSize sz = GetVecSize(op);
+ // Do any games use these a noticeable amount?
+ switch ((op >> 16) & 31) {
+ case 6: // vfad
+ {
+ MappedRegs r = NEONMapDirtyIn(op, V_Single, sz);
+ switch (sz) {
+ case V_Pair:
+ VPADD(F_32, r.vd, r.vs, r.vs);
+ break;
+ case V_Triple:
+ VPADD(F_32, D0, D_0(r.vs), D_0(r.vs));
+ VADD(F_32, r.vd, D0, D_1(r.vs));
+ break;
+ case V_Quad:
+ VADD(F_32, R0, D_0(r.vs), D_1(r.vs));
+ VPADD(F_32, r.vd, R0, R0);
+ break;
+ default:
+ ;
+ }
+ break;
+ }
+
+ case 7: // vavg
+ DISABLE;
+ break;
+ }
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_VRot(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
DISABLE;
+
+ int vd = _VD;
+ int vs = _VS;
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ // ...
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_VIdt(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_NOINIT | MAP_DIRTY);
+ switch (sz) {
+ case V_Pair:
+ VMOV_immf(vd, 1.0f);
+ if ((_VD & 1) == 0) {
+ // Load with 1.0, 0.0
+ VMOV_imm(I_64, D0, VIMMbits2bytes, 0x0F);
+ VAND(vd, vd, D0);
+ } else {
+ VMOV_imm(I_64, D0, VIMMbits2bytes, 0xF0);
+ VAND(vd, vd, D0);
+ }
+ break;
+ case V_Triple:
+ case V_Quad:
+ {
+ // TODO: This can be optimized.
+ VEOR(vd, vd, vd);
+ ARMReg dest = (_VD & 2) ? D_1(vd) : D_0(vd);
+ VMOV_immf(dest, 1.0f);
+ if ((_VD & 1) == 0) {
+ // Load with 1.0, 0.0
+ VMOV_imm(I_64, D0, VIMMbits2bytes, 0x0F);
+ VAND(dest, dest, D0);
+ } else {
+ VMOV_imm(I_64, D0, VIMMbits2bytes, 0xF0);
+ VAND(dest, dest, D0);
+ }
+ }
+ break;
+ default:
+ _dbg_assert_msg_(CPU,0,"Bad vidt instruction");
+ break;
+ }
+
+ NEONApplyPrefixD(vd);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vcmp(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix())
+ DISABLE;
+
+ // Not a chance that this works on the first try :P
DISABLE;
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ VCondition cond = (VCondition)(op & 0xF);
+
+ MappedRegs regs = NEONMapInIn(op, sz, sz);
+
+ ARMReg vs = regs.vs, vt = regs.vt;
+ ARMReg res = fpr.QAllocTemp(sz);
+
+ // Some, we just fall back to the interpreter.
+ // ES is just really equivalent to (value & 0x7F800000) == 0x7F800000.
+ switch (cond) {
+ case VC_EI: // c = my_isinf(s[i]); break;
+ case VC_NI: // c = !my_isinf(s[i]); break;
+ DISABLE;
+ case VC_ES: // c = my_isnan(s[i]) || my_isinf(s[i]); break; // Tekken Dark Resurrection
+ case VC_NS: // c = !my_isnan(s[i]) && !my_isinf(s[i]); break;
+ case VC_EN: // c = my_isnan(s[i]); break;
+ case VC_NN: // c = !my_isnan(s[i]); break;
+ // if (_VS != _VT)
+ DISABLE;
+ break;
+
+ case VC_EZ:
+ case VC_NZ:
+ VMOV_immf(Q0, 0.0f);
+ break;
+ default:
+ ;
+ }
+
+ int affected_bits = (1 << 4) | (1 << 5); // 4 and 5
+ for (int i = 0; i < n; i++) {
+ affected_bits |= 1 << i;
+ }
+
+ // Preload the pointer to our magic mask
+ static const u32 collectorBits[4] = { 1, 2, 4, 8 };
+ MOVP2R(R1, &collectorBits);
+
+ // Do the compare
+ MOVI2R(R0, 0);
+ CCFlags flag = CC_AL;
+
+ bool oneIsFalse = false;
+ switch (cond) {
+ case VC_FL: // c = 0;
+ break;
+
+ case VC_TR: // c = 1
+ MOVI2R(R0, affected_bits);
+ break;
+
+ case VC_ES: // c = my_isnan(s[i]) || my_isinf(s[i]); break; // Tekken Dark Resurrection
+ case VC_NS: // c = !(my_isnan(s[i]) || my_isinf(s[i])); break;
+ DISABLE; // TODO: these shouldn't be that hard
+ break;
+
+ case VC_EN: // c = my_isnan(s[i]); break; // Tekken 6
+ case VC_NN: // c = !my_isnan(s[i]); break;
+ DISABLE; // TODO: these shouldn't be that hard
+ break;
+
+ case VC_EQ: // c = s[i] == t[i]
+ VCEQ(F_32, res, vs, vt);
+ break;
+
+ case VC_LT: // c = s[i] < t[i]
+ VCLT(F_32, res, vs, vt);
+ break;
+
+ case VC_LE: // c = s[i] <= t[i];
+ VCLE(F_32, res, vs, vt);
+ break;
+
+ case VC_NE: // c = s[i] != t[i]
+ VCEQ(F_32, res, vs, vt);
+ oneIsFalse = true;
+ break;
+
+ case VC_GE: // c = s[i] >= t[i]
+ VCGE(F_32, res, vs, vt);
+ break;
+
+ case VC_GT: // c = s[i] > t[i]
+ VCGT(F_32, res, vs, vt);
+ break;
+
+ case VC_EZ: // c = s[i] == 0.0f || s[i] == -0.0f
+ VCEQ(F_32, res, vs);
+ break;
+
+ case VC_NZ: // c = s[i] != 0
+ VCEQ(F_32, res, vs);
+ oneIsFalse = true;
+ break;
+
+ default:
+ DISABLE;
+ }
+ if (oneIsFalse) {
+ VMVN(res, res);
+ }
+ // Somehow collect the bits into a mask.
+
+ // Collect the bits. Where's my PMOVMSKB? :(
+ VLD1(I_32, Q0, R1, n < 2 ? 1 : 2);
+ VAND(Q0, Q0, res);
+ VPADD(I_32, Q0, Q0, Q0);
+ VPADD(I_32, D0, D0, D0);
+ // OK, bits now in S0.
+ VMOV(R0, S0);
+ // Zap irrelevant bits (V_Single, V_Triple)
+ AND(R0, R0, affected_bits);
+
+ // TODO: Now, how in the world do we generate the component OR and AND bits without burning tens of ALU instructions?? Lookup-table?
+
+ gpr.MapReg(MIPS_REG_VFPUCC, MAP_DIRTY);
+ BIC(gpr.R(MIPS_REG_VFPUCC), gpr.R(MIPS_REG_VFPUCC), affected_bits);
+ ORR(gpr.R(MIPS_REG_VFPUCC), gpr.R(MIPS_REG_VFPUCC), R0);
}
void Jit::CompNEON_Vcmov(MIPSOpcode op) {
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE;
+ }
+
DISABLE;
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ ARMReg vs = NEONMapPrefixS(_VS, sz, 0);
+ DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_DIRTY);
+ int tf = (op >> 19) & 1;
+ int imm3 = (op >> 16) & 7;
+
+ if (imm3 < 6) {
+ // Test one bit of CC. This bit decides whether none or all subregisters are copied.
+ gpr.MapReg(MIPS_REG_VFPUCC);
+ TST(gpr.R(MIPS_REG_VFPUCC), 1 << imm3);
+ FixupBranch skip = B_CC(CC_NEQ);
+ VMOV_neon(vd, vs);
+ SetJumpTarget(skip);
+ } else {
+ // Look at the bottom four bits of CC to individually decide if the subregisters should be copied.
+ // This is the nasty one! Need to expand those bits into a full NEON register somehow.
+ DISABLE;
+ /*
+ gpr.MapReg(MIPS_REG_VFPUCC);
+ for (int i = 0; i < n; i++) {
+ TST(gpr.R(MIPS_REG_VFPUCC), 1 << i);
+ SetCC(tf ? CC_EQ : CC_NEQ);
+ VMOV(fpr.V(dregs[i]), fpr.V(sregs[i]));
+ SetCC(CC_AL);
+ }
+ */
+ }
+
+ NEONApplyPrefixD(vd);
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Viim(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE;
+ }
+
+ DestARMReg vt = NEONMapPrefixD(_VT, V_Single, MAP_NOINIT | MAP_DIRTY);
+
+ s32 imm = (s32)(s16)(u16)(op & 0xFFFF);
+ // TODO: Optimize for low registers.
+ MOVI2F(S0, (float)imm, R0);
+ VMOV_neon(vt.rd, D0);
+
+ NEONApplyPrefixD(vt);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vfim(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE;
+ }
+
+ DestARMReg vt = NEONMapPrefixD(_VT, V_Single, MAP_NOINIT | MAP_DIRTY);
+
+ FP16 half;
+ half.u = op & 0xFFFF;
+ FP32 fval = half_to_float_fast5(half);
+ // TODO: Optimize for low registers.
+ MOVI2F(S0, (float)fval.f, R0);
+ VMOV_neon(vt.rd, D0);
+
+ NEONApplyPrefixD(vt);
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
+// https://code.google.com/p/bullet/source/browse/branches/PhysicsEffects/include/vecmath/neon/vectormath_neon_assembly_implementations.S?r=2488
void Jit::CompNEON_VCrossQuat(MIPSOpcode op) {
- DISABLE;
+ // This op does not support prefixes anyway.
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE_UNKNOWN_PREFIX;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ if (sz != V_Triple) {
+ // Quaternion product. Bleh.
+ DISABLE;
+ }
+
+ MappedRegs r = NEONMapDirtyInIn(op, sz, sz, sz, false);
+
+ ARMReg t1 = Q0;
+ ARMReg t2 = fpr.QAllocTemp(V_Triple);
+
+ // There has to be a faster way to do this. This is not really any better than
+ // scalar.
+
+ // d18, d19 (q9) = t1 = r.vt
+ // d16, d17 (q8) = t2 = r.vs
+ // d20, d21 (q10) = t
+ VMOV_neon(t1, r.vs);
+ VMOV_neon(t2, r.vt);
+ VTRN(F_32, D_0(t2), D_1(t2)); // vtrn.32 d18,d19 @ q9 = = d18,d19
+ VREV64(F_32, D_0(t1), D_0(t1)); // vrev64.32 d16,d16 @ q8 = = d16,d17
+ VREV64(F_32, D_0(t2), D_0(t2)); // vrev64.32 d18,d18 @ q9 = = d18,d19
+ VTRN(F_32, D_0(t1), D_1(t1)); // vtrn.32 d16,d17 @ q8 = = d16,d17
+ // perform first half of cross product using rearranged inputs
+ VMUL(F_32, r.vd, t1, t2); // vmul.f32 q10, q8, q9 @ q10 =
+ // @ rearrange inputs again
+ VTRN(F_32, D_0(t2), D_1(t2)); // vtrn.32 d18,d19 @ q9 = = d18,d19
+ VREV64(F_32, D_0(t1), D_0(t1)); // vrev64.32 d16,d16 @ q8 = = d16,d17
+ VREV64(F_32, D_0(t2), D_0(t2)); // vrev64.32 d18,d18 @ q9 = = d18,d19
+ VTRN(F_32, D_0(t1), D_1(t1)); // vtrn.32 d16,d17 @ q8 = = d16,d17
+ // @ perform last half of cross product using rearranged inputs
+ VMLS(F_32, r.vd, t1, t2); // vmls.f32 q10, q8, q9 @ q10 =
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_Vsgn(MIPSOpcode op) {
DISABLE;
+
+ // This will be a bunch of bit magic.
}
void Jit::CompNEON_Vocp(MIPSOpcode op) {
- DISABLE;
+ CONDITIONAL_DISABLE;
+ if (js.HasUnknownPrefix()) {
+ DISABLE;
+ }
+
+ VectorSize sz = GetVecSize(op);
+ int n = GetNumVectorElements(sz);
+
+ MappedRegs regs = NEONMapDirtyIn(op, sz, sz);
+ MOVI2F_neon(Q0, 1.0f, R0);
+ VSUB(F_32, regs.vd, Q0, regs.vs);
+ NEONApplyPrefixD(regs.vd);
+
+ fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Jit::CompNEON_ColorConv(MIPSOpcode op) {
diff --git a/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp
new file mode 100644
index 0000000000..45512df225
--- /dev/null
+++ b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp
@@ -0,0 +1,419 @@
+// Copyright (c) 2013- PPSSPP Project.
+
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU General Public License as published by
+// the Free Software Foundation, version 2.0 or later versions.
+
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU General Public License 2.0 for more details.
+
+// A copy of the GPL 2.0 should have been included with the program.
+// If not, see http://www.gnu.org/licenses/
+
+// Official git repository and contact information can be found at
+// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
+
+// NEON VFPU
+// This is where we will create an alternate implementation of the VFPU emulation
+// that uses NEON Q registers to cache pairs/tris/quads, and so on.
+// Will require major extensions to the reg cache and other things.
+
+// ARM NEON can only do pairs and quads, not tris and scalars.
+// We can do scalars, though, for many operations if all the operands
+// are below Q8 (D16, S32) using regular VFP instructions but really not sure
+// if it's worth it.
+
+#include
+
+#include "base/logging.h"
+#include "math/math_util.h"
+
+#include "Common/CPUDetect.h"
+#include "Core/MemMap.h"
+#include "Core/MIPS/MIPS.h"
+#include "Core/MIPS/MIPSAnalyst.h"
+#include "Core/MIPS/MIPSCodeUtils.h"
+#include "Core/MIPS/MIPSVFPUUtils.h"
+#include "Core/Config.h"
+#include "Core/Reporting.h"
+
+#include "Core/MIPS/ARM/ArmJit.h"
+#include "Core/MIPS/ARM/ArmRegCache.h"
+#include "Core/MIPS/ARM/ArmCompVFPUNEONUtil.h"
+
+// TODO: Somehow #ifdef away on ARMv5eabi, without breaking the linker.
+// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
+
+#define CONDITIONAL_DISABLE ;
+#define DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
+
+
+#define _RS MIPS_GET_RS(op)
+#define _RT MIPS_GET_RT(op)
+#define _RD MIPS_GET_RD(op)
+#define _FS MIPS_GET_FS(op)
+#define _FT MIPS_GET_FT(op)
+#define _FD MIPS_GET_FD(op)
+#define _SA MIPS_GET_SA(op)
+#define _POS ((op>> 6) & 0x1F)
+#define _SIZE ((op>>11) & 0x1F)
+#define _IMM16 (signed short)(op & 0xFFFF)
+#define _IMM26 (op & 0x03FFFFFF)
+
+namespace MIPSComp {
+
+static const float minus_one = -1.0f;
+static const float one = 1.0f;
+static const float zero = 0.0f;
+
+// On NEON, we map triples to Q registers and singles to D registers.
+// Sometimes, as when doing dot products, it matters what's in that unused reg. This zeroes it.
+void Jit::NEONMaskToSize(ARMReg vs, VectorSize sz) {
+ // TODO
+}
+
+ARMReg Jit::NEONMapPrefixST(int mipsReg, VectorSize sz, u32 prefix, int mapFlags) {
+ static const float constantArray[8] = { 0.f, 1.f, 2.f, 0.5f, 3.f, 1.f / 3.f, 0.25f, 1.f / 6.f };
+ static const float constantArrayNegated[8] = { -0.f, -1.f, -2.f, -0.5f, -3.f, -1.f / 3.f, -0.25f, -1.f / 6.f };
+
+ // Applying prefixes in SIMD fashion will actually be a lot easier than the old style.
+ if (prefix == 0xE4) {
+ return fpr.QMapReg(mipsReg, sz, mapFlags);
+ }
+
+ int n = GetNumVectorElements(sz);
+
+ int regnum[4] = { -1, -1, -1, -1 };
+ int abs[4] = { 0 };
+ int negate[4] = { 0 };
+ int constants[4] = { 0 };
+ int constNum[4] = { 0 };
+
+ int full_mask = (1 << n) - 1;
+
+ int abs_mask = (prefix >> 8) & full_mask;
+ int negate_mask = (prefix >> 16) & full_mask;
+ int constants_mask = (prefix >> 12) & full_mask;
+
+ // Decode prefix to keep the rest readable
+ int permuteMask = 0;
+ for (int i = 0; i < n; i++) {
+ permuteMask |= 3 << (i * 2);
+ regnum[i] = (prefix >> (i * 2)) & 3;
+ abs[i] = (prefix >> (8 + i)) & 1;
+ negate[i] = (prefix >> (16 + i)) & 1;
+ constants[i] = (prefix >> (12 + i)) & 1;
+
+ if (constants[i]) {
+ constNum[i] = regnum[i] + (abs[i] << 2);
+ abs[i] = 0;
+ }
+ }
+ abs_mask &= ~constants_mask;
+
+ bool anyPermute = (prefix & permuteMask) != (0xE4 & permuteMask);
+
+ if (constants_mask == full_mask) {
+ // It's all constants! Don't even bother mapping the input register,
+ // just allocate a temp one.
+ // If a single, this can sometimes be done cheaper. But meh.
+ ARMReg ar = fpr.QAllocTemp(sz);
+ for (int i = 0; i < n; i++) {
+ if ((i & 1) == 0) {
+ if (constNum[i] == constNum[i + 1]) {
+ // Replace two loads with a single immediate when easily possible.
+ ARMReg dest = i & 2 ? D_1(ar) : D_0(ar);
+ switch (constNum[i]) {
+ case 0:
+ case 1:
+ {
+ float c = constantArray[constNum[i]];
+ VMOV_immf(dest, negate[i] ? -c : c);
+ }
+ break;
+ // TODO: There are a few more that are doable.
+ default:
+ goto skip;
+ }
+
+ i++;
+ continue;
+ skip:
+ ;
+ }
+ }
+ MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]);
+ VLD1_lane(F_32, ar, R0, i, true);
+ }
+ return ar;
+ }
+
+ // 1. Permute.
+ // 2. Abs
+ // If any constants:
+ // 3. Replace values with constants
+ // 4. Negate
+
+ ARMReg inputAR = fpr.QMapReg(mipsReg, sz, mapFlags);
+ ARMReg ar = fpr.QAllocTemp(sz);
+
+ if (!anyPermute) {
+ VMOV(ar, inputAR);
+ // No permutations!
+ } else {
+ bool allSame = false;
+ for (int i = 1; i < n; i++) {
+ if (regnum[0] == regnum[i])
+ allSame = true;
+ }
+
+ if (allSame) {
+ // Easy, someone is duplicating one value onto all the reg parts.
+ // If this is happening and QMapReg must load, we can combine these two actions
+ // into a VLD1_lane. TODO
+ VDUP(F_32, ar, inputAR, regnum[0]);
+ } else {
+ // Do some special cases
+ if (regnum[0] == 1 && regnum[1] == 0) {
+ INFO_LOG(HLE, "PREFIXST: Bottom swap!");
+ VREV64(I_32, ar, inputAR);
+ regnum[0] = 0;
+ regnum[1] = 1;
+ }
+
+ // TODO: Make a generic fallback using another temp register
+
+ bool match = true;
+ for (int i = 0; i < n; i++) {
+ if (regnum[i] != i)
+ match = false;
+ }
+
+ // TODO: Cannot do this permutation yet!
+ if (!match) {
+ ERROR_LOG(HLE, "PREFIXST: Unsupported permute! %i %i %i %i / %i", regnum[0], regnum[1], regnum[2], regnum[3], n);
+ VMOV(ar, inputAR);
+ }
+ }
+ }
+
+ // ABS
+ // Two methods: If all lanes are "absoluted", it's easy.
+ if (abs_mask == full_mask) {
+ // TODO: elide the above VMOV (in !anyPermute) when possible
+ VABS(F_32, ar, ar);
+ } else if (abs_mask != 0) {
+ // Partial ABS!
+ if (abs_mask == 3) {
+ VABS(F_32, D_0(ar), D_0(ar));
+ } else {
+ // Horrifying fallback: Mov to Q0, abs, move back.
+ // TODO: Optimize for lower quads where we don't need to move.
+ VMOV(MatchSize(Q0, ar), ar);
+ for (int i = 0; i < n; i++) {
+ if (abs_mask & (1 << i)) {
+ VABS((ARMReg)(S0 + i), (ARMReg)(S0 + i));
+ }
+ }
+ VMOV(ar, MatchSize(Q0, ar));
+ INFO_LOG(HLE, "PREFIXST: Partial ABS %i/%i! Slow fallback generated.", abs_mask, full_mask);
+ }
+ }
+
+ if (negate_mask == full_mask) {
+ // TODO: elide the above VMOV when possible
+ VNEG(F_32, ar, ar);
+ } else if (negate_mask != 0) {
+ // Partial negate! I guess we build sign bits in another register
+ // and simply XOR.
+ if (negate_mask == 3) {
+ VNEG(F_32, D_0(ar), D_0(ar));
+ } else {
+ // Horrifying fallback: Mov to Q0, negate, move back.
+ // TODO: Optimize for lower quads where we don't need to move.
+ VMOV(MatchSize(Q0, ar), ar);
+ for (int i = 0; i < n; i++) {
+ if (negate_mask & (1 << i)) {
+ VNEG((ARMReg)(S0 + i), (ARMReg)(S0 + i));
+ }
+ }
+ VMOV(ar, MatchSize(Q0, ar));
+ INFO_LOG(HLE, "PREFIXST: Partial Negate %i/%i! Slow fallback generated.", negate_mask, full_mask);
+ }
+ }
+
+ // Insert constants where requested, and check negate!
+ for (int i = 0; i < n; i++) {
+ if (constants[i]) {
+ MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]);
+ VLD1_lane(F_32, ar, R0, i, true);
+ }
+ }
+
+ return ar;
+}
+
+Jit::DestARMReg Jit::NEONMapPrefixD(int vreg, VectorSize sz, int mapFlags) {
+ // Inverted from the actual bits, easier to reason about 1 == write
+ int writeMask = (~(js.prefixD >> 8)) & 0xF;
+ int n = GetNumVectorElements(sz);
+ int full_mask = (1 << n) - 1;
+
+ DestARMReg dest;
+ dest.sz = sz;
+ if ((writeMask & full_mask) == full_mask) {
+ // No need to apply a write mask.
+ // Let's not make things complicated.
+ dest.rd = fpr.QMapReg(vreg, sz, mapFlags);
+ dest.backingRd = dest.rd;
+ } else {
+ // Allocate a temporary register.
+ ELOG("PREFIXD: Write mask allocated! %i/%i", writeMask, full_mask);
+ dest.rd = fpr.QAllocTemp(sz);
+ dest.backingRd = fpr.QMapReg(vreg, sz, mapFlags & ~MAP_NOINIT); // Force initialization of the backing reg.
+ }
+ return dest;
+}
+
+void Jit::NEONApplyPrefixD(DestARMReg dest) {
+ // Apply clamps to dest.rd
+ int n = GetNumVectorElements(dest.sz);
+
+ int sat1_mask = 0;
+ int sat3_mask = 0;
+ int full_mask = (1 << n) - 1;
+ for (int i = 0; i < n; i++) {
+ int sat = (js.prefixD >> (i * 2)) & 3;
+ if (sat == 1)
+ sat1_mask |= 1 << i;
+ if (sat == 3)
+ sat3_mask |= 1 << i;
+ }
+
+ if (sat1_mask && sat3_mask) {
+ // Why would anyone do this?
+ ELOG("PREFIXD: Can't have both sat[0-1] and sat[-1-1] at the same time yet");
+ }
+
+ if (sat1_mask) {
+ if (sat1_mask != full_mask) {
+ ELOG("PREFIXD: Can't have partial sat1 mask yet (%i vs %i)", sat1_mask, full_mask);
+ }
+ if (IsD(dest.rd)) {
+ VMOV_immf(D0, 0.0);
+ VMOV_immf(D1, 1.0);
+ VMAX(F_32, dest.rd, dest.rd, D0);
+ VMIN(F_32, dest.rd, dest.rd, D1);
+ } else {
+ VMOV_immf(Q0, 1.0);
+ VMIN(F_32, dest.rd, dest.rd, Q0);
+ VMOV_immf(Q0, 0.0);
+ VMAX(F_32, dest.rd, dest.rd, Q0);
+ }
+ }
+
+ if (sat3_mask && sat1_mask != full_mask) {
+ if (sat3_mask != full_mask) {
+ ELOG("PREFIXD: Can't have partial sat3 mask yet (%i vs %i)", sat3_mask, full_mask);
+ }
+ if (IsD(dest.rd)) {
+ VMOV_immf(D0, 0.0);
+ VMOV_immf(D1, 1.0);
+ VMAX(F_32, dest.rd, dest.rd, D0);
+ VMIN(F_32, dest.rd, dest.rd, D1);
+ } else {
+ VMOV_immf(Q0, 1.0);
+ VMIN(F_32, dest.rd, dest.rd, Q0);
+ VMOV_immf(Q0, -1.0);
+ VMAX(F_32, dest.rd, dest.rd, Q0);
+ }
+ }
+
+ // Check for actual mask operation (unrelated to the "masks" above).
+ if (dest.backingRd != dest.rd) {
+ // This means that we need to apply the write mask, from rd to backingRd.
+ // What a pain. We can at least shortcut easy cases like half the register.
+ // And we can generate the masks easily with some of the crazy vector imm modes. (bits2bytes for example).
+ // So no need to load them from RAM.
+ int writeMask = (~(js.prefixD >> 8)) & 0xF;
+
+ if (writeMask == 3) {
+ ILOG("Doing writemask = 3");
+ VMOV(D_0(dest.rd), D_0(dest.backingRd));
+ } else {
+ // TODO
+ ELOG("PREFIXD: Arbitrary write masks not supported (%i / %i)", writeMask, full_mask);
+ VMOV(dest.backingRd, dest.rd);
+ }
+ }
+}
+
+Jit::MappedRegs Jit::NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes) {
+ MappedRegs regs;
+ if (applyPrefixes) {
+ regs.vs = NEONMapPrefixS(_VS, ssize, 0);
+ regs.vt = NEONMapPrefixT(_VT, tsize, 0);
+ } else {
+ regs.vs = fpr.QMapReg(_VS, ssize, 0);
+ regs.vt = fpr.QMapReg(_VT, ssize, 0);
+ }
+
+ regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0 || GetVectorOverlap(_VD, dsize, _VT, ssize);
+ if (applyPrefixes) {
+ regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
+ } else {
+ regs.vd.rd = fpr.QMapReg(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
+ regs.vd.backingRd = regs.vd.rd;
+ regs.vd.sz = dsize;
+ }
+ return regs;
+}
+
+Jit::MappedRegs Jit::NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes) {
+ MappedRegs regs;
+ if (applyPrefixes) {
+ regs.vs = NEONMapPrefixS(_VS, ssize, 0);
+ regs.vt = NEONMapPrefixT(_VT, tsize, 0);
+ } else {
+ regs.vs = fpr.QMapReg(_VS, ssize, 0);
+ regs.vt = fpr.QMapReg(_VT, ssize, 0);
+ }
+ regs.vd.rd = INVALID_REG;
+ regs.vd.sz = V_Invalid;
+ return regs;
+}
+
+Jit::MappedRegs Jit::NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes) {
+ MappedRegs regs;
+ regs.vs = NEONMapPrefixS(_VS, ssize, 0);
+ regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0;
+ regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
+ return regs;
+}
+
+// Requires quad registers.
+void Jit::NEONTranspose4x4(ARMReg cols[4]) {
+ // 0123 _\ 0426
+ // 4567 / 1537
+ VTRN(F_32, cols[0], cols[1]);
+
+ // 89ab _\ 8cae
+ // cdef / 9dbf
+ VTRN(F_32, cols[2], cols[3]);
+
+ // 04[26] 048c
+ // 15 37 -> 1537
+ // [8c]ae 26ae
+ // 9d bf 9dbf
+ VSWP(D_1(cols[0]), D_0(cols[2]));
+
+ // 04 8c 048c
+ // 15[37] -> 159d
+ // 26 ae 26ae
+ // [9d]bf 37bf
+ VSWP(D_1(cols[1]), D_0(cols[3]));
+}
+
+} // namespace MIPSComp
\ No newline at end of file
diff --git a/Core/MIPS/ARM/ArmCompVFPUNEONUtil.h b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.h
new file mode 100644
index 0000000000..3d278e064b
--- /dev/null
+++ b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.h
@@ -0,0 +1,20 @@
+#pragma once
+
+#include "Core/MIPS/ARM/ArmJit.h"
+#include "Core/MIPS/ARM/ArmRegCache.h"
+
+namespace MIPSComp {
+
+
+inline ARMReg MatchSize(ARMReg x, ARMReg target) {
+ if (IsQ(target) && IsQ(x))
+ return x;
+ if (IsD(target) && IsD(x))
+ return x;
+ if (IsD(target) && IsQ(x))
+ return D_0(x);
+ // if (IsQ(target) && IsD(x))
+ return (ARMReg)(D0 + (x - Q0) * 2);
+}
+
+}
\ No newline at end of file
diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp
index 48bd4a7082..adde925f9f 100644
--- a/Core/MIPS/ARM/ArmJit.cpp
+++ b/Core/MIPS/ARM/ArmJit.cpp
@@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() {
useNEONVFPU = false;
}
-Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips)
+Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips)
{
logBlocks = 0;
dontLogBlocks = 0;
@@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b)
while (js.compiling)
{
gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages
- fpr.SetCompilerPC(js.compilerPC);
MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC);
js.downcountAmount += MIPSGetInstructionCycleEstimate(inst);
diff --git a/Core/MIPS/ARM/ArmJit.h b/Core/MIPS/ARM/ArmJit.h
index a086550690..9579f49c09 100644
--- a/Core/MIPS/ARM/ArmJit.h
+++ b/Core/MIPS/ARM/ArmJit.h
@@ -17,6 +17,7 @@
#pragma once
+#include "Common/CPUDetect.h"
#include "Core/MIPS/JitCommon/JitState.h"
#include "Core/MIPS/JitCommon/JitBlockCache.h"
#include "Core/MIPS/ARM/ArmRegCache.h"
@@ -240,6 +241,43 @@ private:
}
void GetVectorRegsPrefixD(u8 *regs, VectorSize sz, int vectorReg);
+
+ // For NEON mappings, it will be easier to deal directly in ARM registers.
+
+ ARMReg NEONMapPrefixST(int vfpuReg, VectorSize sz, u32 prefix, int mapFlags);
+ ARMReg NEONMapPrefixS(int vfpuReg, VectorSize sz, int mapFlags) {
+ return NEONMapPrefixST(vfpuReg, sz, js.prefixS, mapFlags);
+ }
+ ARMReg NEONMapPrefixT(int vfpuReg, VectorSize sz, int mapFlags) {
+ return NEONMapPrefixST(vfpuReg, sz, js.prefixT, mapFlags);
+ }
+
+ struct DestARMReg {
+ ARMReg rd;
+ ARMReg backingRd;
+ VectorSize sz;
+
+ operator ARMReg() const { return rd; }
+ };
+
+ struct MappedRegs {
+ ARMReg vs;
+ ARMReg vt;
+ DestARMReg vd;
+ bool overlap;
+ };
+
+ MappedRegs NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true);
+ MappedRegs NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true);
+ MappedRegs NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes = true);
+
+ DestARMReg NEONMapPrefixD(int vfpuReg, VectorSize sz, int mapFlags);
+ void NEONApplyPrefixD(DestARMReg dest);
+
+ // NEON utils
+ void NEONMaskToSize(ARMReg vs, VectorSize sz);
+ void NEONTranspose4x4(ARMReg cols[4]);
+
// Utils
void SetR0ToEffectiveAddress(MIPSGPReg rs, s16 offset);
void SetCCAndR0ForSafeAddress(MIPSGPReg rs, s16 offset, ARMReg tempReg, bool reverse = false);
diff --git a/Core/MIPS/ARM/ArmRegCache.cpp b/Core/MIPS/ARM/ArmRegCache.cpp
index faf8945b2b..6a1be703e6 100644
--- a/Core/MIPS/ARM/ArmRegCache.cpp
+++ b/Core/MIPS/ARM/ArmRegCache.cpp
@@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) {
return mr[mipsReg].loc == ML_ARMREG_AS_PTR;
}
+bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) {
+ return mr[mipsReg].loc == ML_ARMREG;
+}
+
void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) {
// If we can do it with a simple Operand2, let's do that.
Operand2 op2;
diff --git a/Core/MIPS/ARM/ArmRegCache.h b/Core/MIPS/ARM/ArmRegCache.h
index 9e62de5295..cdc33e3b0d 100644
--- a/Core/MIPS/ARM/ArmRegCache.h
+++ b/Core/MIPS/ARM/ArmRegCache.h
@@ -103,6 +103,7 @@ public:
ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0);
ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty.
+ bool IsMapped(MIPSGPReg reg);
bool IsMappedAsPointer(MIPSGPReg reg);
void MapInIn(MIPSGPReg rd, MIPSGPReg rs);
diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.cpp b/Core/MIPS/ARM/ArmRegCacheFPU.cpp
index 9c390ea491..8628b484de 100644
--- a/Core/MIPS/ARM/ArmRegCacheFPU.cpp
+++ b/Core/MIPS/ARM/ArmRegCacheFPU.cpp
@@ -16,14 +16,17 @@
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
#include
+
#include "base/logging.h"
#include "Common/CPUDetect.h"
+#include "Core/MIPS/MIPS.h"
#include "Core/MIPS/ARM/ArmRegCacheFPU.h"
+#include "Core/MIPS/ARM/ArmJit.h"
#include "Core/MIPS/MIPSTables.h"
using namespace ArmGen;
-ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) {
+ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) {
if (cpu_info.bNEON) {
numARMFpuReg_ = 32;
} else {
@@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init
}
}
-void ArmRegCacheFPU::Init(ARMXEmitter *emitter) {
- emit_ = emitter;
-}
-
void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) {
if (!initialReady) {
SetupInitialRegs();
@@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() {
mrInitial[i].spillLock = false;
mrInitial[i].tempLock = false;
}
+ for (int i = 0; i < MAX_ARMQUADS; i++) {
+ qr[i].isDirty = false;
+ qr[i].mipsVec = -1;
+ qr[i].sz = V_Invalid;
+ qr[i].spillLock = false;
+ qr[i].isTemp = false;
+ memset(qr[i].vregs, 0xff, 4);
+ }
}
-static const ARMReg *GetMIPSAllocationOrder(int &count) {
+const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) {
// We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things.
static const ARMReg allocationOrder[] = {
S2, S3,
@@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
S12, S13, S14, S15
};
- // With NEON, we have many more.
- // In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU.
- // VFPU will use NEON to do SIMD and it will be awkward to mix with FPU.
+ // VFP mapping
+ // VFPU registers and regular FP registers are mapped interchangably on top of the standard
+ // 16 FPU registers.
+ // NEON mapping
+ // We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers.
+ // VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON
+ // (or D16-D31 as doubles, but not relevant).
+ // Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads.
+
// We should attempt to map scalars to low Q registers and wider things to high registers,
// as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be
// able to use regular VFP instructions too.
@@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
S4, S5, S6, S7, // Q1
S8, S9, S10, S11, // Q2
S12, S13, S14, S15, // Q3
- S16, S17, S18, S19, // Q4
- S20, S21, S22, S23, // Q5
- S24, S25, S26, S27, // Q6
- S28, S29, S30, S31, // Q7
- // Q8-Q15 free for NEON tricks
+ // Q4-Q15 free for VFPU
};
- if (cpu_info.bNEON) {
+ if (jo_->useNEONVFPU) {
count = sizeof(allocationOrderNEON) / sizeof(const int);
return allocationOrderNEON;
} else {
@@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
}
}
+bool ArmRegCacheFPU::IsMapped(MIPSReg r) {
+ return mr[r].loc == ML_ARMREG;
+}
+
ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) {
+ // INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags);
+ if (jo_->useNEONVFPU && mipsReg >= 32) {
+ ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC);
+ return S0;
+ }
+
pendingFlush = true;
// Let's see if it's already mapped. If so we just need to update the dirty flag.
// We don't need to check for ML_NOINIT because we assume that anyone who maps
@@ -157,7 +176,7 @@ allocate:
}
// Uh oh, we have all them spilllocked....
- ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc);
+ ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC);
return INVALID_REG;
}
@@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
}
void ArmRegCacheFPU::FlushArmReg(ARMReg r) {
- int reg = r - S0;
- if (ar[reg].mipsReg == -1) {
- // Nothing to do, reg not mapped.
- return;
- }
- if (ar[reg].mipsReg != -1) {
- if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) {
- //INFO_LOG(JIT, "Flushing ARM reg %i", reg);
- emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
+ if (r >= S0 && r <= S31) {
+ int reg = r - S0;
+ if (ar[reg].mipsReg == -1) {
+ // Nothing to do, reg not mapped.
+ return;
}
- // IMMs won't be in an ARM reg.
- mr[ar[reg].mipsReg].loc = ML_MEM;
- mr[ar[reg].mipsReg].reg = INVALID_REG;
- } else {
- ERROR_LOG(JIT, "Dirty but no mipsreg?");
+ if (ar[reg].mipsReg != -1) {
+ if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG)
+ {
+ //INFO_LOG(JIT, "Flushing ARM reg %i", reg);
+ emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
+ }
+ // IMMs won't be in an ARM reg.
+ mr[ar[reg].mipsReg].loc = ML_MEM;
+ mr[ar[reg].mipsReg].reg = INVALID_REG;
+ } else {
+ ERROR_LOG(JIT, "Dirty but no mipsreg?");
+ }
+ ar[reg].isDirty = false;
+ ar[reg].mipsReg = -1;
+ } else if (r >= D0 && r <= D31) {
+ // TODO: Convert to S regs and flush them individually.
+ } else if (r >= Q0 && r <= Q15) {
+ int quad = r - Q0;
+ QFlush(r);
}
- ar[reg].isDirty = false;
- ar[reg].mipsReg = -1;
}
+void ArmRegCacheFPU::FlushV(MIPSReg r) {
+ FlushR(r + 32);
+}
+
+/*
+void ArmRegCacheFPU::FlushQWithV(MIPSReg r) {
+ // Look for it in all the quads. If it's in any, flush that quad clean.
+ int flushCount = 0;
+ for (int i = 0; i < MAX_ARMQUADS; i++) {
+ if (qr[i].sz == V_Invalid)
+ continue;
+
+ int n = qr[i].sz;
+ bool flushThis = false;
+ for (int j = 0; j < n; j++) {
+ if (qr[i].vregs[j] == r) {
+ flushThis = true;
+ }
+ }
+
+ if (flushThis) {
+ QFlush(i);
+ flushCount++;
+ }
+ }
+
+ if (flushCount > 1) {
+ WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r);
+ }
+}
+*/
+
void ArmRegCacheFPU::FlushR(MIPSReg r) {
switch (mr[r].loc) {
case ML_IMM:
@@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) {
if (mr[r].reg == (int)INVALID_REG) {
ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg");
}
- if (ar[mr[r].reg].isDirty) {
- //INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
- emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
- ar[mr[r].reg].isDirty = false;
+
+ if (mr[r].reg >= Q0 && mr[r].reg <= Q15) {
+ // This should happen rarely, but occasionally we need to flush a single stray
+ // mipsreg that's been part of a quad.
+ int quad = mr[r].reg - Q0;
+ if (qr[quad].isDirty) {
+ WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1);
+ emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true);
+ }
+ } else {
+ if (ar[mr[r].reg].isDirty) {
+ //INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
+ emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
+ ar[mr[r].reg].isDirty = false;
+ }
+ ar[mr[r].reg].mipsReg = -1;
}
- ar[mr[r].reg].mipsReg = -1;
break;
case ML_MEM:
@@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() {
return 16;
}
+// Scalar only. Need a similar one for sequential Q vectors.
int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) {
int c = 1;
int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg);
@@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() {
DiscardR(i);
}
+ // Flush quads!
+ // These could also use sequential detection.
+ for (int i = 4; i < MAX_ARMQUADS; i++) {
+ QFlush(i);
+ }
+
// Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are
// sequential. This is necessary because we store VFPU registers in a staggered order to get
// columns sequential (most VFPU math in nearly all games is in columns, not rows).
@@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const {
}
int ArmRegCacheFPU::GetTempR() {
+ if (jo_->useNEONVFPU) {
+ ERROR_LOG(JIT, "VFP temps not allowed in NEON mode");
+ return 0;
+ }
pendingFlush = true;
for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) {
if (mr[r].loc == ML_MEM && !mr[r].tempLock) {
@@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) {
// This is actually pretty slow with all the 160 regs...
void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() {
- for (int i = 0; i < NUM_MIPSFPUREG; i++)
+ for (int i = 0; i < NUM_MIPSFPUREG; i++) {
mr[i].spillLock = false;
- for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i)
+ }
+ for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) {
DiscardR(i);
+ }
+ for (int i = 0; i < MAX_ARMQUADS; i++) {
+ qr[i].spillLock = false;
+ if (qr[i].isTemp) {
+ qr[i].isTemp = false;
+ qr[i].sz = V_Invalid;
+ }
+ }
}
ARMReg ArmRegCacheFPU::R(int mipsReg) {
@@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) {
return (ARMReg)(mr[mipsReg].reg + S0);
} else {
if (mipsReg < 32) {
- ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_));
+ ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
} else if (mipsReg < 32 + 128) {
- ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
+ ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
} else {
- ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
+ ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
}
return INVALID_REG; // BAAAD
}
}
+
+inline ARMReg QuadAsD(int quad) {
+ return (ARMReg)(D0 + quad * 2);
+}
+
+inline ARMReg QuadAsQ(int quad) {
+ return (ARMReg)(Q0 + quad);
+}
+
+bool MappableQ(int quad) {
+ return quad >= 4;
+}
+
+void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) {
+ ERROR_LOG(JIT, "QLoad4x4 not implemented");
+ // TODO
+}
+
+void ArmRegCacheFPU::QFlush(int quad) {
+ if (!MappableQ(quad)) {
+ ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad);
+ return;
+ }
+
+ if (qr[quad].isDirty && !qr[quad].isTemp) {
+ INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz));
+
+ ARMReg q = QuadAsQ(quad);
+ // Unlike reads, when writing to the register file we need to be careful to write the correct
+ // number of floats.
+
+ switch (qr[quad].sz) {
+ case V_Single:
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1_lane(F_32, q, R0, 0, true);
+ // WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC);
+ break;
+ case V_Pair:
+ if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) {
+ // Can combine, it's a column!
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
+ } else {
+ // WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1_lane(F_32, q, R0, 0, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
+ emit_->VST1_lane(F_32, q, R0, 1, true);
+ }
+ break;
+ case V_Triple:
+ if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
+ emit_->VST1_lane(F_32, q, R0, 2, true);
+ } else {
+ // WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1_lane(F_32, q, R0, 0, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
+ emit_->VST1_lane(F_32, q, R0, 1, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
+ emit_->VST1_lane(F_32, q, R0, 2, true);
+ }
+ break;
+ case V_Quad:
+ if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
+ } else {
+ // WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
+ emit_->VST1_lane(F_32, q, R0, 0, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
+ emit_->VST1_lane(F_32, q, R0, 1, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
+ emit_->VST1_lane(F_32, q, R0, 2, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1);
+ emit_->VST1_lane(F_32, q, R0, 3, true);
+ }
+ break;
+ default:
+ ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz);
+ break;
+ }
+
+ qr[quad].isDirty = false;
+
+ int n = GetNumVectorElements(qr[quad].sz);
+ for (int i = 0; i < n; i++) {
+ int vr = qr[quad].vregs[i];
+ if (vr < 0 || vr > 128) {
+ ERROR_LOG(JIT, "Bad vr %i", vr);
+ }
+ FPURegMIPS &m = mr[32 + vr];
+ m.loc = ML_MEM;
+ m.lane = -1;
+ m.reg = -1;
+ }
+
+ } else {
+ if (qr[quad].isTemp) {
+ WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp);
+ }
+ }
+
+ qr[quad].isTemp = false;
+ qr[quad].mipsVec = -1;
+ qr[quad].sz = V_Invalid;
+ memset(qr[quad].vregs, 0xFF, 4);
+}
+
+int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) {
+ // Search for a free quad. A quad is free if the first register in it is free.
+ int quad = -1;
+
+ for (int i = 0; i < count; i++) {
+ int q = (i + start) & 15;
+
+ if (!MappableQ(q))
+ continue;
+
+ // Don't steal temp quads!
+ if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) {
+ // INFO_LOG(JIT, "Free quad: %i", q);
+ // Oh yeah! Free quad!
+ return q;
+ }
+ }
+
+ // Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some
+ // sort of age.
+ int bestQuad = -1;
+ int bestScore = -1;
+ for (int i = 0; i < count; i++) {
+ int q = (i + start) & 15;
+
+ if (!MappableQ(q))
+ continue;
+ if (qr[q].spillLock)
+ continue;
+ if (qr[q].isTemp)
+ continue;
+
+ int score = 0;
+ if (!qr[q].isDirty) {
+ score += 5;
+ }
+
+ if (score > bestScore) {
+ bestQuad = q;
+ bestScore = score;
+ }
+ }
+
+ if (bestQuad == -1) {
+ ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!");
+ return -1;
+ } else {
+ INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason");
+ QFlush(bestQuad);
+ return bestQuad;
+ }
+}
+
+ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) {
+ int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps
+ if (q < 0) {
+ ERROR_LOG(JIT, "Failed to allocate temp quad");
+ q = 0;
+ }
+ qr[q].spillLock = true;
+ qr[q].isTemp = true;
+ qr[q].sz = sz;
+ qr[q].isDirty = false; // doesn't matter
+
+ INFO_LOG(JIT, "Allocated temp quad %i", q);
+
+ if (sz == V_Single || sz == V_Pair) {
+ return D_0(ARMReg(Q0 + q));
+ } else {
+ return ARMReg(Q0 + q);
+ }
+}
+
+bool ArmRegCacheFPU::Consecutive(int v1, int v2) const {
+ return (voffset[v1] + 1) == voffset[v2];
+}
+
+bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const {
+ return Consecutive(v1, v2) && Consecutive(v2, v3);
+}
+
+bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const {
+ return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4);
+}
+
+void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) {
+ u8 vregs[4];
+ if (flags & MAP_MTX_TRANSPOSED) {
+ GetMatrixRows(matrix, mz, vregs);
+ } else {
+ GetMatrixColumns(matrix, mz, vregs);
+ }
+
+ // TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load.
+ int n = GetMatrixSide(mz);
+ VectorSize vsz = GetVectorSize(mz);
+ for (int i = 0; i < n; i++) {
+ regs[i] = QMapReg(vregs[i], vsz, flags);
+ }
+}
+
+ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) {
+ qTime_++;
+
+ int n = GetNumVectorElements(sz);
+ u8 vregs[4];
+ GetVectorRegs(vregs, sz, vreg);
+
+ // Range of registers to consider
+ int start = 0;
+ int count = 16;
+
+ if (flags & MAP_PREFER_HIGH) {
+ start = 8;
+ } else if (flags & MAP_PREFER_LOW) {
+ start = 4;
+ } else if (flags & MAP_FORCE_LOW) {
+ start = 4;
+ count = 4;
+ } else if (flags & MAP_FORCE_HIGH) {
+ start = 8;
+ count = 8;
+ }
+
+ // Let's check if they are all mapped in a quad somewhere.
+ // At the same time, check for the quad already being mapped.
+ // Later we can check for possible transposes as well.
+
+ // First just loop over all registers. If it's here and not in range, or overlapped, kick.
+ std::vector quadsToFlush;
+ for (int i = 0; i < 16; i++) {
+ int q = (i + start) & 15;
+ if (!MappableQ(q))
+ continue;
+
+ // Skip unmapped quads.
+ if (qr[q].sz == V_Invalid)
+ continue;
+
+ // Check if completely there already. If so, set spill-lock, transfer dirty flag and exit.
+ if (vreg == qr[q].mipsVec && sz == qr[q].sz) {
+ if (i < count) {
+ INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz);
+ qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY);
+ qr[q].spillLock = true;
+
+ // Sanity check vregs
+ for (int i = 0; i < n; i++) {
+ if (vregs[i] != qr[q].vregs[i]) {
+ ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]);
+ }
+ }
+
+ return (ARMReg)(Q0 + q);
+ } else {
+ INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count);
+ quadsToFlush.push_back(q);
+ continue;
+ }
+ }
+
+ // Check for any overlap. Overlap == flush.
+ int origN = GetNumVectorElements(qr[q].sz);
+ for (int a = 0; a < n; a++) {
+ for (int b = 0; b < origN; b++) {
+ if (vregs[a] == qr[q].vregs[b]) {
+ quadsToFlush.push_back(q);
+ goto doubleBreak;
+ }
+ }
+ }
+ doubleBreak:
+ ;
+ }
+
+ // We didn't find the extra register, but we got a list of regs to flush. Flush 'em.
+ // Here we can check for opportunities to do a "transpose-flush" of row vectors, etc.
+ if (!quadsToFlush.empty()) {
+ ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size());
+ }
+ for (size_t i = 0; i < quadsToFlush.size(); i++) {
+ QFlush(quadsToFlush[i]);
+ }
+
+ // Find where we want to map it, obeying the constraints we gave.
+ int quad = QGetFreeQuad(start, count, "mapping");
+
+ // If parts of our register are elsewhere, and we are dirty, we need to flush them
+ // before we reload in a new location.
+ // This may be problematic if inputs overlap irregularly with output, say:
+ // vdot S700, R000, C000
+ // It might still work by accident...
+ if (flags & MAP_DIRTY) {
+ for (int i = 0; i < n; i++) {
+ FlushV(vregs[i]);
+ }
+ }
+
+ qr[quad].sz = sz;
+ qr[quad].mipsVec = vreg;
+
+ if (!(flags & MAP_NOINIT)) {
+ // Okay, now we will try to load the whole thing in one go. This is possible
+ // if it's a row and easy if it's a single.
+ // Rows are rare, columns are common - but thanks to our register reordering,
+ // columns are actually in-order in memory.
+ switch (sz) {
+ case V_Single:
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
+ break;
+ case V_Pair:
+ if (Consecutive(vregs[0], vregs[1])) {
+ // Can combine, it's a column!
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
+ } else {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
+ }
+ break;
+ case V_Triple:
+ if (Consecutive(vregs[0], vregs[1], vregs[2])) {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
+ } else {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
+ }
+ break;
+ case V_Quad:
+ if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
+ } else {
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
+ emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1);
+ emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true);
+ }
+ break;
+ default:
+ ;
+ }
+ }
+
+ // OK, let's fill out the arrays to confirm that we have grabbed these registers.
+ for (int i = 0; i < n; i++) {
+ int mipsReg = 32 + vregs[i];
+ mr[mipsReg].loc = ML_ARMREG;
+ mr[mipsReg].reg = QuadAsQ(quad);
+ mr[mipsReg].lane = i;
+ qr[quad].vregs[i] = vregs[i];
+ }
+ qr[quad].isDirty = (flags & MAP_DIRTY) != 0;
+ qr[quad].spillLock = true;
+
+ INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty);
+ if (sz == V_Single || sz == V_Pair) {
+ return D_0(QuadAsQ(quad));
+ } else {
+ return QuadAsQ(quad);
+ }
+}
+
diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.h b/Core/MIPS/ARM/ArmRegCacheFPU.h
index 15e1c7de57..1e796a2a19 100644
--- a/Core/MIPS/ARM/ArmRegCacheFPU.h
+++ b/Core/MIPS/ARM/ArmRegCacheFPU.h
@@ -33,28 +33,56 @@ enum {
TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS,
};
+enum {
+ MAP_READ = 0,
+ MAP_MTX_TRANSPOSED = 16,
+ MAP_PREFER_LOW = 16,
+ MAP_PREFER_HIGH = 32,
+
+ // Force is not yet correctly implemented, if the reg is already mapped it will not move
+ MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15)
+ MAP_FORCE_HIGH = 128, // Only map Q8-Q15
+};
+
struct FPURegARM {
int mipsReg; // if -1, no mipsreg attached.
bool isDirty; // Should the register be written back?
};
+struct FPURegQuad {
+ int mipsVec;
+ VectorSize sz;
+ u8 vregs[4];
+ bool isDirty;
+ bool spillLock;
+ bool isTemp;
+};
+
struct FPURegMIPS {
// Where is this MIPS register?
RegMIPSLoc loc;
// Data (only one of these is used, depending on loc. Could make a union).
int reg;
+ int lane;
+
bool spillLock; // if true, this register cannot be spilled.
bool tempLock;
// If loc == ML_MEM, it's back in its location in the CPU context struct.
};
+namespace MIPSComp {
+ struct ArmJitOptions;
+ struct JitState;
+}
+
class ArmRegCacheFPU
{
public:
- ArmRegCacheFPU(MIPSState *mips);
+ ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo);
~ArmRegCacheFPU() {}
void Init(ARMXEmitter *emitter);
+
void Start(MIPSAnalyst::AnalysisResults &stats);
// Protect the arm register containing a MIPS register from spilling, to ensure that
@@ -80,9 +108,11 @@ public:
void MapDirty(MIPSReg rd);
void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true);
void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true);
+ bool IsMapped(MIPSReg r);
void FlushArmReg(ARMReg r);
void FlushR(MIPSReg r);
void DiscardR(MIPSReg r);
+ ARMReg R(int preg); // Returns a cached register
// VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode!
void MapRegV(int vreg, int flags = 0);
@@ -90,18 +120,41 @@ public:
void MapInInV(int rt, int rs);
void MapDirtyInV(int rd, int rs, bool avoidLoad = true);
void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true);
- bool IsTempX(ARMReg r) const;
- MIPSReg GetTempR();
+ bool IsTempX(ARMReg r) const;
MIPSReg GetTempV() { return GetTempR() - 32; }
+ // VFPU registers as single VFP registers.
+ ARMReg V(int vreg) { return R(vreg + 32); }
int FlushGetSequential(int a, int maxArmReg);
void FlushAll();
- ARMReg R(int preg); // Returns a cached register
+ // This one is allowed at any point.
+ void FlushV(MIPSReg r);
+
+ // VFPU registers mapped to match NEON quads (and doubles, for pairs and singles)
+ // Here we return the ARM register directly instead of providing a "V" accessor
+ // and so on. Might switch to this model for the other regallocs later.
+
+ // Quad mapping does NOT look into the ar array. Instead we use the qr array to keep
+ // track of what's in each quad.
+
+ // Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types.
+ // Need to explicitly allow spilling to get spilling.
+ ARMReg QMapReg(int vreg, VectorSize sz, int flags);
+
+ // TODO
+ // Maps a matrix as a set of columns (yes, even transposed ones, always columns
+ // as those are faster to load/flush). When possible it will map into consecutive
+ // quad registers, enabling blazing-fast full-matrix loads, transposed or not.
+ void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags);
+
+ ARMReg QAllocTemp(VectorSize sz);
- // VFPU registers as single VFP registers
- ARMReg V(int vreg) { return R(vreg + 32); }
+ void QAllowSpill(int quad);
+ void QFlush(int quad);
+ void QLoad4x4(MIPSGPReg regPtr, int vquads[4]);
+ //void FlushQWithV(MIPSReg r);
// NOTE: These require you to release spill locks manually!
void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags);
@@ -112,31 +165,43 @@ public:
void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; }
- // For better log output only.
- void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; }
-
int GetMipsRegOffset(MIPSReg r);
+
+private:
+ bool Consecutive(int v1, int v2) const;
+ bool Consecutive(int v1, int v2, int v3) const;
+ bool Consecutive(int v1, int v2, int v3, int v4) const;
+
+ MIPSReg GetTempR();
+ const ARMReg *GetMIPSAllocationOrder(int &count);
int GetMipsRegOffsetV(MIPSReg r) {
return GetMipsRegOffset(r + 32);
}
+ // This one WILL get a free quad as long as you haven't spill-locked them all.
+ int QGetFreeQuad(int start, int count, const char *reason);
int GetNumARMFPURegs();
-private:
void SetupInitialRegs();
MIPSState *mips_;
ARMXEmitter *emit_;
- u32 compilerPC_;
+ MIPSComp::JitState *js_;
+ MIPSComp::ArmJitOptions *jo_;
int numARMFpuReg_;
+ int qTime_;
enum {
- MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON
+ // With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers
+ // are individually mappable though.
+ MAX_ARMFPUREG = 32,
+ MAX_ARMQUADS = 16,
NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS,
};
FPURegARM ar[MAX_ARMFPUREG];
FPURegMIPS mr[NUM_MIPSFPUREG];
+ FPURegQuad qr[MAX_ARMQUADS];
FPURegMIPS *vr;
bool pendingFlush;
diff --git a/android/jni/Android.mk b/android/jni/Android.mk
index 1a8b82ff8e..8dbbfe2e94 100644
--- a/android/jni/Android.mk
+++ b/android/jni/Android.mk
@@ -64,6 +64,7 @@ ARCH_FILES := \
$(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \
+ $(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \
$(SRC)/Core/MIPS/ARM/ArmAsm.cpp \
$(SRC)/Core/MIPS/ARM/ArmJit.cpp \
@@ -84,6 +85,7 @@ ARCH_FILES := \
$(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \
+ $(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \
$(SRC)/Core/MIPS/ARM/ArmAsm.cpp \
$(SRC)/Core/MIPS/ARM/ArmJit.cpp \