diff --git a/CMakeLists.txt b/CMakeLists.txt index bb71c0c6cf..e520e72672 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1022,6 +1022,7 @@ if(ARM) Core/MIPS/ARM/ArmCompLoadStore.cpp Core/MIPS/ARM/ArmCompVFPU.cpp Core/MIPS/ARM/ArmCompVFPUNEON.cpp + Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp Core/MIPS/ARM/ArmCompReplace.cpp Core/MIPS/ARM/ArmJit.cpp Core/MIPS/ARM/ArmJit.h diff --git a/Core/Core.vcxproj b/Core/Core.vcxproj index ad0cc66e4a..6725309679 100644 --- a/Core/Core.vcxproj +++ b/Core/Core.vcxproj @@ -317,6 +317,18 @@ true true + + true + true + true + true + + + true + true + true + true + true true @@ -563,7 +575,7 @@ true true - + true true true @@ -658,4 +670,4 @@ - \ No newline at end of file + diff --git a/Core/Core.vcxproj.filters b/Core/Core.vcxproj.filters index 3144b4caae..915ae6c43d 100644 --- a/Core/Core.vcxproj.filters +++ b/Core/Core.vcxproj.filters @@ -393,6 +393,12 @@ MIPS\ARM + + MIPS\ARM + + + MIPS\ARM + Ext @@ -969,7 +975,7 @@ MIPS\JitCommon - + MIPS\ARM @@ -1027,4 +1033,4 @@ - \ No newline at end of file + diff --git a/Core/MIPS/ARM/ArmCompFPU.cpp b/Core/MIPS/ARM/ArmCompFPU.cpp index 37986404bb..e0a022477d 100644 --- a/Core/MIPS/ARM/ArmCompFPU.cpp +++ b/Core/MIPS/ARM/ArmCompFPU.cpp @@ -357,9 +357,12 @@ void Jit::Comp_mxc1(MIPSOpcode op) switch ((op >> 21) & 0x1f) { case 0: // R(rt) = FI(fs); break; //mfc1 - fpr.MapReg(fs); gpr.MapReg(rt, MAP_DIRTY | MAP_NOINIT); - VMOV(gpr.R(rt), fpr.R(fs)); + if (fpr.IsMapped(fs)) { + VMOV(gpr.R(rt), fpr.R(fs)); + } else { + LDR(gpr.R(rt), CTXREG, fpr.GetMipsRegOffset(fs)); + } return; case 2: //cfc1 @@ -393,11 +396,11 @@ void Jit::Comp_mxc1(MIPSOpcode op) case 4: //FI(fs) = R(rt); break; //mtc1 if (rt == MIPS_REG_ZERO) { - fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT); + fpr.MapReg(fs, MAP_NOINIT); MOVI2F(fpr.R(fs), 0.0f, R0); } else { gpr.MapReg(rt); - fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT); + fpr.MapReg(fs, MAP_NOINIT); VMOV(fpr.R(fs), gpr.R(rt)); } return; diff --git a/Core/MIPS/ARM/ArmCompVFPU.cpp b/Core/MIPS/ARM/ArmCompVFPU.cpp index 3993b64f48..a05aee73a9 100644 --- a/Core/MIPS/ARM/ArmCompVFPU.cpp +++ b/Core/MIPS/ARM/ArmCompVFPU.cpp @@ -324,8 +324,8 @@ namespace MIPSComp void Jit::Comp_SVQ(MIPSOpcode op) { - CONDITIONAL_DISABLE; NEON_IF_AVAILABLE(CompNEON_SVQ); + CONDITIONAL_DISABLE; int imm = (signed short)(op&0xFFFC); int vt = (((op >> 16) & 0x1f)) | ((op&1) << 5); @@ -506,6 +506,7 @@ namespace MIPSComp void Jit::Comp_VIdt(MIPSOpcode op) { NEON_IF_AVAILABLE(CompNEON_VIdt); + CONDITIONAL_DISABLE; if (js.HasUnknownPrefix()) { DISABLE; @@ -1574,7 +1575,7 @@ namespace MIPSComp VMUL(fpr.V(temp3), fpr.V(sregs[0]), fpr.V(tregs[1])); VMLS(fpr.V(temp3), fpr.V(sregs[1]), fpr.V(tregs[0])); - fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT); + fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT); VMOV(fpr.V(dregs[0]), S0); VMOV(fpr.V(dregs[1]), S1); VMOV(fpr.V(dregs[2]), fpr.V(temp3)); @@ -1609,7 +1610,7 @@ namespace MIPSComp VMLS(fpr.V(temp4), fpr.V(sregs[2]), fpr.V(tregs[2])); VMLA(fpr.V(temp4), fpr.V(sregs[3]), fpr.V(tregs[3])); - fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT); + fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT); VMOV(fpr.V(dregs[0]), S0); VMOV(fpr.V(dregs[1]), S1); VMOV(fpr.V(dregs[2]), fpr.V(temp3)); @@ -2118,5 +2119,4 @@ namespace MIPSComp void Jit::Comp_Vbfy(MIPSOpcode op) { DISABLE; } - } diff --git a/Core/MIPS/ARM/ArmCompVFPUNEON.cpp b/Core/MIPS/ARM/ArmCompVFPUNEON.cpp index a9d17595e7..0ec5ef2061 100644 --- a/Core/MIPS/ARM/ArmCompVFPUNEON.cpp +++ b/Core/MIPS/ARM/ArmCompVFPUNEON.cpp @@ -20,7 +20,16 @@ // that uses NEON Q registers to cache pairs/tris/quads, and so on. // Will require major extensions to the reg cache and other things. +// ARM NEON can only do pairs and quads, not tris and scalars. +// We can do scalars, though, for many operations if all the operands +// are below Q8 (D16, S32) using regular VFP instructions but really not sure +// if it's worth it. + + + #include + +#include "base/logging.h" #include "math/math_util.h" #include "Common/CPUDetect.h" @@ -28,49 +37,703 @@ #include "Core/MIPS/MIPS.h" #include "Core/MIPS/MIPSAnalyst.h" #include "Core/MIPS/MIPSCodeUtils.h" +#include "Core/MIPS/MIPSVFPUUtils.h" #include "Core/Config.h" #include "Core/Reporting.h" #include "Core/MIPS/ARM/ArmJit.h" #include "Core/MIPS/ARM/ArmRegCache.h" +#include "Core/MIPS/ARM/ArmCompVFPUNEONUtil.h" // TODO: Somehow #ifdef away on ARMv5eabi, without breaking the linker. +// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; } #define CONDITIONAL_DISABLE ; #define DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; } +#define DISABLE_UNKNOWN_PREFIX { WLOG("DISABLE: Unknown Prefix in %s", __FUNCTION__); fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; } + + +#define _RS MIPS_GET_RS(op) +#define _RT MIPS_GET_RT(op) +#define _RD MIPS_GET_RD(op) +#define _FS MIPS_GET_FS(op) +#define _FT MIPS_GET_FT(op) +#define _FD MIPS_GET_FD(op) +#define _SA MIPS_GET_SA(op) +#define _POS ((op>> 6) & 0x1F) +#define _SIZE ((op>>11) & 0x1F) +#define _IMM16 (signed short)(op & 0xFFFF) +#define _IMM26 (op & 0x03FFFFFF) + namespace MIPSComp { +static const float minus_one = -1.0f; +static const float one = 1.0f; +static const float zero = 0.0f; + + +void Jit::CompNEON_VecDo3(MIPSOpcode op) { + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + MappedRegs r = NEONMapDirtyInIn(op, sz, sz, sz); + ARMReg temp = MatchSize(Q0, r.vs); + // TODO: Special case for scalar + switch (op >> 26) { + case 24: //VFPU0 + switch ((op >> 23) & 7) { + case 0: VADD(F_32, r.vd, r.vs, r.vt); break; // vadd + case 1: VSUB(F_32, r.vd, r.vs, r.vt); break; // vsub + case 7: // vdiv // vdiv THERE IS NO NEON SIMD VDIV :( There's a fast reciprocal iterator thing though. + { + // Implement by falling back to VFP + VMOV(D0, D_0(r.vs)); + VMOV(D1, D_0(r.vt)); + VDIV(S0, S0, S2); + if (sz >= V_Pair) + VDIV(S1, S1, S3); + VMOV(D_0(r.vd), D0); + if (sz >= V_Triple) { + VMOV(D0, D_1(r.vs)); + VMOV(D1, D_1(r.vt)); + VDIV(S0, S0, S2); + if (sz == V_Quad) + VDIV(S1, S1, S3); + VMOV(D_1(r.vd), D0); + } + } + break; + default: + DISABLE; + } + break; + case 25: //VFPU1 + switch ((op >> 23) & 7) { + case 0: VMUL(F_32, r.vd, r.vs, r.vt); break; // vmul + default: + DISABLE; + } + break; + case 27: //VFPU3 + switch ((op >> 23) & 7) { + case 2: VMIN(F_32, r.vd, r.vs, r.vt); break; // vmin + case 3: VMAX(F_32, r.vd, r.vs, r.vt); break; // vmax + case 6: // vsge + VMOV_immf(temp, 1.0f); + VCGE(F_32, r.vd, r.vs, r.vt); + VAND(r.vd, r.vd, temp); + break; + case 7: // vslt + VMOV_immf(temp, 1.0f); + VCLT(F_32, r.vd, r.vs, r.vt); + VAND(r.vd, r.vd, temp); + break; + } + break; + + default: + DISABLE; + } + + NEONApplyPrefixD(r.vd); + + fpr.ReleaseSpillLocksAndDiscardTemps(); +} + + +// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; } + void Jit::CompNEON_SV(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + + // Remember to use single lane stores here and not VLDR/VSTR - switching usage + // between NEON and VFPU can be expensive on some chips. + + // Here's a common idiom we should optimize: + // lv.s S200, 0(s4) + // lv.s S201, 4(s4) + // lv.s S202, 8(s4) + // vone.s S203 + // vtfm4.q C000, E600, C200 + // Would be great if we could somehow combine the lv.s into one vector instead of mapping three + // separate quads. + + s32 offset = (signed short)(op & 0xFFFC); + int vt = ((op >> 16) & 0x1f) | ((op & 3) << 5); + MIPSGPReg rs = _RS; + + bool doCheck = false; + switch (op >> 26) + { + case 50: //lv.s // VI(vt) = Memory::Read_U32(addr); + { + if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && (offset & 3) == 0 && offset < 0x400 && offset > -0x400) { + INFO_LOG(HLE, "LV.S fastmode!"); + // TODO: Also look forward and combine multiple loads. + gpr.MapRegAsPointer(rs); + ARMReg ar = fpr.QMapReg(vt, V_Single, MAP_NOINIT | MAP_DIRTY); + if (offset) { + ADDI2R(R0, gpr.RPtr(rs), offset, R1); + VLD1_lane(F_32, ar, R0, 0, true); + } else { + VLD1_lane(F_32, ar, gpr.RPtr(rs), 0, true); + } + break; + } + INFO_LOG(HLE, "LV.S slowmode!"); + + // CC might be set by slow path below, so load regs first. + ARMReg ar = fpr.QMapReg(vt, V_Single, MAP_DIRTY | MAP_NOINIT); + if (gpr.IsImm(rs)) { + u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF; + gpr.SetRegImm(R0, addr + (u32)Memory::base); + } else { + gpr.MapReg(rs); + if (g_Config.bFastMemory) { + SetR0ToEffectiveAddress(rs, offset); + } else { + SetCCAndR0ForSafeAddress(rs, offset, R1); + doCheck = true; + } + ADD(R0, R0, MEMBASEREG); + } + FixupBranch skip; + if (doCheck) { + skip = B_CC(CC_EQ); + } + VLD1_lane(F_32, ar, R0, 0, true); + if (doCheck) { + SetJumpTarget(skip); + SetCC(CC_AL); + } + } + break; + + case 58: //sv.s // Memory::Write_U32(VI(vt), addr); + { + if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && (offset & 3) == 0 && offset < 0x400 && offset > -0x400) { + INFO_LOG(HLE, "SV.S fastmode!"); + // TODO: Also look forward and combine multiple stores. + gpr.MapRegAsPointer(rs); + ARMReg ar = fpr.QMapReg(vt, V_Single, 0); + if (offset) { + ADDI2R(R0, gpr.RPtr(rs), offset, R1); + VST1_lane(F_32, ar, R0, 0, true); + } else { + VST1_lane(F_32, ar, gpr.RPtr(rs), 0, true); + } + break; + } + + INFO_LOG(HLE, "SV.S slowmode!"); + // CC might be set by slow path below, so load regs first. + ARMReg ar = fpr.QMapReg(vt, V_Single, 0); + if (gpr.IsImm(rs)) { + u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF; + gpr.SetRegImm(R0, addr + (u32)Memory::base); + } else { + gpr.MapReg(rs); + if (g_Config.bFastMemory) { + SetR0ToEffectiveAddress(rs, offset); + } else { + SetCCAndR0ForSafeAddress(rs, offset, R1); + doCheck = true; + } + ADD(R0, R0, MEMBASEREG); + } + FixupBranch skip; + if (doCheck) { + skip = B_CC(CC_EQ); + } + VST1_lane(F_32, ar, R0, 0, true); + if (doCheck) { + SetJumpTarget(skip); + SetCC(CC_AL); + } + } + break; + } + fpr.ReleaseSpillLocksAndDiscardTemps(); +} + +inline int MIPS_GET_VQVT(u32 op) { + return (((op >> 16) & 0x1f)) | ((op & 1) << 5); } void Jit::CompNEON_SVQ(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + + int offset = (signed short)(op & 0xFFFC); + int vt = MIPS_GET_VQVT(op.encoding); + MIPSGPReg rs = _RS; + bool doCheck = false; + switch (op >> 26) + { + case 54: //lv.q + { + // Check for four-in-a-row + const u32 ops[4] = { + op.encoding, + Memory::Read_Instruction(js.compilerPC + 4).encoding, + Memory::Read_Instruction(js.compilerPC + 8).encoding, + Memory::Read_Instruction(js.compilerPC + 12).encoding + }; + if (g_Config.bFastMemory && (ops[1] >> 26) == 54 && (ops[2] >> 26) == 54 && (ops[3] >> 26) == 54) { + int offsets[4] = {offset, (s16)(ops[1] & 0xFFFC), (s16)(ops[2] & 0xFFFC), (s16)(ops[3] & 0xFFFC)}; + int rss[4] = {MIPS_GET_RS(op), MIPS_GET_RS(ops[1]), MIPS_GET_RS(ops[2]), MIPS_GET_RS(ops[3])}; + if (offsets[1] == offset + 16 && offsets[2] == offsets[1] + 16 && offsets[3] == offsets[2] + 16 && + rss[0] == rss[1] && rss[1] == rss[2] && rss[2] == rss[3]) { + int vts[4] = {MIPS_GET_VQVT(op.encoding), MIPS_GET_VQVT(ops[1]), MIPS_GET_VQVT(ops[2]), MIPS_GET_VQVT(ops[3])}; + // Also check the destination registers! + + // Detected four consecutive ones! + // gpr.MapRegAsPointer(rs); + // fpr.QLoad4x4(vts[4], rs, offset); + INFO_LOG(JIT, "Matrix load detected! TODO: optimize"); + // break; + } + } + + if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && offset < 0x400-16 && offset > -0x400-16) { + gpr.MapRegAsPointer(rs); + ARMReg ar = fpr.QMapReg(vt, V_Quad, MAP_DIRTY | MAP_NOINIT); + if (offset) { + ADDI2R(R0, gpr.RPtr(rs), offset, R1); + VLD1(F_32, ar, R0, 2, ALIGN_128); + } else { + VLD1(F_32, ar, gpr.RPtr(rs), 2, ALIGN_128); + } + break; + } + + // CC might be set by slow path below, so load regs first. + ARMReg ar = fpr.QMapReg(vt, V_Quad, MAP_DIRTY | MAP_NOINIT); + if (gpr.IsImm(rs)) { + u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF; + gpr.SetRegImm(R0, addr + (u32)Memory::base); + } else { + gpr.MapReg(rs); + if (g_Config.bFastMemory) { + SetR0ToEffectiveAddress(rs, offset); + } else { + SetCCAndR0ForSafeAddress(rs, offset, R1); + doCheck = true; + } + ADD(R0, R0, MEMBASEREG); + } + + FixupBranch skip; + if (doCheck) { + skip = B_CC(CC_EQ); + } + + VLD1(F_32, ar, R0, 2, ALIGN_128); + + if (doCheck) { + SetJumpTarget(skip); + SetCC(CC_AL); + } + } + break; + + case 62: //sv.q + { + const u32 ops[4] = { + op.encoding, + Memory::Read_Instruction(js.compilerPC + 4).encoding, + Memory::Read_Instruction(js.compilerPC + 8).encoding, + Memory::Read_Instruction(js.compilerPC + 12).encoding + }; + if (g_Config.bFastMemory && (ops[1] >> 26) == 54 && (ops[2] >> 26) == 54 && (ops[3] >> 26) == 54) { + int offsets[4] = { offset, (s16)(ops[1] & 0xFFFC), (s16)(ops[2] & 0xFFFC), (s16)(ops[3] & 0xFFFC) }; + int rss[4] = { MIPS_GET_RS(op), MIPS_GET_RS(ops[1]), MIPS_GET_RS(ops[2]), MIPS_GET_RS(ops[3]) }; + if (offsets[1] == offset + 16 && offsets[2] == offsets[1] + 16 && offsets[3] == offsets[2] + 16 && + rss[0] == rss[1] && rss[1] == rss[2] && rss[2] == rss[3]) { + int vts[4] = { MIPS_GET_VQVT(op.encoding), MIPS_GET_VQVT(ops[1]), MIPS_GET_VQVT(ops[2]), MIPS_GET_VQVT(ops[3]) }; + // Also check the destination registers! + + // Detected four consecutive ones! + // gpr.MapRegAsPointer(rs); + // fpr.QLoad4x4(vts[4], rs, offset); + INFO_LOG(JIT, "Matrix store detected! TODO: optimize"); + // break; + } + } + + if (!gpr.IsImm(rs) && jo.cachePointers && g_Config.bFastMemory && offset < 0x400-16 && offset > -0x400-16) { + gpr.MapRegAsPointer(rs); + ARMReg ar = fpr.QMapReg(vt, V_Quad, 0); + if (offset) { + ADDI2R(R0, gpr.RPtr(rs), offset, R1); + VST1(F_32, ar, R0, 2, ALIGN_128); + } else { + VST1(F_32, ar, gpr.RPtr(rs), 2, ALIGN_128); + } + break; + } + + // CC might be set by slow path below, so load regs first. + u8 vregs[4]; + ARMReg ar = fpr.QMapReg(vt, V_Quad, 0); + + if (gpr.IsImm(rs)) { + u32 addr = (offset + gpr.GetImm(rs)) & 0x3FFFFFFF; + gpr.SetRegImm(R0, addr + (u32)Memory::base); + } else { + gpr.MapReg(rs); + if (g_Config.bFastMemory) { + SetR0ToEffectiveAddress(rs, offset); + } else { + SetCCAndR0ForSafeAddress(rs, offset, R1); + doCheck = true; + } + ADD(R0, R0, MEMBASEREG); + } + + FixupBranch skip; + if (doCheck) { + skip = B_CC(CC_EQ); + } + + VST1(F_32, ar, R0, 2, ALIGN_128); + + if (doCheck) { + SetJumpTarget(skip); + SetCC(CC_AL); + } + } + break; + + default: + DISABLE; + break; + } + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_VVectorInit(MIPSOpcode op) { - DISABLE; -} + CONDITIONAL_DISABLE; + // WARNING: No prefix support! + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + VectorSize sz = GetVecSize(op); + DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_NOINIT | MAP_DIRTY); -void Jit::CompNEON_VMatrixInit(MIPSOpcode op) { - DISABLE; + switch ((op >> 16) & 0xF) { + case 6: // vzero + VEOR(vd.rd, vd.rd, vd.rd); + break; + case 7: // vone + VMOV_immf(vd.rd, 1.0f); + break; + default: + DISABLE; + break; + } + NEONApplyPrefixD(vd); + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_VDot(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + VectorSize sz = GetVecSize(op); + MappedRegs r = NEONMapDirtyInIn(op, V_Single, sz, sz); + + switch (sz) { + case V_Pair: + VMUL(F_32, r.vd, r.vs, r.vt); + VPADD(F_32, r.vd, r.vd, r.vd); + break; + case V_Triple: + VMUL(F_32, Q0, r.vs, r.vt); + VPADD(F_32, D0, D0, D0); + VADD(F_32, r.vd, D0, D1); + break; + case V_Quad: + VMUL(F_32, D0, D_0(r.vs), D_0(r.vt)); + VMLA(F_32, D0, D_1(r.vs), D_1(r.vt)); + VPADD(F_32, r.vd, D0, D0); + break; + case V_Single: + case V_Invalid: + ; + } + + NEONApplyPrefixD(r.vd); + fpr.ReleaseSpillLocksAndDiscardTemps(); } -void Jit::CompNEON_VecDo3(MIPSOpcode op) { + +void Jit::CompNEON_VHdp(MIPSOpcode op) { + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + DISABLE; + + // Similar to VDot but the last component is only s instead of s * t. + // A bit tricky on NEON... +} + +void Jit::CompNEON_VScl(MIPSOpcode op) { + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + VectorSize sz = GetVecSize(op); + MappedRegs r = NEONMapDirtyInIn(op, sz, sz, V_Single); + + ARMReg temp = MatchSize(Q0, r.vt); + + // TODO: VMUL_scalar directly when possible + VMOV_neon(temp, r.vt); + VMUL_scalar(F_32, r.vd, r.vs, DScalar(Q0, 0)); + + NEONApplyPrefixD(r.vd); + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_VV2Op(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + // Pre-processing: Eliminate silly no-op VMOVs, common in Wipeout Pure + if (((op >> 16) & 0x1f) == 0 && _VS == _VD && js.HasNoPrefix()) { + return; + } + + // Must bail before we start mapping registers. + switch ((op >> 16) & 0x1f) { + case 0: // d[i] = s[i]; break; //vmov + case 1: // d[i] = fabsf(s[i]); break; //vabs + case 2: // d[i] = -s[i]; break; //vneg + case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq + break; + + default: + DISABLE; + break; + } + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + MappedRegs r = NEONMapDirtyIn(op, sz, sz); + + ARMReg temp = MatchSize(Q0, r.vs); + + switch ((op >> 16) & 0x1f) { + case 0: // d[i] = s[i]; break; //vmov + // Probably for swizzle. + VMOV_neon(r.vd, r.vs); + break; + case 1: // d[i] = fabsf(s[i]); break; //vabs + VABS(F_32, r.vd, r.vs); + break; + case 2: // d[i] = -s[i]; break; //vneg + VNEG(F_32, r.vd, r.vs); + break; + + case 4: // if (s[i] < 0) d[i] = 0; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat0 + if (IsD(r.vd)) { + VMOV_immf(D0, 0.0f); + VMOV_immf(D1, 1.0f); + VMAX(F_32, r.vd, r.vs, D0); + VMIN(F_32, r.vd, r.vd, D1); + } else { + VMOV_immf(Q0, 1.0f); + VMIN(F_32, r.vd, r.vs, Q0); + VMOV_immf(Q0, 0.0f); + VMAX(F_32, r.vd, r.vd, Q0); + } + break; + case 5: // if (s[i] < -1.0f) d[i] = -1.0f; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat1 + if (IsD(r.vd)) { + VMOV_immf(D0, -1.0f); + VMOV_immf(D1, 1.0f); + VMAX(F_32, r.vd, r.vs, D0); + VMIN(F_32, r.vd, r.vd, D1); + } else { + VMOV_immf(Q0, 1.0f); + VMIN(F_32, r.vd, r.vs, Q0); + VMOV_immf(Q0, -1.0f); + VMAX(F_32, r.vd, r.vd, Q0); + } + break; + + case 16: // d[i] = 1.0f / s[i]; break; //vrcp + // Can just fallback to VFP and use VDIV. + DISABLE; + { + ARMReg temp2 = fpr.QAllocTemp(sz); + // Needs iterations on NEON. And two temps - which is a problem if vs == vd! Argh! + VRECPE(F_32, temp, r.vs); + VRECPS(temp2, r.vs, temp); + VMUL(F_32, temp2, temp2, temp); + VRECPS(temp2, r.vs, temp); + VMUL(F_32, temp2, temp2, temp); + } + // http://stackoverflow.com/questions/6759897/how-to-divide-in-neon-intrinsics-by-a-float-number + // reciprocal = vrecpeq_f32(b); + // reciprocal = vmulq_f32(vrecpsq_f32(b, reciprocal), reciprocal); + // reciprocal = vmulq_f32(vrecpsq_f32(b, reciprocal), reciprocal); + DISABLE; + break; + + case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq + DISABLE; + // Needs iterations on NEON + { + if (true) { + // Not-very-accurate estimate + VRSQRTE(F_32, r.vd, r.vs); + } else { + ARMReg temp2 = fpr.QAllocTemp(sz); + // TODO: It's likely that some games will require one or two Newton-Raphson + // iterations to refine the estimate. + VRSQRTE(F_32, temp, r.vs); + VRSQRTS(temp2, r.vs, temp); + VMUL(F_32, r.vd, temp2, temp); + //VRSQRTS(temp2, r.vs, temp); + // VMUL(F_32, r.vd, temp2, temp); + } + } + break; + case 18: // d[i] = sinf((float)M_PI_2 * s[i]); break; //vsin + DISABLE; + break; + case 19: // d[i] = cosf((float)M_PI_2 * s[i]); break; //vcos + DISABLE; + break; + case 20: // d[i] = powf(2.0f, s[i]); break; //vexp2 + DISABLE; + break; + case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2 + DISABLE; + break; + case 22: // d[i] = sqrtf(s[i]); break; //vsqrt + // Let's just defer to VFP for now. Better than calling the interpreter for sure. + VMOV_neon(MatchSize(Q0, r.vs), r.vs); + for (int i = 0; i < n; i++) { + VSQRT((ARMReg)(S0 + i), (ARMReg)(S0 + i)); + } + VMOV_neon(MatchSize(Q0, r.vd), r.vd); + break; + case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin + DISABLE; + break; + case 24: // d[i] = -1.0f / s[i]; break; // vnrcp + // Needs iterations on NEON. Just do the same as vrcp and negate. + DISABLE; + break; + case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin + DISABLE; + break; + case 28: // d[i] = 1.0f / expf(s[i] * (float)M_LOG2E); break; // vrexp2 + DISABLE; + break; + default: + DISABLE; + break; + } + + NEONApplyPrefixD(r.vd); + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Mftv(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + int imm = op & 0xFF; + MIPSGPReg rt = _RT; + switch ((op >> 21) & 0x1f) { + case 3: //mfv / mfvc + // rt = 0, imm = 255 appears to be used as a CPU interlock by some games. + if (rt != 0) { + if (imm < 128) { //R(rt) = VI(imm); + ARMReg r = fpr.QMapReg(imm, V_Single, MAP_READ); + gpr.MapReg(rt, MAP_NOINIT | MAP_DIRTY); + // TODO: Gotta be a faster way + VMOV_neon(MatchSize(Q0, r), r); + VMOV(gpr.R(rt), S0); + } else if (imm < 128 + VFPU_CTRL_MAX) { //mtvc + // In case we have a saved prefix. + FlushPrefixV(); + if (imm - 128 == VFPU_CTRL_CC) { + gpr.MapDirtyIn(rt, MIPS_REG_VFPUCC); + MOV(gpr.R(rt), gpr.R(MIPS_REG_VFPUCC)); + } else { + gpr.MapReg(rt, MAP_NOINIT | MAP_DIRTY); + LDR(gpr.R(rt), CTXREG, offsetof(MIPSState, vfpuCtrl) + 4 * (imm - 128)); + } + } else { + //ERROR - maybe need to make this value too an "interlock" value? + ERROR_LOG(CPU, "mfv - invalid register %i", imm); + } + } + break; + + case 7: // mtv + if (imm < 128) { + // TODO: It's pretty common that this is preceded by mfc1, that is, a value is being + // moved from the regular floating point registers. It would probably be faster to do + // the copy directly in the FPRs instead of going through the GPRs. + + ARMReg r = fpr.QMapReg(imm, V_Single, MAP_DIRTY | MAP_NOINIT); + if (gpr.IsMapped(rt)) { + VMOV(S0, gpr.R(rt)); + VMOV_neon(r, MatchSize(Q0, r)); + } else { + ADDI2R(R0, CTXREG, gpr.GetMipsRegOffset(rt), R1); + VLD1_lane(F_32, r, R0, 0, true); + } + } else if (imm < 128 + VFPU_CTRL_MAX) { //mtvc //currentMIPS->vfpuCtrl[imm - 128] = R(rt); + if (imm - 128 == VFPU_CTRL_CC) { + gpr.MapDirtyIn(MIPS_REG_VFPUCC, rt); + MOV(gpr.R(MIPS_REG_VFPUCC), rt); + } else { + gpr.MapReg(rt); + STR(gpr.R(rt), CTXREG, offsetof(MIPSState, vfpuCtrl) + 4 * (imm - 128)); + } + //gpr.BindToRegister(rt, true, false); + //MOV(32, M(¤tMIPS->vfpuCtrl[imm - 128]), gpr.R(rt)); + + // TODO: Optimization if rt is Imm? + // Set these BEFORE disable! + if (imm - 128 == VFPU_CTRL_SPREFIX) { + js.prefixSFlag = JitState::PREFIX_UNKNOWN; + } else if (imm - 128 == VFPU_CTRL_TPREFIX) { + js.prefixTFlag = JitState::PREFIX_UNKNOWN; + } else if (imm - 128 == VFPU_CTRL_DPREFIX) { + js.prefixDFlag = JitState::PREFIX_UNKNOWN; + } + } else { + //ERROR + _dbg_assert_msg_(CPU,0,"mtv - invalid register"); + } + break; + + default: + DISABLE; + } + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vmfvc(MIPSOpcode op) { @@ -78,31 +741,223 @@ void Jit::CompNEON_Vmfvc(MIPSOpcode op) { } void Jit::CompNEON_Vmtvc(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + + int vs = _VS; + int imm = op & 0xFF; + if (imm >= 128 && imm < 128 + VFPU_CTRL_MAX) { + ARMReg r = fpr.QMapReg(vs, V_Single, 0); + ADDI2R(R0, CTXREG, offsetof(MIPSState, vfpuCtrl[0]) + (imm - 128) * 4, R1); + VST1_lane(F_32, r, R0, 0, true); + fpr.ReleaseSpillLocksAndDiscardTemps(); + + if (imm - 128 == VFPU_CTRL_SPREFIX) { + js.prefixSFlag = JitState::PREFIX_UNKNOWN; + } else if (imm - 128 == VFPU_CTRL_TPREFIX) { + js.prefixTFlag = JitState::PREFIX_UNKNOWN; + } else if (imm - 128 == VFPU_CTRL_DPREFIX) { + js.prefixDFlag = JitState::PREFIX_UNKNOWN; + } + } +} + +void Jit::CompNEON_VMatrixInit(MIPSOpcode op) { + CONDITIONAL_DISABLE; + + MatrixSize msz = GetMtxSize(op); + int n = GetMatrixSide(msz); + + ARMReg cols[4]; + fpr.QMapMatrix(cols, _VD, msz, MAP_NOINIT | MAP_DIRTY); + + switch ((op >> 16) & 0xF) { + case 3: // vmidt + // There has to be a better way to synthesize: 1.0, 0.0, 0.0, 1.0 in a quad + VEOR(D0, D0, D0); + VMOV_immf(D1, 1.0f); + VTRN(F_32, D0, D1); + VREV64(I_32, D0, D0); + switch (msz) { + case M_2x2: + VMOV_neon(cols[0], D0); + VMOV_neon(cols[1], D1); + break; + case M_3x3: + VMOV_neon(D_0(cols[0]), D0); + VMOV_imm(I_8, D_1(cols[0]), VIMMxxxxxxxx, 0); + VMOV_neon(D_0(cols[1]), D1); + VMOV_imm(I_8, D_1(cols[1]), VIMMxxxxxxxx, 0); + VMOV_imm(I_8, D_0(cols[2]), VIMMxxxxxxxx, 0); + VMOV_neon(D_1(cols[2]), D0); + break; + case M_4x4: + VMOV_neon(D_0(cols[0]), D0); + VMOV_imm(I_8, D_1(cols[0]), VIMMxxxxxxxx, 0); + VMOV_neon(D_0(cols[1]), D1); + VMOV_imm(I_8, D_1(cols[1]), VIMMxxxxxxxx, 0); + VMOV_imm(I_8, D_0(cols[2]), VIMMxxxxxxxx, 0); + VMOV_neon(D_1(cols[2]), D0); + VMOV_imm(I_8, D_0(cols[3]), VIMMxxxxxxxx, 0); + VMOV_neon(D_1(cols[3]), D1); + + // NEONTranspose4x4(cols); + break; + default: + _assert_msg_(JIT, 0, "Bad matrix size"); + break; + } + break; + case 6: // vmzero + for (int i = 0; i < n; i++) { + VEOR(cols[i], cols[i], cols[i]); + } + break; + case 7: // vmone + for (int i = 0; i < n; i++) { + VMOV_immf(cols[i], 1.0f); + } + break; + } + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vmmov(MIPSOpcode op) { - DISABLE; -} + CONDITIONAL_DISABLE; + if (_VS == _VD) { + // A lot of these no-op matrix moves in Wipeout... Just drop the instruction entirely. + return; + } -void Jit::CompNEON_VScl(MIPSOpcode op) { - DISABLE; + MatrixSize msz = GetMtxSize(op); + + MatrixOverlapType overlap = GetMatrixOverlap(_VD, _VS, msz); + if (overlap != OVERLAP_NONE) { + // Too complicated to bother handling in the JIT. + // TODO: Special case for in-place (and other) transpose, etc. + DISABLE; + } + + ARMReg s_cols[4], d_cols[4]; + fpr.QMapMatrix(s_cols, _VS, msz, 0); + fpr.QMapMatrix(d_cols, _VD, msz, MAP_DIRTY | MAP_NOINIT); + + int n = GetMatrixSide(msz); + for (int i = 0; i < n; i++) { + VMOV_neon(d_cols[i], s_cols[i]); + } + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vmmul(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + + MatrixSize msz = GetMtxSize(op); + int n = GetMatrixSide(msz); + + bool overlap = GetMatrixOverlap(_VD, _VS, msz) || GetMatrixOverlap(_VD, _VT, msz); + if (overlap) { + // Later. Fortunately, the VFPU also seems to prohibit overlap for matrix mul. + ILOG("Matrix overlap, ignoring."); + DISABLE; + } + + // Having problems with 2x2s for some reason. + if (msz == M_2x2) { + DISABLE; + } + + ARMReg s_cols[4], t_cols[4], d_cols[4]; + + // For some reason, vmmul is encoded with the first matrix (S) transposed from the real meaning. + fpr.QMapMatrix(t_cols, _VT, msz, MAP_FORCE_LOW); // Need to see if we can avoid having to force it low in some sane way. Will need crazy prediction logic for loads otherwise. + fpr.QMapMatrix(s_cols, Xpose(_VS), msz, MAP_PREFER_HIGH); + fpr.QMapMatrix(d_cols, _VD, msz, MAP_PREFER_HIGH | MAP_NOINIT | MAP_DIRTY); + + // TODO: Getting there but still getting wrong results. + for (int i = 0; i < n; i++) { + for (int j = 0; j < n; j++) { + if (i == 0) { + VMUL_scalar(F_32, d_cols[j], s_cols[i], XScalar(t_cols[j], i)); + } else { + VMLA_scalar(F_32, d_cols[j], s_cols[i], XScalar(t_cols[j], i)); + } + } + } + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vmscl(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + + MatrixSize msz = GetMtxSize(op); + + bool overlap = GetMatrixOverlap(_VD, _VS, msz); + if (overlap) { + DISABLE; + } + + int n = GetMatrixSide(msz); + + ARMReg s_cols[4], t, d_cols[4]; + fpr.QMapMatrix(s_cols, _VS, msz, 0); + fpr.QMapMatrix(d_cols, _VD, msz, MAP_NOINIT | MAP_DIRTY); + + t = fpr.QMapReg(_VT, V_Single, 0); + VMOV_neon(D0, t); + for (int i = 0; i < n; i++) { + VMUL_scalar(F_32, d_cols[i], s_cols[i], DScalar(D0, 0)); + } + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vtfm(MIPSOpcode op) { - DISABLE; -} + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE; + } -void Jit::CompNEON_VHdp(MIPSOpcode op) { - DISABLE; + if (_VT == _VD) { + DISABLE; + } + + VectorSize sz = GetVecSize(op); + MatrixSize msz = GetMtxSize(op); + int n = GetNumVectorElements(sz); + int ins = (op >> 23) & 7; + + bool homogenous = false; + if (n == ins) { + n++; + sz = (VectorSize)((int)(sz)+1); + msz = (MatrixSize)((int)(msz)+1); + homogenous = true; + } + // Otherwise, n should already be ins + 1. + else if (n != ins + 1) { + DISABLE; + } + + ARMReg s_cols[4], t, d; + t = fpr.QMapReg(_VT, sz, MAP_FORCE_LOW); + fpr.QMapMatrix(s_cols, Xpose(_VS), msz, MAP_PREFER_HIGH); + d = fpr.QMapReg(_VD, sz, MAP_DIRTY | MAP_NOINIT | MAP_PREFER_HIGH); + + VMUL_scalar(F_32, d, s_cols[0], XScalar(t, 0)); + for (int i = 1; i < n; i++) { + if (homogenous && i == n - 1) { + VADD(F_32, d, d, s_cols[i]); + } else { + VMLA_scalar(F_32, d, s_cols[i], XScalar(t, i)); + } + } + + // VTFM does not have prefix support. + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_VCrs(MIPSOpcode op) { @@ -126,55 +981,457 @@ void Jit::CompNEON_Vf2i(MIPSOpcode op) { } void Jit::CompNEON_Vi2f(MIPSOpcode op) { + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE; + } + DISABLE; + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + int imm = (op >> 16) & 0x1f; + const float mult = 1.0f / (float)(1UL << imm); + + MappedRegs regs = NEONMapDirtyIn(op, sz, sz); + + MOVI2F_neon(MatchSize(Q0, regs.vd), mult, R0); + + VCVT(F_32, regs.vd, regs.vs); + VMUL(F_32, regs.vd, regs.vd, Q0); + + NEONApplyPrefixD(regs.vd); + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vh2f(MIPSOpcode op) { - DISABLE; + if (!cpu_info.bHalf) { + // No hardware support for half-to-float, fallback to interpreter + // TODO: Translate the fast SSE solution to standard integer/VFP stuff + // for the weaker CPUs. + DISABLE; + } + + VectorSize sz = GetVecSize(op); + + VectorSize outsize = V_Pair; + switch (sz) { + case V_Single: + outsize = V_Pair; + break; + case V_Pair: + outsize = V_Quad; + break; + default: + ERROR_LOG(JIT, "Vh2f: Must be pair or quad"); + break; + } + + ARMReg vs = NEONMapPrefixS(_VS, sz, 0); + DestARMReg vd = NEONMapPrefixD(_VD, outsize, MAP_DIRTY | (vd == vs ? 0 : MAP_NOINIT)); + + VCVTF32F16(vd.rd, vs); + + NEONApplyPrefixD(vd); + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vcst(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + int conNum = (op >> 16) & 0x1f; + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_DIRTY | MAP_NOINIT); + gpr.SetRegImm(R0, (u32)(void *)&cst_constants[conNum]); + VLD1_all_lanes(F_32, vd, R0, true); + NEONApplyPrefixD(vd); // TODO: Could bake this into the constant we load. + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vhoriz(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + VectorSize sz = GetVecSize(op); + // Do any games use these a noticeable amount? + switch ((op >> 16) & 31) { + case 6: // vfad + { + MappedRegs r = NEONMapDirtyIn(op, V_Single, sz); + switch (sz) { + case V_Pair: + VPADD(F_32, r.vd, r.vs, r.vs); + break; + case V_Triple: + VPADD(F_32, D0, D_0(r.vs), D_0(r.vs)); + VADD(F_32, r.vd, D0, D_1(r.vs)); + break; + case V_Quad: + VADD(F_32, R0, D_0(r.vs), D_1(r.vs)); + VPADD(F_32, r.vd, R0, R0); + break; + default: + ; + } + break; + } + + case 7: // vavg + DISABLE; + break; + } + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_VRot(MIPSOpcode op) { + CONDITIONAL_DISABLE; + + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + DISABLE; + + int vd = _VD; + int vs = _VS; + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + // ... + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_VIdt(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + VectorSize sz = GetVecSize(op); + DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_NOINIT | MAP_DIRTY); + switch (sz) { + case V_Pair: + VMOV_immf(vd, 1.0f); + if ((_VD & 1) == 0) { + // Load with 1.0, 0.0 + VMOV_imm(I_64, D0, VIMMbits2bytes, 0x0F); + VAND(vd, vd, D0); + } else { + VMOV_imm(I_64, D0, VIMMbits2bytes, 0xF0); + VAND(vd, vd, D0); + } + break; + case V_Triple: + case V_Quad: + { + // TODO: This can be optimized. + VEOR(vd, vd, vd); + ARMReg dest = (_VD & 2) ? D_1(vd) : D_0(vd); + VMOV_immf(dest, 1.0f); + if ((_VD & 1) == 0) { + // Load with 1.0, 0.0 + VMOV_imm(I_64, D0, VIMMbits2bytes, 0x0F); + VAND(dest, dest, D0); + } else { + VMOV_imm(I_64, D0, VIMMbits2bytes, 0xF0); + VAND(dest, dest, D0); + } + } + break; + default: + _dbg_assert_msg_(CPU,0,"Bad vidt instruction"); + break; + } + + NEONApplyPrefixD(vd); + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vcmp(MIPSOpcode op) { + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) + DISABLE; + + // Not a chance that this works on the first try :P DISABLE; + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + VCondition cond = (VCondition)(op & 0xF); + + MappedRegs regs = NEONMapInIn(op, sz, sz); + + ARMReg vs = regs.vs, vt = regs.vt; + ARMReg res = fpr.QAllocTemp(sz); + + // Some, we just fall back to the interpreter. + // ES is just really equivalent to (value & 0x7F800000) == 0x7F800000. + switch (cond) { + case VC_EI: // c = my_isinf(s[i]); break; + case VC_NI: // c = !my_isinf(s[i]); break; + DISABLE; + case VC_ES: // c = my_isnan(s[i]) || my_isinf(s[i]); break; // Tekken Dark Resurrection + case VC_NS: // c = !my_isnan(s[i]) && !my_isinf(s[i]); break; + case VC_EN: // c = my_isnan(s[i]); break; + case VC_NN: // c = !my_isnan(s[i]); break; + // if (_VS != _VT) + DISABLE; + break; + + case VC_EZ: + case VC_NZ: + VMOV_immf(Q0, 0.0f); + break; + default: + ; + } + + int affected_bits = (1 << 4) | (1 << 5); // 4 and 5 + for (int i = 0; i < n; i++) { + affected_bits |= 1 << i; + } + + // Preload the pointer to our magic mask + static const u32 collectorBits[4] = { 1, 2, 4, 8 }; + MOVP2R(R1, &collectorBits); + + // Do the compare + MOVI2R(R0, 0); + CCFlags flag = CC_AL; + + bool oneIsFalse = false; + switch (cond) { + case VC_FL: // c = 0; + break; + + case VC_TR: // c = 1 + MOVI2R(R0, affected_bits); + break; + + case VC_ES: // c = my_isnan(s[i]) || my_isinf(s[i]); break; // Tekken Dark Resurrection + case VC_NS: // c = !(my_isnan(s[i]) || my_isinf(s[i])); break; + DISABLE; // TODO: these shouldn't be that hard + break; + + case VC_EN: // c = my_isnan(s[i]); break; // Tekken 6 + case VC_NN: // c = !my_isnan(s[i]); break; + DISABLE; // TODO: these shouldn't be that hard + break; + + case VC_EQ: // c = s[i] == t[i] + VCEQ(F_32, res, vs, vt); + break; + + case VC_LT: // c = s[i] < t[i] + VCLT(F_32, res, vs, vt); + break; + + case VC_LE: // c = s[i] <= t[i]; + VCLE(F_32, res, vs, vt); + break; + + case VC_NE: // c = s[i] != t[i] + VCEQ(F_32, res, vs, vt); + oneIsFalse = true; + break; + + case VC_GE: // c = s[i] >= t[i] + VCGE(F_32, res, vs, vt); + break; + + case VC_GT: // c = s[i] > t[i] + VCGT(F_32, res, vs, vt); + break; + + case VC_EZ: // c = s[i] == 0.0f || s[i] == -0.0f + VCEQ(F_32, res, vs); + break; + + case VC_NZ: // c = s[i] != 0 + VCEQ(F_32, res, vs); + oneIsFalse = true; + break; + + default: + DISABLE; + } + if (oneIsFalse) { + VMVN(res, res); + } + // Somehow collect the bits into a mask. + + // Collect the bits. Where's my PMOVMSKB? :( + VLD1(I_32, Q0, R1, n < 2 ? 1 : 2); + VAND(Q0, Q0, res); + VPADD(I_32, Q0, Q0, Q0); + VPADD(I_32, D0, D0, D0); + // OK, bits now in S0. + VMOV(R0, S0); + // Zap irrelevant bits (V_Single, V_Triple) + AND(R0, R0, affected_bits); + + // TODO: Now, how in the world do we generate the component OR and AND bits without burning tens of ALU instructions?? Lookup-table? + + gpr.MapReg(MIPS_REG_VFPUCC, MAP_DIRTY); + BIC(gpr.R(MIPS_REG_VFPUCC), gpr.R(MIPS_REG_VFPUCC), affected_bits); + ORR(gpr.R(MIPS_REG_VFPUCC), gpr.R(MIPS_REG_VFPUCC), R0); } void Jit::CompNEON_Vcmov(MIPSOpcode op) { + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE; + } + DISABLE; + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + ARMReg vs = NEONMapPrefixS(_VS, sz, 0); + DestARMReg vd = NEONMapPrefixD(_VD, sz, MAP_DIRTY); + int tf = (op >> 19) & 1; + int imm3 = (op >> 16) & 7; + + if (imm3 < 6) { + // Test one bit of CC. This bit decides whether none or all subregisters are copied. + gpr.MapReg(MIPS_REG_VFPUCC); + TST(gpr.R(MIPS_REG_VFPUCC), 1 << imm3); + FixupBranch skip = B_CC(CC_NEQ); + VMOV_neon(vd, vs); + SetJumpTarget(skip); + } else { + // Look at the bottom four bits of CC to individually decide if the subregisters should be copied. + // This is the nasty one! Need to expand those bits into a full NEON register somehow. + DISABLE; + /* + gpr.MapReg(MIPS_REG_VFPUCC); + for (int i = 0; i < n; i++) { + TST(gpr.R(MIPS_REG_VFPUCC), 1 << i); + SetCC(tf ? CC_EQ : CC_NEQ); + VMOV(fpr.V(dregs[i]), fpr.V(sregs[i])); + SetCC(CC_AL); + } + */ + } + + NEONApplyPrefixD(vd); + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Viim(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE; + } + + DestARMReg vt = NEONMapPrefixD(_VT, V_Single, MAP_NOINIT | MAP_DIRTY); + + s32 imm = (s32)(s16)(u16)(op & 0xFFFF); + // TODO: Optimize for low registers. + MOVI2F(S0, (float)imm, R0); + VMOV_neon(vt.rd, D0); + + NEONApplyPrefixD(vt); + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vfim(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE; + } + + DestARMReg vt = NEONMapPrefixD(_VT, V_Single, MAP_NOINIT | MAP_DIRTY); + + FP16 half; + half.u = op & 0xFFFF; + FP32 fval = half_to_float_fast5(half); + // TODO: Optimize for low registers. + MOVI2F(S0, (float)fval.f, R0); + VMOV_neon(vt.rd, D0); + + NEONApplyPrefixD(vt); + fpr.ReleaseSpillLocksAndDiscardTemps(); } +// https://code.google.com/p/bullet/source/browse/branches/PhysicsEffects/include/vecmath/neon/vectormath_neon_assembly_implementations.S?r=2488 void Jit::CompNEON_VCrossQuat(MIPSOpcode op) { - DISABLE; + // This op does not support prefixes anyway. + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE_UNKNOWN_PREFIX; + } + + VectorSize sz = GetVecSize(op); + if (sz != V_Triple) { + // Quaternion product. Bleh. + DISABLE; + } + + MappedRegs r = NEONMapDirtyInIn(op, sz, sz, sz, false); + + ARMReg t1 = Q0; + ARMReg t2 = fpr.QAllocTemp(V_Triple); + + // There has to be a faster way to do this. This is not really any better than + // scalar. + + // d18, d19 (q9) = t1 = r.vt + // d16, d17 (q8) = t2 = r.vs + // d20, d21 (q10) = t + VMOV_neon(t1, r.vs); + VMOV_neon(t2, r.vt); + VTRN(F_32, D_0(t2), D_1(t2)); // vtrn.32 d18,d19 @ q9 = = d18,d19 + VREV64(F_32, D_0(t1), D_0(t1)); // vrev64.32 d16,d16 @ q8 = = d16,d17 + VREV64(F_32, D_0(t2), D_0(t2)); // vrev64.32 d18,d18 @ q9 = = d18,d19 + VTRN(F_32, D_0(t1), D_1(t1)); // vtrn.32 d16,d17 @ q8 = = d16,d17 + // perform first half of cross product using rearranged inputs + VMUL(F_32, r.vd, t1, t2); // vmul.f32 q10, q8, q9 @ q10 = + // @ rearrange inputs again + VTRN(F_32, D_0(t2), D_1(t2)); // vtrn.32 d18,d19 @ q9 = = d18,d19 + VREV64(F_32, D_0(t1), D_0(t1)); // vrev64.32 d16,d16 @ q8 = = d16,d17 + VREV64(F_32, D_0(t2), D_0(t2)); // vrev64.32 d18,d18 @ q9 = = d18,d19 + VTRN(F_32, D_0(t1), D_1(t1)); // vtrn.32 d16,d17 @ q8 = = d16,d17 + // @ perform last half of cross product using rearranged inputs + VMLS(F_32, r.vd, t1, t2); // vmls.f32 q10, q8, q9 @ q10 = + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_Vsgn(MIPSOpcode op) { DISABLE; + + // This will be a bunch of bit magic. } void Jit::CompNEON_Vocp(MIPSOpcode op) { - DISABLE; + CONDITIONAL_DISABLE; + if (js.HasUnknownPrefix()) { + DISABLE; + } + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + + MappedRegs regs = NEONMapDirtyIn(op, sz, sz); + MOVI2F_neon(Q0, 1.0f, R0); + VSUB(F_32, regs.vd, Q0, regs.vs); + NEONApplyPrefixD(regs.vd); + + fpr.ReleaseSpillLocksAndDiscardTemps(); } void Jit::CompNEON_ColorConv(MIPSOpcode op) { diff --git a/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp new file mode 100644 index 0000000000..45512df225 --- /dev/null +++ b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp @@ -0,0 +1,419 @@ +// Copyright (c) 2013- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +// NEON VFPU +// This is where we will create an alternate implementation of the VFPU emulation +// that uses NEON Q registers to cache pairs/tris/quads, and so on. +// Will require major extensions to the reg cache and other things. + +// ARM NEON can only do pairs and quads, not tris and scalars. +// We can do scalars, though, for many operations if all the operands +// are below Q8 (D16, S32) using regular VFP instructions but really not sure +// if it's worth it. + +#include + +#include "base/logging.h" +#include "math/math_util.h" + +#include "Common/CPUDetect.h" +#include "Core/MemMap.h" +#include "Core/MIPS/MIPS.h" +#include "Core/MIPS/MIPSAnalyst.h" +#include "Core/MIPS/MIPSCodeUtils.h" +#include "Core/MIPS/MIPSVFPUUtils.h" +#include "Core/Config.h" +#include "Core/Reporting.h" + +#include "Core/MIPS/ARM/ArmJit.h" +#include "Core/MIPS/ARM/ArmRegCache.h" +#include "Core/MIPS/ARM/ArmCompVFPUNEONUtil.h" + +// TODO: Somehow #ifdef away on ARMv5eabi, without breaking the linker. +// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; } + +#define CONDITIONAL_DISABLE ; +#define DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; } + + +#define _RS MIPS_GET_RS(op) +#define _RT MIPS_GET_RT(op) +#define _RD MIPS_GET_RD(op) +#define _FS MIPS_GET_FS(op) +#define _FT MIPS_GET_FT(op) +#define _FD MIPS_GET_FD(op) +#define _SA MIPS_GET_SA(op) +#define _POS ((op>> 6) & 0x1F) +#define _SIZE ((op>>11) & 0x1F) +#define _IMM16 (signed short)(op & 0xFFFF) +#define _IMM26 (op & 0x03FFFFFF) + +namespace MIPSComp { + +static const float minus_one = -1.0f; +static const float one = 1.0f; +static const float zero = 0.0f; + +// On NEON, we map triples to Q registers and singles to D registers. +// Sometimes, as when doing dot products, it matters what's in that unused reg. This zeroes it. +void Jit::NEONMaskToSize(ARMReg vs, VectorSize sz) { + // TODO +} + +ARMReg Jit::NEONMapPrefixST(int mipsReg, VectorSize sz, u32 prefix, int mapFlags) { + static const float constantArray[8] = { 0.f, 1.f, 2.f, 0.5f, 3.f, 1.f / 3.f, 0.25f, 1.f / 6.f }; + static const float constantArrayNegated[8] = { -0.f, -1.f, -2.f, -0.5f, -3.f, -1.f / 3.f, -0.25f, -1.f / 6.f }; + + // Applying prefixes in SIMD fashion will actually be a lot easier than the old style. + if (prefix == 0xE4) { + return fpr.QMapReg(mipsReg, sz, mapFlags); + } + + int n = GetNumVectorElements(sz); + + int regnum[4] = { -1, -1, -1, -1 }; + int abs[4] = { 0 }; + int negate[4] = { 0 }; + int constants[4] = { 0 }; + int constNum[4] = { 0 }; + + int full_mask = (1 << n) - 1; + + int abs_mask = (prefix >> 8) & full_mask; + int negate_mask = (prefix >> 16) & full_mask; + int constants_mask = (prefix >> 12) & full_mask; + + // Decode prefix to keep the rest readable + int permuteMask = 0; + for (int i = 0; i < n; i++) { + permuteMask |= 3 << (i * 2); + regnum[i] = (prefix >> (i * 2)) & 3; + abs[i] = (prefix >> (8 + i)) & 1; + negate[i] = (prefix >> (16 + i)) & 1; + constants[i] = (prefix >> (12 + i)) & 1; + + if (constants[i]) { + constNum[i] = regnum[i] + (abs[i] << 2); + abs[i] = 0; + } + } + abs_mask &= ~constants_mask; + + bool anyPermute = (prefix & permuteMask) != (0xE4 & permuteMask); + + if (constants_mask == full_mask) { + // It's all constants! Don't even bother mapping the input register, + // just allocate a temp one. + // If a single, this can sometimes be done cheaper. But meh. + ARMReg ar = fpr.QAllocTemp(sz); + for (int i = 0; i < n; i++) { + if ((i & 1) == 0) { + if (constNum[i] == constNum[i + 1]) { + // Replace two loads with a single immediate when easily possible. + ARMReg dest = i & 2 ? D_1(ar) : D_0(ar); + switch (constNum[i]) { + case 0: + case 1: + { + float c = constantArray[constNum[i]]; + VMOV_immf(dest, negate[i] ? -c : c); + } + break; + // TODO: There are a few more that are doable. + default: + goto skip; + } + + i++; + continue; + skip: + ; + } + } + MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]); + VLD1_lane(F_32, ar, R0, i, true); + } + return ar; + } + + // 1. Permute. + // 2. Abs + // If any constants: + // 3. Replace values with constants + // 4. Negate + + ARMReg inputAR = fpr.QMapReg(mipsReg, sz, mapFlags); + ARMReg ar = fpr.QAllocTemp(sz); + + if (!anyPermute) { + VMOV(ar, inputAR); + // No permutations! + } else { + bool allSame = false; + for (int i = 1; i < n; i++) { + if (regnum[0] == regnum[i]) + allSame = true; + } + + if (allSame) { + // Easy, someone is duplicating one value onto all the reg parts. + // If this is happening and QMapReg must load, we can combine these two actions + // into a VLD1_lane. TODO + VDUP(F_32, ar, inputAR, regnum[0]); + } else { + // Do some special cases + if (regnum[0] == 1 && regnum[1] == 0) { + INFO_LOG(HLE, "PREFIXST: Bottom swap!"); + VREV64(I_32, ar, inputAR); + regnum[0] = 0; + regnum[1] = 1; + } + + // TODO: Make a generic fallback using another temp register + + bool match = true; + for (int i = 0; i < n; i++) { + if (regnum[i] != i) + match = false; + } + + // TODO: Cannot do this permutation yet! + if (!match) { + ERROR_LOG(HLE, "PREFIXST: Unsupported permute! %i %i %i %i / %i", regnum[0], regnum[1], regnum[2], regnum[3], n); + VMOV(ar, inputAR); + } + } + } + + // ABS + // Two methods: If all lanes are "absoluted", it's easy. + if (abs_mask == full_mask) { + // TODO: elide the above VMOV (in !anyPermute) when possible + VABS(F_32, ar, ar); + } else if (abs_mask != 0) { + // Partial ABS! + if (abs_mask == 3) { + VABS(F_32, D_0(ar), D_0(ar)); + } else { + // Horrifying fallback: Mov to Q0, abs, move back. + // TODO: Optimize for lower quads where we don't need to move. + VMOV(MatchSize(Q0, ar), ar); + for (int i = 0; i < n; i++) { + if (abs_mask & (1 << i)) { + VABS((ARMReg)(S0 + i), (ARMReg)(S0 + i)); + } + } + VMOV(ar, MatchSize(Q0, ar)); + INFO_LOG(HLE, "PREFIXST: Partial ABS %i/%i! Slow fallback generated.", abs_mask, full_mask); + } + } + + if (negate_mask == full_mask) { + // TODO: elide the above VMOV when possible + VNEG(F_32, ar, ar); + } else if (negate_mask != 0) { + // Partial negate! I guess we build sign bits in another register + // and simply XOR. + if (negate_mask == 3) { + VNEG(F_32, D_0(ar), D_0(ar)); + } else { + // Horrifying fallback: Mov to Q0, negate, move back. + // TODO: Optimize for lower quads where we don't need to move. + VMOV(MatchSize(Q0, ar), ar); + for (int i = 0; i < n; i++) { + if (negate_mask & (1 << i)) { + VNEG((ARMReg)(S0 + i), (ARMReg)(S0 + i)); + } + } + VMOV(ar, MatchSize(Q0, ar)); + INFO_LOG(HLE, "PREFIXST: Partial Negate %i/%i! Slow fallback generated.", negate_mask, full_mask); + } + } + + // Insert constants where requested, and check negate! + for (int i = 0; i < n; i++) { + if (constants[i]) { + MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]); + VLD1_lane(F_32, ar, R0, i, true); + } + } + + return ar; +} + +Jit::DestARMReg Jit::NEONMapPrefixD(int vreg, VectorSize sz, int mapFlags) { + // Inverted from the actual bits, easier to reason about 1 == write + int writeMask = (~(js.prefixD >> 8)) & 0xF; + int n = GetNumVectorElements(sz); + int full_mask = (1 << n) - 1; + + DestARMReg dest; + dest.sz = sz; + if ((writeMask & full_mask) == full_mask) { + // No need to apply a write mask. + // Let's not make things complicated. + dest.rd = fpr.QMapReg(vreg, sz, mapFlags); + dest.backingRd = dest.rd; + } else { + // Allocate a temporary register. + ELOG("PREFIXD: Write mask allocated! %i/%i", writeMask, full_mask); + dest.rd = fpr.QAllocTemp(sz); + dest.backingRd = fpr.QMapReg(vreg, sz, mapFlags & ~MAP_NOINIT); // Force initialization of the backing reg. + } + return dest; +} + +void Jit::NEONApplyPrefixD(DestARMReg dest) { + // Apply clamps to dest.rd + int n = GetNumVectorElements(dest.sz); + + int sat1_mask = 0; + int sat3_mask = 0; + int full_mask = (1 << n) - 1; + for (int i = 0; i < n; i++) { + int sat = (js.prefixD >> (i * 2)) & 3; + if (sat == 1) + sat1_mask |= 1 << i; + if (sat == 3) + sat3_mask |= 1 << i; + } + + if (sat1_mask && sat3_mask) { + // Why would anyone do this? + ELOG("PREFIXD: Can't have both sat[0-1] and sat[-1-1] at the same time yet"); + } + + if (sat1_mask) { + if (sat1_mask != full_mask) { + ELOG("PREFIXD: Can't have partial sat1 mask yet (%i vs %i)", sat1_mask, full_mask); + } + if (IsD(dest.rd)) { + VMOV_immf(D0, 0.0); + VMOV_immf(D1, 1.0); + VMAX(F_32, dest.rd, dest.rd, D0); + VMIN(F_32, dest.rd, dest.rd, D1); + } else { + VMOV_immf(Q0, 1.0); + VMIN(F_32, dest.rd, dest.rd, Q0); + VMOV_immf(Q0, 0.0); + VMAX(F_32, dest.rd, dest.rd, Q0); + } + } + + if (sat3_mask && sat1_mask != full_mask) { + if (sat3_mask != full_mask) { + ELOG("PREFIXD: Can't have partial sat3 mask yet (%i vs %i)", sat3_mask, full_mask); + } + if (IsD(dest.rd)) { + VMOV_immf(D0, 0.0); + VMOV_immf(D1, 1.0); + VMAX(F_32, dest.rd, dest.rd, D0); + VMIN(F_32, dest.rd, dest.rd, D1); + } else { + VMOV_immf(Q0, 1.0); + VMIN(F_32, dest.rd, dest.rd, Q0); + VMOV_immf(Q0, -1.0); + VMAX(F_32, dest.rd, dest.rd, Q0); + } + } + + // Check for actual mask operation (unrelated to the "masks" above). + if (dest.backingRd != dest.rd) { + // This means that we need to apply the write mask, from rd to backingRd. + // What a pain. We can at least shortcut easy cases like half the register. + // And we can generate the masks easily with some of the crazy vector imm modes. (bits2bytes for example). + // So no need to load them from RAM. + int writeMask = (~(js.prefixD >> 8)) & 0xF; + + if (writeMask == 3) { + ILOG("Doing writemask = 3"); + VMOV(D_0(dest.rd), D_0(dest.backingRd)); + } else { + // TODO + ELOG("PREFIXD: Arbitrary write masks not supported (%i / %i)", writeMask, full_mask); + VMOV(dest.backingRd, dest.rd); + } + } +} + +Jit::MappedRegs Jit::NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes) { + MappedRegs regs; + if (applyPrefixes) { + regs.vs = NEONMapPrefixS(_VS, ssize, 0); + regs.vt = NEONMapPrefixT(_VT, tsize, 0); + } else { + regs.vs = fpr.QMapReg(_VS, ssize, 0); + regs.vt = fpr.QMapReg(_VT, ssize, 0); + } + + regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0 || GetVectorOverlap(_VD, dsize, _VT, ssize); + if (applyPrefixes) { + regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT)); + } else { + regs.vd.rd = fpr.QMapReg(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT)); + regs.vd.backingRd = regs.vd.rd; + regs.vd.sz = dsize; + } + return regs; +} + +Jit::MappedRegs Jit::NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes) { + MappedRegs regs; + if (applyPrefixes) { + regs.vs = NEONMapPrefixS(_VS, ssize, 0); + regs.vt = NEONMapPrefixT(_VT, tsize, 0); + } else { + regs.vs = fpr.QMapReg(_VS, ssize, 0); + regs.vt = fpr.QMapReg(_VT, ssize, 0); + } + regs.vd.rd = INVALID_REG; + regs.vd.sz = V_Invalid; + return regs; +} + +Jit::MappedRegs Jit::NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes) { + MappedRegs regs; + regs.vs = NEONMapPrefixS(_VS, ssize, 0); + regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0; + regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT)); + return regs; +} + +// Requires quad registers. +void Jit::NEONTranspose4x4(ARMReg cols[4]) { + // 0123 _\ 0426 + // 4567 / 1537 + VTRN(F_32, cols[0], cols[1]); + + // 89ab _\ 8cae + // cdef / 9dbf + VTRN(F_32, cols[2], cols[3]); + + // 04[26] 048c + // 15 37 -> 1537 + // [8c]ae 26ae + // 9d bf 9dbf + VSWP(D_1(cols[0]), D_0(cols[2])); + + // 04 8c 048c + // 15[37] -> 159d + // 26 ae 26ae + // [9d]bf 37bf + VSWP(D_1(cols[1]), D_0(cols[3])); +} + +} // namespace MIPSComp \ No newline at end of file diff --git a/Core/MIPS/ARM/ArmCompVFPUNEONUtil.h b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.h new file mode 100644 index 0000000000..3d278e064b --- /dev/null +++ b/Core/MIPS/ARM/ArmCompVFPUNEONUtil.h @@ -0,0 +1,20 @@ +#pragma once + +#include "Core/MIPS/ARM/ArmJit.h" +#include "Core/MIPS/ARM/ArmRegCache.h" + +namespace MIPSComp { + + +inline ARMReg MatchSize(ARMReg x, ARMReg target) { + if (IsQ(target) && IsQ(x)) + return x; + if (IsD(target) && IsD(x)) + return x; + if (IsD(target) && IsQ(x)) + return D_0(x); + // if (IsQ(target) && IsD(x)) + return (ARMReg)(D0 + (x - Q0) * 2); +} + +} \ No newline at end of file diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp index 48bd4a7082..adde925f9f 100644 --- a/Core/MIPS/ARM/ArmJit.cpp +++ b/Core/MIPS/ARM/ArmJit.cpp @@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() { useNEONVFPU = false; } -Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips) +Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips) { logBlocks = 0; dontLogBlocks = 0; @@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b) while (js.compiling) { gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages - fpr.SetCompilerPC(js.compilerPC); MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC); js.downcountAmount += MIPSGetInstructionCycleEstimate(inst); diff --git a/Core/MIPS/ARM/ArmJit.h b/Core/MIPS/ARM/ArmJit.h index a086550690..9579f49c09 100644 --- a/Core/MIPS/ARM/ArmJit.h +++ b/Core/MIPS/ARM/ArmJit.h @@ -17,6 +17,7 @@ #pragma once +#include "Common/CPUDetect.h" #include "Core/MIPS/JitCommon/JitState.h" #include "Core/MIPS/JitCommon/JitBlockCache.h" #include "Core/MIPS/ARM/ArmRegCache.h" @@ -240,6 +241,43 @@ private: } void GetVectorRegsPrefixD(u8 *regs, VectorSize sz, int vectorReg); + + // For NEON mappings, it will be easier to deal directly in ARM registers. + + ARMReg NEONMapPrefixST(int vfpuReg, VectorSize sz, u32 prefix, int mapFlags); + ARMReg NEONMapPrefixS(int vfpuReg, VectorSize sz, int mapFlags) { + return NEONMapPrefixST(vfpuReg, sz, js.prefixS, mapFlags); + } + ARMReg NEONMapPrefixT(int vfpuReg, VectorSize sz, int mapFlags) { + return NEONMapPrefixST(vfpuReg, sz, js.prefixT, mapFlags); + } + + struct DestARMReg { + ARMReg rd; + ARMReg backingRd; + VectorSize sz; + + operator ARMReg() const { return rd; } + }; + + struct MappedRegs { + ARMReg vs; + ARMReg vt; + DestARMReg vd; + bool overlap; + }; + + MappedRegs NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true); + MappedRegs NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true); + MappedRegs NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes = true); + + DestARMReg NEONMapPrefixD(int vfpuReg, VectorSize sz, int mapFlags); + void NEONApplyPrefixD(DestARMReg dest); + + // NEON utils + void NEONMaskToSize(ARMReg vs, VectorSize sz); + void NEONTranspose4x4(ARMReg cols[4]); + // Utils void SetR0ToEffectiveAddress(MIPSGPReg rs, s16 offset); void SetCCAndR0ForSafeAddress(MIPSGPReg rs, s16 offset, ARMReg tempReg, bool reverse = false); diff --git a/Core/MIPS/ARM/ArmRegCache.cpp b/Core/MIPS/ARM/ArmRegCache.cpp index faf8945b2b..6a1be703e6 100644 --- a/Core/MIPS/ARM/ArmRegCache.cpp +++ b/Core/MIPS/ARM/ArmRegCache.cpp @@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) { return mr[mipsReg].loc == ML_ARMREG_AS_PTR; } +bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) { + return mr[mipsReg].loc == ML_ARMREG; +} + void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) { // If we can do it with a simple Operand2, let's do that. Operand2 op2; diff --git a/Core/MIPS/ARM/ArmRegCache.h b/Core/MIPS/ARM/ArmRegCache.h index 9e62de5295..cdc33e3b0d 100644 --- a/Core/MIPS/ARM/ArmRegCache.h +++ b/Core/MIPS/ARM/ArmRegCache.h @@ -103,6 +103,7 @@ public: ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0); ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty. + bool IsMapped(MIPSGPReg reg); bool IsMappedAsPointer(MIPSGPReg reg); void MapInIn(MIPSGPReg rd, MIPSGPReg rs); diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.cpp b/Core/MIPS/ARM/ArmRegCacheFPU.cpp index 9c390ea491..8628b484de 100644 --- a/Core/MIPS/ARM/ArmRegCacheFPU.cpp +++ b/Core/MIPS/ARM/ArmRegCacheFPU.cpp @@ -16,14 +16,17 @@ // https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. #include + #include "base/logging.h" #include "Common/CPUDetect.h" +#include "Core/MIPS/MIPS.h" #include "Core/MIPS/ARM/ArmRegCacheFPU.h" +#include "Core/MIPS/ARM/ArmJit.h" #include "Core/MIPS/MIPSTables.h" using namespace ArmGen; -ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) { +ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) { if (cpu_info.bNEON) { numARMFpuReg_ = 32; } else { @@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init } } -void ArmRegCacheFPU::Init(ARMXEmitter *emitter) { - emit_ = emitter; -} - void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) { if (!initialReady) { SetupInitialRegs(); @@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() { mrInitial[i].spillLock = false; mrInitial[i].tempLock = false; } + for (int i = 0; i < MAX_ARMQUADS; i++) { + qr[i].isDirty = false; + qr[i].mipsVec = -1; + qr[i].sz = V_Invalid; + qr[i].spillLock = false; + qr[i].isTemp = false; + memset(qr[i].vregs, 0xff, 4); + } } -static const ARMReg *GetMIPSAllocationOrder(int &count) { +const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) { // We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things. static const ARMReg allocationOrder[] = { S2, S3, @@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) { S12, S13, S14, S15 }; - // With NEON, we have many more. - // In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU. - // VFPU will use NEON to do SIMD and it will be awkward to mix with FPU. + // VFP mapping + // VFPU registers and regular FP registers are mapped interchangably on top of the standard + // 16 FPU registers. + // NEON mapping + // We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers. + // VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON + // (or D16-D31 as doubles, but not relevant). + // Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads. + // We should attempt to map scalars to low Q registers and wider things to high registers, // as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be // able to use regular VFP instructions too. @@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) { S4, S5, S6, S7, // Q1 S8, S9, S10, S11, // Q2 S12, S13, S14, S15, // Q3 - S16, S17, S18, S19, // Q4 - S20, S21, S22, S23, // Q5 - S24, S25, S26, S27, // Q6 - S28, S29, S30, S31, // Q7 - // Q8-Q15 free for NEON tricks + // Q4-Q15 free for VFPU }; - if (cpu_info.bNEON) { + if (jo_->useNEONVFPU) { count = sizeof(allocationOrderNEON) / sizeof(const int); return allocationOrderNEON; } else { @@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) { } } +bool ArmRegCacheFPU::IsMapped(MIPSReg r) { + return mr[r].loc == ML_ARMREG; +} + ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) { + // INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags); + if (jo_->useNEONVFPU && mipsReg >= 32) { + ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC); + return S0; + } + pendingFlush = true; // Let's see if it's already mapped. If so we just need to update the dirty flag. // We don't need to check for ML_NOINIT because we assume that anyone who maps @@ -157,7 +176,7 @@ allocate: } // Uh oh, we have all them spilllocked.... - ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc); + ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC); return INVALID_REG; } @@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) { } void ArmRegCacheFPU::FlushArmReg(ARMReg r) { - int reg = r - S0; - if (ar[reg].mipsReg == -1) { - // Nothing to do, reg not mapped. - return; - } - if (ar[reg].mipsReg != -1) { - if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) { - //INFO_LOG(JIT, "Flushing ARM reg %i", reg); - emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg)); + if (r >= S0 && r <= S31) { + int reg = r - S0; + if (ar[reg].mipsReg == -1) { + // Nothing to do, reg not mapped. + return; } - // IMMs won't be in an ARM reg. - mr[ar[reg].mipsReg].loc = ML_MEM; - mr[ar[reg].mipsReg].reg = INVALID_REG; - } else { - ERROR_LOG(JIT, "Dirty but no mipsreg?"); + if (ar[reg].mipsReg != -1) { + if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) + { + //INFO_LOG(JIT, "Flushing ARM reg %i", reg); + emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg)); + } + // IMMs won't be in an ARM reg. + mr[ar[reg].mipsReg].loc = ML_MEM; + mr[ar[reg].mipsReg].reg = INVALID_REG; + } else { + ERROR_LOG(JIT, "Dirty but no mipsreg?"); + } + ar[reg].isDirty = false; + ar[reg].mipsReg = -1; + } else if (r >= D0 && r <= D31) { + // TODO: Convert to S regs and flush them individually. + } else if (r >= Q0 && r <= Q15) { + int quad = r - Q0; + QFlush(r); } - ar[reg].isDirty = false; - ar[reg].mipsReg = -1; } +void ArmRegCacheFPU::FlushV(MIPSReg r) { + FlushR(r + 32); +} + +/* +void ArmRegCacheFPU::FlushQWithV(MIPSReg r) { + // Look for it in all the quads. If it's in any, flush that quad clean. + int flushCount = 0; + for (int i = 0; i < MAX_ARMQUADS; i++) { + if (qr[i].sz == V_Invalid) + continue; + + int n = qr[i].sz; + bool flushThis = false; + for (int j = 0; j < n; j++) { + if (qr[i].vregs[j] == r) { + flushThis = true; + } + } + + if (flushThis) { + QFlush(i); + flushCount++; + } + } + + if (flushCount > 1) { + WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r); + } +} +*/ + void ArmRegCacheFPU::FlushR(MIPSReg r) { switch (mr[r].loc) { case ML_IMM: @@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) { if (mr[r].reg == (int)INVALID_REG) { ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg"); } - if (ar[mr[r].reg].isDirty) { - //INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg); - emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r)); - ar[mr[r].reg].isDirty = false; + + if (mr[r].reg >= Q0 && mr[r].reg <= Q15) { + // This should happen rarely, but occasionally we need to flush a single stray + // mipsreg that's been part of a quad. + int quad = mr[r].reg - Q0; + if (qr[quad].isDirty) { + WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1); + emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true); + } + } else { + if (ar[mr[r].reg].isDirty) { + //INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg); + emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r)); + ar[mr[r].reg].isDirty = false; + } + ar[mr[r].reg].mipsReg = -1; } - ar[mr[r].reg].mipsReg = -1; break; case ML_MEM: @@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() { return 16; } +// Scalar only. Need a similar one for sequential Q vectors. int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) { int c = 1; int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg); @@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() { DiscardR(i); } + // Flush quads! + // These could also use sequential detection. + for (int i = 4; i < MAX_ARMQUADS; i++) { + QFlush(i); + } + // Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are // sequential. This is necessary because we store VFPU registers in a staggered order to get // columns sequential (most VFPU math in nearly all games is in columns, not rows). @@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const { } int ArmRegCacheFPU::GetTempR() { + if (jo_->useNEONVFPU) { + ERROR_LOG(JIT, "VFP temps not allowed in NEON mode"); + return 0; + } pendingFlush = true; for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) { if (mr[r].loc == ML_MEM && !mr[r].tempLock) { @@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) { // This is actually pretty slow with all the 160 regs... void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() { - for (int i = 0; i < NUM_MIPSFPUREG; i++) + for (int i = 0; i < NUM_MIPSFPUREG; i++) { mr[i].spillLock = false; - for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) + } + for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) { DiscardR(i); + } + for (int i = 0; i < MAX_ARMQUADS; i++) { + qr[i].spillLock = false; + if (qr[i].isTemp) { + qr[i].isTemp = false; + qr[i].sz = V_Invalid; + } + } } ARMReg ArmRegCacheFPU::R(int mipsReg) { @@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) { return (ARMReg)(mr[mipsReg].reg + S0); } else { if (mipsReg < 32) { - ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_)); + ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC)); } else if (mipsReg < 32 + 128) { - ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_)); + ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC)); } else { - ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_)); + ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC)); } return INVALID_REG; // BAAAD } } + +inline ARMReg QuadAsD(int quad) { + return (ARMReg)(D0 + quad * 2); +} + +inline ARMReg QuadAsQ(int quad) { + return (ARMReg)(Q0 + quad); +} + +bool MappableQ(int quad) { + return quad >= 4; +} + +void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) { + ERROR_LOG(JIT, "QLoad4x4 not implemented"); + // TODO +} + +void ArmRegCacheFPU::QFlush(int quad) { + if (!MappableQ(quad)) { + ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad); + return; + } + + if (qr[quad].isDirty && !qr[quad].isTemp) { + INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz)); + + ARMReg q = QuadAsQ(quad); + // Unlike reads, when writing to the register file we need to be careful to write the correct + // number of floats. + + switch (qr[quad].sz) { + case V_Single: + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + // WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC); + break; + case V_Pair: + if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) { + // Can combine, it's a column! + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + // WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1); + emit_->VST1_lane(F_32, q, R0, 1, true); + } + break; + case V_Triple: + if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable + emit_->VST1_lane(F_32, q, R0, 2, true); + } else { + // WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1); + emit_->VST1_lane(F_32, q, R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1); + emit_->VST1_lane(F_32, q, R0, 2, true); + } + break; + case V_Quad: + if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + // WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1); + emit_->VST1_lane(F_32, q, R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1); + emit_->VST1_lane(F_32, q, R0, 2, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1); + emit_->VST1_lane(F_32, q, R0, 3, true); + } + break; + default: + ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz); + break; + } + + qr[quad].isDirty = false; + + int n = GetNumVectorElements(qr[quad].sz); + for (int i = 0; i < n; i++) { + int vr = qr[quad].vregs[i]; + if (vr < 0 || vr > 128) { + ERROR_LOG(JIT, "Bad vr %i", vr); + } + FPURegMIPS &m = mr[32 + vr]; + m.loc = ML_MEM; + m.lane = -1; + m.reg = -1; + } + + } else { + if (qr[quad].isTemp) { + WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp); + } + } + + qr[quad].isTemp = false; + qr[quad].mipsVec = -1; + qr[quad].sz = V_Invalid; + memset(qr[quad].vregs, 0xFF, 4); +} + +int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) { + // Search for a free quad. A quad is free if the first register in it is free. + int quad = -1; + + for (int i = 0; i < count; i++) { + int q = (i + start) & 15; + + if (!MappableQ(q)) + continue; + + // Don't steal temp quads! + if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) { + // INFO_LOG(JIT, "Free quad: %i", q); + // Oh yeah! Free quad! + return q; + } + } + + // Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some + // sort of age. + int bestQuad = -1; + int bestScore = -1; + for (int i = 0; i < count; i++) { + int q = (i + start) & 15; + + if (!MappableQ(q)) + continue; + if (qr[q].spillLock) + continue; + if (qr[q].isTemp) + continue; + + int score = 0; + if (!qr[q].isDirty) { + score += 5; + } + + if (score > bestScore) { + bestQuad = q; + bestScore = score; + } + } + + if (bestQuad == -1) { + ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!"); + return -1; + } else { + INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason"); + QFlush(bestQuad); + return bestQuad; + } +} + +ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) { + int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps + if (q < 0) { + ERROR_LOG(JIT, "Failed to allocate temp quad"); + q = 0; + } + qr[q].spillLock = true; + qr[q].isTemp = true; + qr[q].sz = sz; + qr[q].isDirty = false; // doesn't matter + + INFO_LOG(JIT, "Allocated temp quad %i", q); + + if (sz == V_Single || sz == V_Pair) { + return D_0(ARMReg(Q0 + q)); + } else { + return ARMReg(Q0 + q); + } +} + +bool ArmRegCacheFPU::Consecutive(int v1, int v2) const { + return (voffset[v1] + 1) == voffset[v2]; +} + +bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const { + return Consecutive(v1, v2) && Consecutive(v2, v3); +} + +bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const { + return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4); +} + +void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) { + u8 vregs[4]; + if (flags & MAP_MTX_TRANSPOSED) { + GetMatrixRows(matrix, mz, vregs); + } else { + GetMatrixColumns(matrix, mz, vregs); + } + + // TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load. + int n = GetMatrixSide(mz); + VectorSize vsz = GetVectorSize(mz); + for (int i = 0; i < n; i++) { + regs[i] = QMapReg(vregs[i], vsz, flags); + } +} + +ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) { + qTime_++; + + int n = GetNumVectorElements(sz); + u8 vregs[4]; + GetVectorRegs(vregs, sz, vreg); + + // Range of registers to consider + int start = 0; + int count = 16; + + if (flags & MAP_PREFER_HIGH) { + start = 8; + } else if (flags & MAP_PREFER_LOW) { + start = 4; + } else if (flags & MAP_FORCE_LOW) { + start = 4; + count = 4; + } else if (flags & MAP_FORCE_HIGH) { + start = 8; + count = 8; + } + + // Let's check if they are all mapped in a quad somewhere. + // At the same time, check for the quad already being mapped. + // Later we can check for possible transposes as well. + + // First just loop over all registers. If it's here and not in range, or overlapped, kick. + std::vector quadsToFlush; + for (int i = 0; i < 16; i++) { + int q = (i + start) & 15; + if (!MappableQ(q)) + continue; + + // Skip unmapped quads. + if (qr[q].sz == V_Invalid) + continue; + + // Check if completely there already. If so, set spill-lock, transfer dirty flag and exit. + if (vreg == qr[q].mipsVec && sz == qr[q].sz) { + if (i < count) { + INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz); + qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY); + qr[q].spillLock = true; + + // Sanity check vregs + for (int i = 0; i < n; i++) { + if (vregs[i] != qr[q].vregs[i]) { + ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]); + } + } + + return (ARMReg)(Q0 + q); + } else { + INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count); + quadsToFlush.push_back(q); + continue; + } + } + + // Check for any overlap. Overlap == flush. + int origN = GetNumVectorElements(qr[q].sz); + for (int a = 0; a < n; a++) { + for (int b = 0; b < origN; b++) { + if (vregs[a] == qr[q].vregs[b]) { + quadsToFlush.push_back(q); + goto doubleBreak; + } + } + } + doubleBreak: + ; + } + + // We didn't find the extra register, but we got a list of regs to flush. Flush 'em. + // Here we can check for opportunities to do a "transpose-flush" of row vectors, etc. + if (!quadsToFlush.empty()) { + ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size()); + } + for (size_t i = 0; i < quadsToFlush.size(); i++) { + QFlush(quadsToFlush[i]); + } + + // Find where we want to map it, obeying the constraints we gave. + int quad = QGetFreeQuad(start, count, "mapping"); + + // If parts of our register are elsewhere, and we are dirty, we need to flush them + // before we reload in a new location. + // This may be problematic if inputs overlap irregularly with output, say: + // vdot S700, R000, C000 + // It might still work by accident... + if (flags & MAP_DIRTY) { + for (int i = 0; i < n; i++) { + FlushV(vregs[i]); + } + } + + qr[quad].sz = sz; + qr[quad].mipsVec = vreg; + + if (!(flags & MAP_NOINIT)) { + // Okay, now we will try to load the whole thing in one go. This is possible + // if it's a row and easy if it's a single. + // Rows are rare, columns are common - but thanks to our register reordering, + // columns are actually in-order in memory. + switch (sz) { + case V_Single: + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + break; + case V_Pair: + if (Consecutive(vregs[0], vregs[1])) { + // Can combine, it's a column! + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true); + } + break; + case V_Triple: + if (Consecutive(vregs[0], vregs[1], vregs[2])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true); + } else { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true); + } + break; + case V_Quad: + if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true); + } + break; + default: + ; + } + } + + // OK, let's fill out the arrays to confirm that we have grabbed these registers. + for (int i = 0; i < n; i++) { + int mipsReg = 32 + vregs[i]; + mr[mipsReg].loc = ML_ARMREG; + mr[mipsReg].reg = QuadAsQ(quad); + mr[mipsReg].lane = i; + qr[quad].vregs[i] = vregs[i]; + } + qr[quad].isDirty = (flags & MAP_DIRTY) != 0; + qr[quad].spillLock = true; + + INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty); + if (sz == V_Single || sz == V_Pair) { + return D_0(QuadAsQ(quad)); + } else { + return QuadAsQ(quad); + } +} + diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.h b/Core/MIPS/ARM/ArmRegCacheFPU.h index 15e1c7de57..1e796a2a19 100644 --- a/Core/MIPS/ARM/ArmRegCacheFPU.h +++ b/Core/MIPS/ARM/ArmRegCacheFPU.h @@ -33,28 +33,56 @@ enum { TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS, }; +enum { + MAP_READ = 0, + MAP_MTX_TRANSPOSED = 16, + MAP_PREFER_LOW = 16, + MAP_PREFER_HIGH = 32, + + // Force is not yet correctly implemented, if the reg is already mapped it will not move + MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15) + MAP_FORCE_HIGH = 128, // Only map Q8-Q15 +}; + struct FPURegARM { int mipsReg; // if -1, no mipsreg attached. bool isDirty; // Should the register be written back? }; +struct FPURegQuad { + int mipsVec; + VectorSize sz; + u8 vregs[4]; + bool isDirty; + bool spillLock; + bool isTemp; +}; + struct FPURegMIPS { // Where is this MIPS register? RegMIPSLoc loc; // Data (only one of these is used, depending on loc. Could make a union). int reg; + int lane; + bool spillLock; // if true, this register cannot be spilled. bool tempLock; // If loc == ML_MEM, it's back in its location in the CPU context struct. }; +namespace MIPSComp { + struct ArmJitOptions; + struct JitState; +} + class ArmRegCacheFPU { public: - ArmRegCacheFPU(MIPSState *mips); + ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo); ~ArmRegCacheFPU() {} void Init(ARMXEmitter *emitter); + void Start(MIPSAnalyst::AnalysisResults &stats); // Protect the arm register containing a MIPS register from spilling, to ensure that @@ -80,9 +108,11 @@ public: void MapDirty(MIPSReg rd); void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true); void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true); + bool IsMapped(MIPSReg r); void FlushArmReg(ARMReg r); void FlushR(MIPSReg r); void DiscardR(MIPSReg r); + ARMReg R(int preg); // Returns a cached register // VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode! void MapRegV(int vreg, int flags = 0); @@ -90,18 +120,41 @@ public: void MapInInV(int rt, int rs); void MapDirtyInV(int rd, int rs, bool avoidLoad = true); void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true); - bool IsTempX(ARMReg r) const; - MIPSReg GetTempR(); + bool IsTempX(ARMReg r) const; MIPSReg GetTempV() { return GetTempR() - 32; } + // VFPU registers as single VFP registers. + ARMReg V(int vreg) { return R(vreg + 32); } int FlushGetSequential(int a, int maxArmReg); void FlushAll(); - ARMReg R(int preg); // Returns a cached register + // This one is allowed at any point. + void FlushV(MIPSReg r); + + // VFPU registers mapped to match NEON quads (and doubles, for pairs and singles) + // Here we return the ARM register directly instead of providing a "V" accessor + // and so on. Might switch to this model for the other regallocs later. + + // Quad mapping does NOT look into the ar array. Instead we use the qr array to keep + // track of what's in each quad. + + // Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types. + // Need to explicitly allow spilling to get spilling. + ARMReg QMapReg(int vreg, VectorSize sz, int flags); + + // TODO + // Maps a matrix as a set of columns (yes, even transposed ones, always columns + // as those are faster to load/flush). When possible it will map into consecutive + // quad registers, enabling blazing-fast full-matrix loads, transposed or not. + void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags); + + ARMReg QAllocTemp(VectorSize sz); - // VFPU registers as single VFP registers - ARMReg V(int vreg) { return R(vreg + 32); } + void QAllowSpill(int quad); + void QFlush(int quad); + void QLoad4x4(MIPSGPReg regPtr, int vquads[4]); + //void FlushQWithV(MIPSReg r); // NOTE: These require you to release spill locks manually! void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags); @@ -112,31 +165,43 @@ public: void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; } - // For better log output only. - void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; } - int GetMipsRegOffset(MIPSReg r); + +private: + bool Consecutive(int v1, int v2) const; + bool Consecutive(int v1, int v2, int v3) const; + bool Consecutive(int v1, int v2, int v3, int v4) const; + + MIPSReg GetTempR(); + const ARMReg *GetMIPSAllocationOrder(int &count); int GetMipsRegOffsetV(MIPSReg r) { return GetMipsRegOffset(r + 32); } + // This one WILL get a free quad as long as you haven't spill-locked them all. + int QGetFreeQuad(int start, int count, const char *reason); int GetNumARMFPURegs(); -private: void SetupInitialRegs(); MIPSState *mips_; ARMXEmitter *emit_; - u32 compilerPC_; + MIPSComp::JitState *js_; + MIPSComp::ArmJitOptions *jo_; int numARMFpuReg_; + int qTime_; enum { - MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON + // With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers + // are individually mappable though. + MAX_ARMFPUREG = 32, + MAX_ARMQUADS = 16, NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS, }; FPURegARM ar[MAX_ARMFPUREG]; FPURegMIPS mr[NUM_MIPSFPUREG]; + FPURegQuad qr[MAX_ARMQUADS]; FPURegMIPS *vr; bool pendingFlush; diff --git a/android/jni/Android.mk b/android/jni/Android.mk index 1a8b82ff8e..8dbbfe2e94 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -64,6 +64,7 @@ ARCH_FILES := \ $(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \ $(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \ $(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \ + $(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \ $(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \ $(SRC)/Core/MIPS/ARM/ArmAsm.cpp \ $(SRC)/Core/MIPS/ARM/ArmJit.cpp \ @@ -84,6 +85,7 @@ ARCH_FILES := \ $(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \ $(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \ $(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \ + $(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \ $(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \ $(SRC)/Core/MIPS/ARM/ArmAsm.cpp \ $(SRC)/Core/MIPS/ARM/ArmJit.cpp \