From d98bde8e503bda7decffcb9285eb3ce19be3ad9f Mon Sep 17 00:00:00 2001 From: Henrik Rydgard Date: Sat, 6 Dec 2014 12:26:58 +0100 Subject: [PATCH] Merge the RegCache changes from the old neon-vfpu branch --- Core/MIPS/ARM/ArmJit.cpp | 3 +- Core/MIPS/ARM/ArmRegCache.cpp | 4 + Core/MIPS/ARM/ArmRegCache.h | 1 + Core/MIPS/ARM/ArmRegCacheFPU.cpp | 563 ++++++++++++++++++++++++++++--- Core/MIPS/ARM/ArmRegCacheFPU.h | 88 ++++- 5 files changed, 603 insertions(+), 56 deletions(-) diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp index 48bd4a7082..adde925f9f 100644 --- a/Core/MIPS/ARM/ArmJit.cpp +++ b/Core/MIPS/ARM/ArmJit.cpp @@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() { useNEONVFPU = false; } -Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips) +Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips) { logBlocks = 0; dontLogBlocks = 0; @@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b) while (js.compiling) { gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages - fpr.SetCompilerPC(js.compilerPC); MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC); js.downcountAmount += MIPSGetInstructionCycleEstimate(inst); diff --git a/Core/MIPS/ARM/ArmRegCache.cpp b/Core/MIPS/ARM/ArmRegCache.cpp index faf8945b2b..6a1be703e6 100644 --- a/Core/MIPS/ARM/ArmRegCache.cpp +++ b/Core/MIPS/ARM/ArmRegCache.cpp @@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) { return mr[mipsReg].loc == ML_ARMREG_AS_PTR; } +bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) { + return mr[mipsReg].loc == ML_ARMREG; +} + void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) { // If we can do it with a simple Operand2, let's do that. Operand2 op2; diff --git a/Core/MIPS/ARM/ArmRegCache.h b/Core/MIPS/ARM/ArmRegCache.h index 9e62de5295..cdc33e3b0d 100644 --- a/Core/MIPS/ARM/ArmRegCache.h +++ b/Core/MIPS/ARM/ArmRegCache.h @@ -103,6 +103,7 @@ public: ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0); ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty. + bool IsMapped(MIPSGPReg reg); bool IsMappedAsPointer(MIPSGPReg reg); void MapInIn(MIPSGPReg rd, MIPSGPReg rs); diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.cpp b/Core/MIPS/ARM/ArmRegCacheFPU.cpp index 9c390ea491..8628b484de 100644 --- a/Core/MIPS/ARM/ArmRegCacheFPU.cpp +++ b/Core/MIPS/ARM/ArmRegCacheFPU.cpp @@ -16,14 +16,17 @@ // https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. #include + #include "base/logging.h" #include "Common/CPUDetect.h" +#include "Core/MIPS/MIPS.h" #include "Core/MIPS/ARM/ArmRegCacheFPU.h" +#include "Core/MIPS/ARM/ArmJit.h" #include "Core/MIPS/MIPSTables.h" using namespace ArmGen; -ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) { +ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) { if (cpu_info.bNEON) { numARMFpuReg_ = 32; } else { @@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init } } -void ArmRegCacheFPU::Init(ARMXEmitter *emitter) { - emit_ = emitter; -} - void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) { if (!initialReady) { SetupInitialRegs(); @@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() { mrInitial[i].spillLock = false; mrInitial[i].tempLock = false; } + for (int i = 0; i < MAX_ARMQUADS; i++) { + qr[i].isDirty = false; + qr[i].mipsVec = -1; + qr[i].sz = V_Invalid; + qr[i].spillLock = false; + qr[i].isTemp = false; + memset(qr[i].vregs, 0xff, 4); + } } -static const ARMReg *GetMIPSAllocationOrder(int &count) { +const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) { // We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things. static const ARMReg allocationOrder[] = { S2, S3, @@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) { S12, S13, S14, S15 }; - // With NEON, we have many more. - // In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU. - // VFPU will use NEON to do SIMD and it will be awkward to mix with FPU. + // VFP mapping + // VFPU registers and regular FP registers are mapped interchangably on top of the standard + // 16 FPU registers. + // NEON mapping + // We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers. + // VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON + // (or D16-D31 as doubles, but not relevant). + // Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads. + // We should attempt to map scalars to low Q registers and wider things to high registers, // as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be // able to use regular VFP instructions too. @@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) { S4, S5, S6, S7, // Q1 S8, S9, S10, S11, // Q2 S12, S13, S14, S15, // Q3 - S16, S17, S18, S19, // Q4 - S20, S21, S22, S23, // Q5 - S24, S25, S26, S27, // Q6 - S28, S29, S30, S31, // Q7 - // Q8-Q15 free for NEON tricks + // Q4-Q15 free for VFPU }; - if (cpu_info.bNEON) { + if (jo_->useNEONVFPU) { count = sizeof(allocationOrderNEON) / sizeof(const int); return allocationOrderNEON; } else { @@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) { } } +bool ArmRegCacheFPU::IsMapped(MIPSReg r) { + return mr[r].loc == ML_ARMREG; +} + ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) { + // INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags); + if (jo_->useNEONVFPU && mipsReg >= 32) { + ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC); + return S0; + } + pendingFlush = true; // Let's see if it's already mapped. If so we just need to update the dirty flag. // We don't need to check for ML_NOINIT because we assume that anyone who maps @@ -157,7 +176,7 @@ allocate: } // Uh oh, we have all them spilllocked.... - ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc); + ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC); return INVALID_REG; } @@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) { } void ArmRegCacheFPU::FlushArmReg(ARMReg r) { - int reg = r - S0; - if (ar[reg].mipsReg == -1) { - // Nothing to do, reg not mapped. - return; - } - if (ar[reg].mipsReg != -1) { - if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) { - //INFO_LOG(JIT, "Flushing ARM reg %i", reg); - emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg)); + if (r >= S0 && r <= S31) { + int reg = r - S0; + if (ar[reg].mipsReg == -1) { + // Nothing to do, reg not mapped. + return; } - // IMMs won't be in an ARM reg. - mr[ar[reg].mipsReg].loc = ML_MEM; - mr[ar[reg].mipsReg].reg = INVALID_REG; - } else { - ERROR_LOG(JIT, "Dirty but no mipsreg?"); + if (ar[reg].mipsReg != -1) { + if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) + { + //INFO_LOG(JIT, "Flushing ARM reg %i", reg); + emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg)); + } + // IMMs won't be in an ARM reg. + mr[ar[reg].mipsReg].loc = ML_MEM; + mr[ar[reg].mipsReg].reg = INVALID_REG; + } else { + ERROR_LOG(JIT, "Dirty but no mipsreg?"); + } + ar[reg].isDirty = false; + ar[reg].mipsReg = -1; + } else if (r >= D0 && r <= D31) { + // TODO: Convert to S regs and flush them individually. + } else if (r >= Q0 && r <= Q15) { + int quad = r - Q0; + QFlush(r); } - ar[reg].isDirty = false; - ar[reg].mipsReg = -1; } +void ArmRegCacheFPU::FlushV(MIPSReg r) { + FlushR(r + 32); +} + +/* +void ArmRegCacheFPU::FlushQWithV(MIPSReg r) { + // Look for it in all the quads. If it's in any, flush that quad clean. + int flushCount = 0; + for (int i = 0; i < MAX_ARMQUADS; i++) { + if (qr[i].sz == V_Invalid) + continue; + + int n = qr[i].sz; + bool flushThis = false; + for (int j = 0; j < n; j++) { + if (qr[i].vregs[j] == r) { + flushThis = true; + } + } + + if (flushThis) { + QFlush(i); + flushCount++; + } + } + + if (flushCount > 1) { + WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r); + } +} +*/ + void ArmRegCacheFPU::FlushR(MIPSReg r) { switch (mr[r].loc) { case ML_IMM: @@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) { if (mr[r].reg == (int)INVALID_REG) { ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg"); } - if (ar[mr[r].reg].isDirty) { - //INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg); - emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r)); - ar[mr[r].reg].isDirty = false; + + if (mr[r].reg >= Q0 && mr[r].reg <= Q15) { + // This should happen rarely, but occasionally we need to flush a single stray + // mipsreg that's been part of a quad. + int quad = mr[r].reg - Q0; + if (qr[quad].isDirty) { + WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1); + emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true); + } + } else { + if (ar[mr[r].reg].isDirty) { + //INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg); + emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r)); + ar[mr[r].reg].isDirty = false; + } + ar[mr[r].reg].mipsReg = -1; } - ar[mr[r].reg].mipsReg = -1; break; case ML_MEM: @@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() { return 16; } +// Scalar only. Need a similar one for sequential Q vectors. int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) { int c = 1; int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg); @@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() { DiscardR(i); } + // Flush quads! + // These could also use sequential detection. + for (int i = 4; i < MAX_ARMQUADS; i++) { + QFlush(i); + } + // Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are // sequential. This is necessary because we store VFPU registers in a staggered order to get // columns sequential (most VFPU math in nearly all games is in columns, not rows). @@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const { } int ArmRegCacheFPU::GetTempR() { + if (jo_->useNEONVFPU) { + ERROR_LOG(JIT, "VFP temps not allowed in NEON mode"); + return 0; + } pendingFlush = true; for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) { if (mr[r].loc == ML_MEM && !mr[r].tempLock) { @@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) { // This is actually pretty slow with all the 160 regs... void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() { - for (int i = 0; i < NUM_MIPSFPUREG; i++) + for (int i = 0; i < NUM_MIPSFPUREG; i++) { mr[i].spillLock = false; - for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) + } + for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) { DiscardR(i); + } + for (int i = 0; i < MAX_ARMQUADS; i++) { + qr[i].spillLock = false; + if (qr[i].isTemp) { + qr[i].isTemp = false; + qr[i].sz = V_Invalid; + } + } } ARMReg ArmRegCacheFPU::R(int mipsReg) { @@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) { return (ARMReg)(mr[mipsReg].reg + S0); } else { if (mipsReg < 32) { - ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_)); + ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC)); } else if (mipsReg < 32 + 128) { - ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_)); + ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC)); } else { - ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_)); + ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC)); } return INVALID_REG; // BAAAD } } + +inline ARMReg QuadAsD(int quad) { + return (ARMReg)(D0 + quad * 2); +} + +inline ARMReg QuadAsQ(int quad) { + return (ARMReg)(Q0 + quad); +} + +bool MappableQ(int quad) { + return quad >= 4; +} + +void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) { + ERROR_LOG(JIT, "QLoad4x4 not implemented"); + // TODO +} + +void ArmRegCacheFPU::QFlush(int quad) { + if (!MappableQ(quad)) { + ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad); + return; + } + + if (qr[quad].isDirty && !qr[quad].isTemp) { + INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz)); + + ARMReg q = QuadAsQ(quad); + // Unlike reads, when writing to the register file we need to be careful to write the correct + // number of floats. + + switch (qr[quad].sz) { + case V_Single: + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + // WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC); + break; + case V_Pair: + if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) { + // Can combine, it's a column! + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + // WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1); + emit_->VST1_lane(F_32, q, R0, 1, true); + } + break; + case V_Triple: + if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable + emit_->VST1_lane(F_32, q, R0, 2, true); + } else { + // WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1); + emit_->VST1_lane(F_32, q, R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1); + emit_->VST1_lane(F_32, q, R0, 2, true); + } + break; + case V_Quad: + if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + // WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1); + emit_->VST1_lane(F_32, q, R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1); + emit_->VST1_lane(F_32, q, R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1); + emit_->VST1_lane(F_32, q, R0, 2, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1); + emit_->VST1_lane(F_32, q, R0, 3, true); + } + break; + default: + ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz); + break; + } + + qr[quad].isDirty = false; + + int n = GetNumVectorElements(qr[quad].sz); + for (int i = 0; i < n; i++) { + int vr = qr[quad].vregs[i]; + if (vr < 0 || vr > 128) { + ERROR_LOG(JIT, "Bad vr %i", vr); + } + FPURegMIPS &m = mr[32 + vr]; + m.loc = ML_MEM; + m.lane = -1; + m.reg = -1; + } + + } else { + if (qr[quad].isTemp) { + WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp); + } + } + + qr[quad].isTemp = false; + qr[quad].mipsVec = -1; + qr[quad].sz = V_Invalid; + memset(qr[quad].vregs, 0xFF, 4); +} + +int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) { + // Search for a free quad. A quad is free if the first register in it is free. + int quad = -1; + + for (int i = 0; i < count; i++) { + int q = (i + start) & 15; + + if (!MappableQ(q)) + continue; + + // Don't steal temp quads! + if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) { + // INFO_LOG(JIT, "Free quad: %i", q); + // Oh yeah! Free quad! + return q; + } + } + + // Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some + // sort of age. + int bestQuad = -1; + int bestScore = -1; + for (int i = 0; i < count; i++) { + int q = (i + start) & 15; + + if (!MappableQ(q)) + continue; + if (qr[q].spillLock) + continue; + if (qr[q].isTemp) + continue; + + int score = 0; + if (!qr[q].isDirty) { + score += 5; + } + + if (score > bestScore) { + bestQuad = q; + bestScore = score; + } + } + + if (bestQuad == -1) { + ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!"); + return -1; + } else { + INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason"); + QFlush(bestQuad); + return bestQuad; + } +} + +ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) { + int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps + if (q < 0) { + ERROR_LOG(JIT, "Failed to allocate temp quad"); + q = 0; + } + qr[q].spillLock = true; + qr[q].isTemp = true; + qr[q].sz = sz; + qr[q].isDirty = false; // doesn't matter + + INFO_LOG(JIT, "Allocated temp quad %i", q); + + if (sz == V_Single || sz == V_Pair) { + return D_0(ARMReg(Q0 + q)); + } else { + return ARMReg(Q0 + q); + } +} + +bool ArmRegCacheFPU::Consecutive(int v1, int v2) const { + return (voffset[v1] + 1) == voffset[v2]; +} + +bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const { + return Consecutive(v1, v2) && Consecutive(v2, v3); +} + +bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const { + return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4); +} + +void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) { + u8 vregs[4]; + if (flags & MAP_MTX_TRANSPOSED) { + GetMatrixRows(matrix, mz, vregs); + } else { + GetMatrixColumns(matrix, mz, vregs); + } + + // TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load. + int n = GetMatrixSide(mz); + VectorSize vsz = GetVectorSize(mz); + for (int i = 0; i < n; i++) { + regs[i] = QMapReg(vregs[i], vsz, flags); + } +} + +ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) { + qTime_++; + + int n = GetNumVectorElements(sz); + u8 vregs[4]; + GetVectorRegs(vregs, sz, vreg); + + // Range of registers to consider + int start = 0; + int count = 16; + + if (flags & MAP_PREFER_HIGH) { + start = 8; + } else if (flags & MAP_PREFER_LOW) { + start = 4; + } else if (flags & MAP_FORCE_LOW) { + start = 4; + count = 4; + } else if (flags & MAP_FORCE_HIGH) { + start = 8; + count = 8; + } + + // Let's check if they are all mapped in a quad somewhere. + // At the same time, check for the quad already being mapped. + // Later we can check for possible transposes as well. + + // First just loop over all registers. If it's here and not in range, or overlapped, kick. + std::vector quadsToFlush; + for (int i = 0; i < 16; i++) { + int q = (i + start) & 15; + if (!MappableQ(q)) + continue; + + // Skip unmapped quads. + if (qr[q].sz == V_Invalid) + continue; + + // Check if completely there already. If so, set spill-lock, transfer dirty flag and exit. + if (vreg == qr[q].mipsVec && sz == qr[q].sz) { + if (i < count) { + INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz); + qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY); + qr[q].spillLock = true; + + // Sanity check vregs + for (int i = 0; i < n; i++) { + if (vregs[i] != qr[q].vregs[i]) { + ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]); + } + } + + return (ARMReg)(Q0 + q); + } else { + INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count); + quadsToFlush.push_back(q); + continue; + } + } + + // Check for any overlap. Overlap == flush. + int origN = GetNumVectorElements(qr[q].sz); + for (int a = 0; a < n; a++) { + for (int b = 0; b < origN; b++) { + if (vregs[a] == qr[q].vregs[b]) { + quadsToFlush.push_back(q); + goto doubleBreak; + } + } + } + doubleBreak: + ; + } + + // We didn't find the extra register, but we got a list of regs to flush. Flush 'em. + // Here we can check for opportunities to do a "transpose-flush" of row vectors, etc. + if (!quadsToFlush.empty()) { + ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size()); + } + for (size_t i = 0; i < quadsToFlush.size(); i++) { + QFlush(quadsToFlush[i]); + } + + // Find where we want to map it, obeying the constraints we gave. + int quad = QGetFreeQuad(start, count, "mapping"); + + // If parts of our register are elsewhere, and we are dirty, we need to flush them + // before we reload in a new location. + // This may be problematic if inputs overlap irregularly with output, say: + // vdot S700, R000, C000 + // It might still work by accident... + if (flags & MAP_DIRTY) { + for (int i = 0; i < n; i++) { + FlushV(vregs[i]); + } + } + + qr[quad].sz = sz; + qr[quad].mipsVec = vreg; + + if (!(flags & MAP_NOINIT)) { + // Okay, now we will try to load the whole thing in one go. This is possible + // if it's a row and easy if it's a single. + // Rows are rare, columns are common - but thanks to our register reordering, + // columns are actually in-order in memory. + switch (sz) { + case V_Single: + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + break; + case V_Pair: + if (Consecutive(vregs[0], vregs[1])) { + // Can combine, it's a column! + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true); + } + break; + case V_Triple: + if (Consecutive(vregs[0], vregs[1], vregs[2])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true); + } else { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true); + } + break; + case V_Quad: + if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable + } else { + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true); + emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1); + emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true); + } + break; + default: + ; + } + } + + // OK, let's fill out the arrays to confirm that we have grabbed these registers. + for (int i = 0; i < n; i++) { + int mipsReg = 32 + vregs[i]; + mr[mipsReg].loc = ML_ARMREG; + mr[mipsReg].reg = QuadAsQ(quad); + mr[mipsReg].lane = i; + qr[quad].vregs[i] = vregs[i]; + } + qr[quad].isDirty = (flags & MAP_DIRTY) != 0; + qr[quad].spillLock = true; + + INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty); + if (sz == V_Single || sz == V_Pair) { + return D_0(QuadAsQ(quad)); + } else { + return QuadAsQ(quad); + } +} + diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.h b/Core/MIPS/ARM/ArmRegCacheFPU.h index 15e1c7de57..f62f09e2d8 100644 --- a/Core/MIPS/ARM/ArmRegCacheFPU.h +++ b/Core/MIPS/ARM/ArmRegCacheFPU.h @@ -33,28 +33,55 @@ enum { TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS, }; +enum { + MAP_MTX_TRANSPOSED = 16, + MAP_PREFER_LOW = 16, + MAP_PREFER_HIGH = 32, + + // Force is not yet correctly implemented, if the reg is already mapped it will not move + MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15) + MAP_FORCE_HIGH = 128, // Only map Q8-Q15 +}; + struct FPURegARM { int mipsReg; // if -1, no mipsreg attached. bool isDirty; // Should the register be written back? }; +struct FPURegQuad { + int mipsVec; + VectorSize sz; + u8 vregs[4]; + bool isDirty; + bool spillLock; + bool isTemp; +}; + struct FPURegMIPS { // Where is this MIPS register? RegMIPSLoc loc; // Data (only one of these is used, depending on loc. Could make a union). int reg; + int lane; + bool spillLock; // if true, this register cannot be spilled. bool tempLock; // If loc == ML_MEM, it's back in its location in the CPU context struct. }; +namespace MIPSComp { + struct ArmJitOptions; + struct JitState; +} + class ArmRegCacheFPU { public: - ArmRegCacheFPU(MIPSState *mips); + ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo); ~ArmRegCacheFPU() {} void Init(ARMXEmitter *emitter); + void Start(MIPSAnalyst::AnalysisResults &stats); // Protect the arm register containing a MIPS register from spilling, to ensure that @@ -80,9 +107,11 @@ public: void MapDirty(MIPSReg rd); void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true); void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true); + bool IsMapped(MIPSReg r); void FlushArmReg(ARMReg r); void FlushR(MIPSReg r); void DiscardR(MIPSReg r); + ARMReg R(int preg); // Returns a cached register // VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode! void MapRegV(int vreg, int flags = 0); @@ -90,18 +119,41 @@ public: void MapInInV(int rt, int rs); void MapDirtyInV(int rd, int rs, bool avoidLoad = true); void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true); - bool IsTempX(ARMReg r) const; - MIPSReg GetTempR(); + bool IsTempX(ARMReg r) const; MIPSReg GetTempV() { return GetTempR() - 32; } + // VFPU registers as single VFP registers. + ARMReg V(int vreg) { return R(vreg + 32); } int FlushGetSequential(int a, int maxArmReg); void FlushAll(); - ARMReg R(int preg); // Returns a cached register + // This one is allowed at any point. + void FlushV(MIPSReg r); + + // VFPU registers mapped to match NEON quads (and doubles, for pairs and singles) + // Here we return the ARM register directly instead of providing a "V" accessor + // and so on. Might switch to this model for the other regallocs later. + + // Quad mapping does NOT look into the ar array. Instead we use the qr array to keep + // track of what's in each quad. + + // Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types. + // Need to explicitly allow spilling to get spilling. + ARMReg QMapReg(int vreg, VectorSize sz, int flags); + + // TODO + // Maps a matrix as a set of columns (yes, even transposed ones, always columns + // as those are faster to load/flush). When possible it will map into consecutive + // quad registers, enabling blazing-fast full-matrix loads, transposed or not. + void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags); + + ARMReg QAllocTemp(VectorSize sz); - // VFPU registers as single VFP registers - ARMReg V(int vreg) { return R(vreg + 32); } + void QAllowSpill(int quad); + void QFlush(int quad); + void QLoad4x4(MIPSGPReg regPtr, int vquads[4]); + //void FlushQWithV(MIPSReg r); // NOTE: These require you to release spill locks manually! void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags); @@ -112,31 +164,43 @@ public: void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; } - // For better log output only. - void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; } - int GetMipsRegOffset(MIPSReg r); + +private: + bool Consecutive(int v1, int v2) const; + bool Consecutive(int v1, int v2, int v3) const; + bool Consecutive(int v1, int v2, int v3, int v4) const; + + MIPSReg GetTempR(); + const ARMReg *GetMIPSAllocationOrder(int &count); int GetMipsRegOffsetV(MIPSReg r) { return GetMipsRegOffset(r + 32); } + // This one WILL get a free quad as long as you haven't spill-locked them all. + int QGetFreeQuad(int start, int count, const char *reason); int GetNumARMFPURegs(); -private: void SetupInitialRegs(); MIPSState *mips_; ARMXEmitter *emit_; - u32 compilerPC_; + MIPSComp::JitState *js_; + MIPSComp::ArmJitOptions *jo_; int numARMFpuReg_; + int qTime_; enum { - MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON + // With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers + // are individually mappable though. + MAX_ARMFPUREG = 32, + MAX_ARMQUADS = 16, NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS, }; FPURegARM ar[MAX_ARMFPUREG]; FPURegMIPS mr[NUM_MIPSFPUREG]; + FPURegQuad qr[MAX_ARMQUADS]; FPURegMIPS *vr; bool pendingFlush;