mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-11 13:36:24 +02:00
Merge the RegCache changes from the old neon-vfpu branch
This commit is contained in:
1 parent
29dcc0a303
commit
d98bde8e50
5 files changed
+603
-56
No files matched your search
@@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() {
|
||||
useNEONVFPU = false;
|
||||
}
|
||||
|
||||
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips)
|
||||
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips)
|
||||
{
|
||||
logBlocks = 0;
|
||||
dontLogBlocks = 0;
|
||||
@@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b)
|
||||
while (js.compiling)
|
||||
{
|
||||
gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages
|
||||
fpr.SetCompilerPC(js.compilerPC);
|
||||
MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC);
|
||||
js.downcountAmount += MIPSGetInstructionCycleEstimate(inst);
|
||||
|
||||
|
||||
@@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) {
|
||||
return mr[mipsReg].loc == ML_ARMREG_AS_PTR;
|
||||
}
|
||||
|
||||
bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) {
|
||||
return mr[mipsReg].loc == ML_ARMREG;
|
||||
}
|
||||
|
||||
void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) {
|
||||
// If we can do it with a simple Operand2, let's do that.
|
||||
Operand2 op2;
|
||||
|
||||
@@ -103,6 +103,7 @@ public:
|
||||
ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0);
|
||||
ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty.
|
||||
|
||||
bool IsMapped(MIPSGPReg reg);
|
||||
bool IsMappedAsPointer(MIPSGPReg reg);
|
||||
|
||||
void MapInIn(MIPSGPReg rd, MIPSGPReg rs);
|
||||
|
||||
@@ -16,14 +16,17 @@
|
||||
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
|
||||
|
||||
#include <cstring>
|
||||
|
||||
#include "base/logging.h"
|
||||
#include "Common/CPUDetect.h"
|
||||
#include "Core/MIPS/MIPS.h"
|
||||
#include "Core/MIPS/ARM/ArmRegCacheFPU.h"
|
||||
#include "Core/MIPS/ARM/ArmJit.h"
|
||||
#include "Core/MIPS/MIPSTables.h"
|
||||
|
||||
using namespace ArmGen;
|
||||
|
||||
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) {
|
||||
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) {
|
||||
if (cpu_info.bNEON) {
|
||||
numARMFpuReg_ = 32;
|
||||
} else {
|
||||
@@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init
|
||||
}
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::Init(ARMXEmitter *emitter) {
|
||||
emit_ = emitter;
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) {
|
||||
if (!initialReady) {
|
||||
SetupInitialRegs();
|
||||
@@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() {
|
||||
mrInitial[i].spillLock = false;
|
||||
mrInitial[i].tempLock = false;
|
||||
}
|
||||
for (int i = 0; i < MAX_ARMQUADS; i++) {
|
||||
qr[i].isDirty = false;
|
||||
qr[i].mipsVec = -1;
|
||||
qr[i].sz = V_Invalid;
|
||||
qr[i].spillLock = false;
|
||||
qr[i].isTemp = false;
|
||||
memset(qr[i].vregs, 0xff, 4);
|
||||
}
|
||||
}
|
||||
|
||||
static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) {
|
||||
// We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things.
|
||||
static const ARMReg allocationOrder[] = {
|
||||
S2, S3,
|
||||
@@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
S12, S13, S14, S15
|
||||
};
|
||||
|
||||
// With NEON, we have many more.
|
||||
// In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU.
|
||||
// VFPU will use NEON to do SIMD and it will be awkward to mix with FPU.
|
||||
// VFP mapping
|
||||
// VFPU registers and regular FP registers are mapped interchangably on top of the standard
|
||||
// 16 FPU registers.
|
||||
|
||||
// NEON mapping
|
||||
// We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers.
|
||||
// VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON
|
||||
// (or D16-D31 as doubles, but not relevant).
|
||||
// Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads.
|
||||
|
||||
// We should attempt to map scalars to low Q registers and wider things to high registers,
|
||||
// as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be
|
||||
// able to use regular VFP instructions too.
|
||||
@@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
S4, S5, S6, S7, // Q1
|
||||
S8, S9, S10, S11, // Q2
|
||||
S12, S13, S14, S15, // Q3
|
||||
S16, S17, S18, S19, // Q4
|
||||
S20, S21, S22, S23, // Q5
|
||||
S24, S25, S26, S27, // Q6
|
||||
S28, S29, S30, S31, // Q7
|
||||
// Q8-Q15 free for NEON tricks
|
||||
// Q4-Q15 free for VFPU
|
||||
};
|
||||
|
||||
if (cpu_info.bNEON) {
|
||||
if (jo_->useNEONVFPU) {
|
||||
count = sizeof(allocationOrderNEON) / sizeof(const int);
|
||||
return allocationOrderNEON;
|
||||
} else {
|
||||
@@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
|
||||
}
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::IsMapped(MIPSReg r) {
|
||||
return mr[r].loc == ML_ARMREG;
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) {
|
||||
// INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags);
|
||||
if (jo_->useNEONVFPU && mipsReg >= 32) {
|
||||
ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC);
|
||||
return S0;
|
||||
}
|
||||
|
||||
pendingFlush = true;
|
||||
// Let's see if it's already mapped. If so we just need to update the dirty flag.
|
||||
// We don't need to check for ML_NOINIT because we assume that anyone who maps
|
||||
@@ -157,7 +176,7 @@ allocate:
|
||||
}
|
||||
|
||||
// Uh oh, we have all them spilllocked....
|
||||
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc);
|
||||
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC);
|
||||
return INVALID_REG;
|
||||
}
|
||||
|
||||
@@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::FlushArmReg(ARMReg r) {
|
||||
int reg = r - S0;
|
||||
if (ar[reg].mipsReg == -1) {
|
||||
// Nothing to do, reg not mapped.
|
||||
return;
|
||||
}
|
||||
if (ar[reg].mipsReg != -1) {
|
||||
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) {
|
||||
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
|
||||
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
|
||||
if (r >= S0 && r <= S31) {
|
||||
int reg = r - S0;
|
||||
if (ar[reg].mipsReg == -1) {
|
||||
// Nothing to do, reg not mapped.
|
||||
return;
|
||||
}
|
||||
// IMMs won't be in an ARM reg.
|
||||
mr[ar[reg].mipsReg].loc = ML_MEM;
|
||||
mr[ar[reg].mipsReg].reg = INVALID_REG;
|
||||
} else {
|
||||
ERROR_LOG(JIT, "Dirty but no mipsreg?");
|
||||
if (ar[reg].mipsReg != -1) {
|
||||
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG)
|
||||
{
|
||||
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
|
||||
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
|
||||
}
|
||||
// IMMs won't be in an ARM reg.
|
||||
mr[ar[reg].mipsReg].loc = ML_MEM;
|
||||
mr[ar[reg].mipsReg].reg = INVALID_REG;
|
||||
} else {
|
||||
ERROR_LOG(JIT, "Dirty but no mipsreg?");
|
||||
}
|
||||
ar[reg].isDirty = false;
|
||||
ar[reg].mipsReg = -1;
|
||||
} else if (r >= D0 && r <= D31) {
|
||||
// TODO: Convert to S regs and flush them individually.
|
||||
} else if (r >= Q0 && r <= Q15) {
|
||||
int quad = r - Q0;
|
||||
QFlush(r);
|
||||
}
|
||||
ar[reg].isDirty = false;
|
||||
ar[reg].mipsReg = -1;
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::FlushV(MIPSReg r) {
|
||||
FlushR(r + 32);
|
||||
}
|
||||
|
||||
/*
|
||||
void ArmRegCacheFPU::FlushQWithV(MIPSReg r) {
|
||||
// Look for it in all the quads. If it's in any, flush that quad clean.
|
||||
int flushCount = 0;
|
||||
for (int i = 0; i < MAX_ARMQUADS; i++) {
|
||||
if (qr[i].sz == V_Invalid)
|
||||
continue;
|
||||
|
||||
int n = qr[i].sz;
|
||||
bool flushThis = false;
|
||||
for (int j = 0; j < n; j++) {
|
||||
if (qr[i].vregs[j] == r) {
|
||||
flushThis = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (flushThis) {
|
||||
QFlush(i);
|
||||
flushCount++;
|
||||
}
|
||||
}
|
||||
|
||||
if (flushCount > 1) {
|
||||
WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r);
|
||||
}
|
||||
}
|
||||
*/
|
||||
|
||||
void ArmRegCacheFPU::FlushR(MIPSReg r) {
|
||||
switch (mr[r].loc) {
|
||||
case ML_IMM:
|
||||
@@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) {
|
||||
if (mr[r].reg == (int)INVALID_REG) {
|
||||
ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg");
|
||||
}
|
||||
if (ar[mr[r].reg].isDirty) {
|
||||
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
|
||||
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
|
||||
ar[mr[r].reg].isDirty = false;
|
||||
|
||||
if (mr[r].reg >= Q0 && mr[r].reg <= Q15) {
|
||||
// This should happen rarely, but occasionally we need to flush a single stray
|
||||
// mipsreg that's been part of a quad.
|
||||
int quad = mr[r].reg - Q0;
|
||||
if (qr[quad].isDirty) {
|
||||
WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1);
|
||||
emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true);
|
||||
}
|
||||
} else {
|
||||
if (ar[mr[r].reg].isDirty) {
|
||||
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
|
||||
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
|
||||
ar[mr[r].reg].isDirty = false;
|
||||
}
|
||||
ar[mr[r].reg].mipsReg = -1;
|
||||
}
|
||||
ar[mr[r].reg].mipsReg = -1;
|
||||
break;
|
||||
|
||||
case ML_MEM:
|
||||
@@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() {
|
||||
return 16;
|
||||
}
|
||||
|
||||
// Scalar only. Need a similar one for sequential Q vectors.
|
||||
int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) {
|
||||
int c = 1;
|
||||
int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg);
|
||||
@@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() {
|
||||
DiscardR(i);
|
||||
}
|
||||
|
||||
// Flush quads!
|
||||
// These could also use sequential detection.
|
||||
for (int i = 4; i < MAX_ARMQUADS; i++) {
|
||||
QFlush(i);
|
||||
}
|
||||
|
||||
// Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are
|
||||
// sequential. This is necessary because we store VFPU registers in a staggered order to get
|
||||
// columns sequential (most VFPU math in nearly all games is in columns, not rows).
|
||||
@@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const {
|
||||
}
|
||||
|
||||
int ArmRegCacheFPU::GetTempR() {
|
||||
if (jo_->useNEONVFPU) {
|
||||
ERROR_LOG(JIT, "VFP temps not allowed in NEON mode");
|
||||
return 0;
|
||||
}
|
||||
pendingFlush = true;
|
||||
for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) {
|
||||
if (mr[r].loc == ML_MEM && !mr[r].tempLock) {
|
||||
@@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) {
|
||||
|
||||
// This is actually pretty slow with all the 160 regs...
|
||||
void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() {
|
||||
for (int i = 0; i < NUM_MIPSFPUREG; i++)
|
||||
for (int i = 0; i < NUM_MIPSFPUREG; i++) {
|
||||
mr[i].spillLock = false;
|
||||
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i)
|
||||
}
|
||||
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) {
|
||||
DiscardR(i);
|
||||
}
|
||||
for (int i = 0; i < MAX_ARMQUADS; i++) {
|
||||
qr[i].spillLock = false;
|
||||
if (qr[i].isTemp) {
|
||||
qr[i].isTemp = false;
|
||||
qr[i].sz = V_Invalid;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::R(int mipsReg) {
|
||||
@@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) {
|
||||
return (ARMReg)(mr[mipsReg].reg + S0);
|
||||
} else {
|
||||
if (mipsReg < 32) {
|
||||
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_));
|
||||
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
|
||||
} else if (mipsReg < 32 + 128) {
|
||||
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
|
||||
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
|
||||
} else {
|
||||
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
|
||||
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
|
||||
}
|
||||
return INVALID_REG; // BAAAD
|
||||
}
|
||||
}
|
||||
|
||||
inline ARMReg QuadAsD(int quad) {
|
||||
return (ARMReg)(D0 + quad * 2);
|
||||
}
|
||||
|
||||
inline ARMReg QuadAsQ(int quad) {
|
||||
return (ARMReg)(Q0 + quad);
|
||||
}
|
||||
|
||||
bool MappableQ(int quad) {
|
||||
return quad >= 4;
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) {
|
||||
ERROR_LOG(JIT, "QLoad4x4 not implemented");
|
||||
// TODO
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::QFlush(int quad) {
|
||||
if (!MappableQ(quad)) {
|
||||
ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad);
|
||||
return;
|
||||
}
|
||||
|
||||
if (qr[quad].isDirty && !qr[quad].isTemp) {
|
||||
INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz));
|
||||
|
||||
ARMReg q = QuadAsQ(quad);
|
||||
// Unlike reads, when writing to the register file we need to be careful to write the correct
|
||||
// number of floats.
|
||||
|
||||
switch (qr[quad].sz) {
|
||||
case V_Single:
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
// WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
break;
|
||||
case V_Pair:
|
||||
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) {
|
||||
// Can combine, it's a column!
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
// WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 1, true);
|
||||
}
|
||||
break;
|
||||
case V_Triple:
|
||||
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
|
||||
emit_->VST1_lane(F_32, q, R0, 2, true);
|
||||
} else {
|
||||
// WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 2, true);
|
||||
}
|
||||
break;
|
||||
case V_Quad:
|
||||
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
// WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 2, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1);
|
||||
emit_->VST1_lane(F_32, q, R0, 3, true);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz);
|
||||
break;
|
||||
}
|
||||
|
||||
qr[quad].isDirty = false;
|
||||
|
||||
int n = GetNumVectorElements(qr[quad].sz);
|
||||
for (int i = 0; i < n; i++) {
|
||||
int vr = qr[quad].vregs[i];
|
||||
if (vr < 0 || vr > 128) {
|
||||
ERROR_LOG(JIT, "Bad vr %i", vr);
|
||||
}
|
||||
FPURegMIPS &m = mr[32 + vr];
|
||||
m.loc = ML_MEM;
|
||||
m.lane = -1;
|
||||
m.reg = -1;
|
||||
}
|
||||
|
||||
} else {
|
||||
if (qr[quad].isTemp) {
|
||||
WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp);
|
||||
}
|
||||
}
|
||||
|
||||
qr[quad].isTemp = false;
|
||||
qr[quad].mipsVec = -1;
|
||||
qr[quad].sz = V_Invalid;
|
||||
memset(qr[quad].vregs, 0xFF, 4);
|
||||
}
|
||||
|
||||
int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) {
|
||||
// Search for a free quad. A quad is free if the first register in it is free.
|
||||
int quad = -1;
|
||||
|
||||
for (int i = 0; i < count; i++) {
|
||||
int q = (i + start) & 15;
|
||||
|
||||
if (!MappableQ(q))
|
||||
continue;
|
||||
|
||||
// Don't steal temp quads!
|
||||
if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) {
|
||||
// INFO_LOG(JIT, "Free quad: %i", q);
|
||||
// Oh yeah! Free quad!
|
||||
return q;
|
||||
}
|
||||
}
|
||||
|
||||
// Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some
|
||||
// sort of age.
|
||||
int bestQuad = -1;
|
||||
int bestScore = -1;
|
||||
for (int i = 0; i < count; i++) {
|
||||
int q = (i + start) & 15;
|
||||
|
||||
if (!MappableQ(q))
|
||||
continue;
|
||||
if (qr[q].spillLock)
|
||||
continue;
|
||||
if (qr[q].isTemp)
|
||||
continue;
|
||||
|
||||
int score = 0;
|
||||
if (!qr[q].isDirty) {
|
||||
score += 5;
|
||||
}
|
||||
|
||||
if (score > bestScore) {
|
||||
bestQuad = q;
|
||||
bestScore = score;
|
||||
}
|
||||
}
|
||||
|
||||
if (bestQuad == -1) {
|
||||
ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!");
|
||||
return -1;
|
||||
} else {
|
||||
INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason");
|
||||
QFlush(bestQuad);
|
||||
return bestQuad;
|
||||
}
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) {
|
||||
int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps
|
||||
if (q < 0) {
|
||||
ERROR_LOG(JIT, "Failed to allocate temp quad");
|
||||
q = 0;
|
||||
}
|
||||
qr[q].spillLock = true;
|
||||
qr[q].isTemp = true;
|
||||
qr[q].sz = sz;
|
||||
qr[q].isDirty = false; // doesn't matter
|
||||
|
||||
INFO_LOG(JIT, "Allocated temp quad %i", q);
|
||||
|
||||
if (sz == V_Single || sz == V_Pair) {
|
||||
return D_0(ARMReg(Q0 + q));
|
||||
} else {
|
||||
return ARMReg(Q0 + q);
|
||||
}
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::Consecutive(int v1, int v2) const {
|
||||
return (voffset[v1] + 1) == voffset[v2];
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const {
|
||||
return Consecutive(v1, v2) && Consecutive(v2, v3);
|
||||
}
|
||||
|
||||
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const {
|
||||
return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4);
|
||||
}
|
||||
|
||||
void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) {
|
||||
u8 vregs[4];
|
||||
if (flags & MAP_MTX_TRANSPOSED) {
|
||||
GetMatrixRows(matrix, mz, vregs);
|
||||
} else {
|
||||
GetMatrixColumns(matrix, mz, vregs);
|
||||
}
|
||||
|
||||
// TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load.
|
||||
int n = GetMatrixSide(mz);
|
||||
VectorSize vsz = GetVectorSize(mz);
|
||||
for (int i = 0; i < n; i++) {
|
||||
regs[i] = QMapReg(vregs[i], vsz, flags);
|
||||
}
|
||||
}
|
||||
|
||||
ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) {
|
||||
qTime_++;
|
||||
|
||||
int n = GetNumVectorElements(sz);
|
||||
u8 vregs[4];
|
||||
GetVectorRegs(vregs, sz, vreg);
|
||||
|
||||
// Range of registers to consider
|
||||
int start = 0;
|
||||
int count = 16;
|
||||
|
||||
if (flags & MAP_PREFER_HIGH) {
|
||||
start = 8;
|
||||
} else if (flags & MAP_PREFER_LOW) {
|
||||
start = 4;
|
||||
} else if (flags & MAP_FORCE_LOW) {
|
||||
start = 4;
|
||||
count = 4;
|
||||
} else if (flags & MAP_FORCE_HIGH) {
|
||||
start = 8;
|
||||
count = 8;
|
||||
}
|
||||
|
||||
// Let's check if they are all mapped in a quad somewhere.
|
||||
// At the same time, check for the quad already being mapped.
|
||||
// Later we can check for possible transposes as well.
|
||||
|
||||
// First just loop over all registers. If it's here and not in range, or overlapped, kick.
|
||||
std::vector<int> quadsToFlush;
|
||||
for (int i = 0; i < 16; i++) {
|
||||
int q = (i + start) & 15;
|
||||
if (!MappableQ(q))
|
||||
continue;
|
||||
|
||||
// Skip unmapped quads.
|
||||
if (qr[q].sz == V_Invalid)
|
||||
continue;
|
||||
|
||||
// Check if completely there already. If so, set spill-lock, transfer dirty flag and exit.
|
||||
if (vreg == qr[q].mipsVec && sz == qr[q].sz) {
|
||||
if (i < count) {
|
||||
INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz);
|
||||
qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY);
|
||||
qr[q].spillLock = true;
|
||||
|
||||
// Sanity check vregs
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (vregs[i] != qr[q].vregs[i]) {
|
||||
ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]);
|
||||
}
|
||||
}
|
||||
|
||||
return (ARMReg)(Q0 + q);
|
||||
} else {
|
||||
INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count);
|
||||
quadsToFlush.push_back(q);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for any overlap. Overlap == flush.
|
||||
int origN = GetNumVectorElements(qr[q].sz);
|
||||
for (int a = 0; a < n; a++) {
|
||||
for (int b = 0; b < origN; b++) {
|
||||
if (vregs[a] == qr[q].vregs[b]) {
|
||||
quadsToFlush.push_back(q);
|
||||
goto doubleBreak;
|
||||
}
|
||||
}
|
||||
}
|
||||
doubleBreak:
|
||||
;
|
||||
}
|
||||
|
||||
// We didn't find the extra register, but we got a list of regs to flush. Flush 'em.
|
||||
// Here we can check for opportunities to do a "transpose-flush" of row vectors, etc.
|
||||
if (!quadsToFlush.empty()) {
|
||||
ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size());
|
||||
}
|
||||
for (size_t i = 0; i < quadsToFlush.size(); i++) {
|
||||
QFlush(quadsToFlush[i]);
|
||||
}
|
||||
|
||||
// Find where we want to map it, obeying the constraints we gave.
|
||||
int quad = QGetFreeQuad(start, count, "mapping");
|
||||
|
||||
// If parts of our register are elsewhere, and we are dirty, we need to flush them
|
||||
// before we reload in a new location.
|
||||
// This may be problematic if inputs overlap irregularly with output, say:
|
||||
// vdot S700, R000, C000
|
||||
// It might still work by accident...
|
||||
if (flags & MAP_DIRTY) {
|
||||
for (int i = 0; i < n; i++) {
|
||||
FlushV(vregs[i]);
|
||||
}
|
||||
}
|
||||
|
||||
qr[quad].sz = sz;
|
||||
qr[quad].mipsVec = vreg;
|
||||
|
||||
if (!(flags & MAP_NOINIT)) {
|
||||
// Okay, now we will try to load the whole thing in one go. This is possible
|
||||
// if it's a row and easy if it's a single.
|
||||
// Rows are rare, columns are common - but thanks to our register reordering,
|
||||
// columns are actually in-order in memory.
|
||||
switch (sz) {
|
||||
case V_Single:
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
break;
|
||||
case V_Pair:
|
||||
if (Consecutive(vregs[0], vregs[1])) {
|
||||
// Can combine, it's a column!
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
|
||||
}
|
||||
break;
|
||||
case V_Triple:
|
||||
if (Consecutive(vregs[0], vregs[1], vregs[2])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
|
||||
} else {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
|
||||
}
|
||||
break;
|
||||
case V_Quad:
|
||||
if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
|
||||
} else {
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
|
||||
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1);
|
||||
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
;
|
||||
}
|
||||
}
|
||||
|
||||
// OK, let's fill out the arrays to confirm that we have grabbed these registers.
|
||||
for (int i = 0; i < n; i++) {
|
||||
int mipsReg = 32 + vregs[i];
|
||||
mr[mipsReg].loc = ML_ARMREG;
|
||||
mr[mipsReg].reg = QuadAsQ(quad);
|
||||
mr[mipsReg].lane = i;
|
||||
qr[quad].vregs[i] = vregs[i];
|
||||
}
|
||||
qr[quad].isDirty = (flags & MAP_DIRTY) != 0;
|
||||
qr[quad].spillLock = true;
|
||||
|
||||
INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty);
|
||||
if (sz == V_Single || sz == V_Pair) {
|
||||
return D_0(QuadAsQ(quad));
|
||||
} else {
|
||||
return QuadAsQ(quad);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -33,28 +33,55 @@ enum {
|
||||
TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS,
|
||||
};
|
||||
|
||||
enum {
|
||||
MAP_MTX_TRANSPOSED = 16,
|
||||
MAP_PREFER_LOW = 16,
|
||||
MAP_PREFER_HIGH = 32,
|
||||
|
||||
// Force is not yet correctly implemented, if the reg is already mapped it will not move
|
||||
MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15)
|
||||
MAP_FORCE_HIGH = 128, // Only map Q8-Q15
|
||||
};
|
||||
|
||||
struct FPURegARM {
|
||||
int mipsReg; // if -1, no mipsreg attached.
|
||||
bool isDirty; // Should the register be written back?
|
||||
};
|
||||
|
||||
struct FPURegQuad {
|
||||
int mipsVec;
|
||||
VectorSize sz;
|
||||
u8 vregs[4];
|
||||
bool isDirty;
|
||||
bool spillLock;
|
||||
bool isTemp;
|
||||
};
|
||||
|
||||
struct FPURegMIPS {
|
||||
// Where is this MIPS register?
|
||||
RegMIPSLoc loc;
|
||||
// Data (only one of these is used, depending on loc. Could make a union).
|
||||
int reg;
|
||||
int lane;
|
||||
|
||||
bool spillLock; // if true, this register cannot be spilled.
|
||||
bool tempLock;
|
||||
// If loc == ML_MEM, it's back in its location in the CPU context struct.
|
||||
};
|
||||
|
||||
namespace MIPSComp {
|
||||
struct ArmJitOptions;
|
||||
struct JitState;
|
||||
}
|
||||
|
||||
class ArmRegCacheFPU
|
||||
{
|
||||
public:
|
||||
ArmRegCacheFPU(MIPSState *mips);
|
||||
ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo);
|
||||
~ArmRegCacheFPU() {}
|
||||
|
||||
void Init(ARMXEmitter *emitter);
|
||||
|
||||
void Start(MIPSAnalyst::AnalysisResults &stats);
|
||||
|
||||
// Protect the arm register containing a MIPS register from spilling, to ensure that
|
||||
@@ -80,9 +107,11 @@ public:
|
||||
void MapDirty(MIPSReg rd);
|
||||
void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true);
|
||||
void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true);
|
||||
bool IsMapped(MIPSReg r);
|
||||
void FlushArmReg(ARMReg r);
|
||||
void FlushR(MIPSReg r);
|
||||
void DiscardR(MIPSReg r);
|
||||
ARMReg R(int preg); // Returns a cached register
|
||||
|
||||
// VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode!
|
||||
void MapRegV(int vreg, int flags = 0);
|
||||
@@ -90,18 +119,41 @@ public:
|
||||
void MapInInV(int rt, int rs);
|
||||
void MapDirtyInV(int rd, int rs, bool avoidLoad = true);
|
||||
void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true);
|
||||
bool IsTempX(ARMReg r) const;
|
||||
|
||||
MIPSReg GetTempR();
|
||||
bool IsTempX(ARMReg r) const;
|
||||
MIPSReg GetTempV() { return GetTempR() - 32; }
|
||||
// VFPU registers as single VFP registers.
|
||||
ARMReg V(int vreg) { return R(vreg + 32); }
|
||||
|
||||
int FlushGetSequential(int a, int maxArmReg);
|
||||
void FlushAll();
|
||||
|
||||
ARMReg R(int preg); // Returns a cached register
|
||||
// This one is allowed at any point.
|
||||
void FlushV(MIPSReg r);
|
||||
|
||||
// VFPU registers mapped to match NEON quads (and doubles, for pairs and singles)
|
||||
// Here we return the ARM register directly instead of providing a "V" accessor
|
||||
// and so on. Might switch to this model for the other regallocs later.
|
||||
|
||||
// Quad mapping does NOT look into the ar array. Instead we use the qr array to keep
|
||||
// track of what's in each quad.
|
||||
|
||||
// Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types.
|
||||
// Need to explicitly allow spilling to get spilling.
|
||||
ARMReg QMapReg(int vreg, VectorSize sz, int flags);
|
||||
|
||||
// TODO
|
||||
// Maps a matrix as a set of columns (yes, even transposed ones, always columns
|
||||
// as those are faster to load/flush). When possible it will map into consecutive
|
||||
// quad registers, enabling blazing-fast full-matrix loads, transposed or not.
|
||||
void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags);
|
||||
|
||||
ARMReg QAllocTemp(VectorSize sz);
|
||||
|
||||
// VFPU registers as single VFP registers
|
||||
ARMReg V(int vreg) { return R(vreg + 32); }
|
||||
void QAllowSpill(int quad);
|
||||
void QFlush(int quad);
|
||||
void QLoad4x4(MIPSGPReg regPtr, int vquads[4]);
|
||||
//void FlushQWithV(MIPSReg r);
|
||||
|
||||
// NOTE: These require you to release spill locks manually!
|
||||
void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags);
|
||||
@@ -112,31 +164,43 @@ public:
|
||||
|
||||
void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; }
|
||||
|
||||
// For better log output only.
|
||||
void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; }
|
||||
|
||||
int GetMipsRegOffset(MIPSReg r);
|
||||
|
||||
private:
|
||||
bool Consecutive(int v1, int v2) const;
|
||||
bool Consecutive(int v1, int v2, int v3) const;
|
||||
bool Consecutive(int v1, int v2, int v3, int v4) const;
|
||||
|
||||
MIPSReg GetTempR();
|
||||
const ARMReg *GetMIPSAllocationOrder(int &count);
|
||||
int GetMipsRegOffsetV(MIPSReg r) {
|
||||
return GetMipsRegOffset(r + 32);
|
||||
}
|
||||
// This one WILL get a free quad as long as you haven't spill-locked them all.
|
||||
int QGetFreeQuad(int start, int count, const char *reason);
|
||||
int GetNumARMFPURegs();
|
||||
|
||||
private:
|
||||
void SetupInitialRegs();
|
||||
|
||||
MIPSState *mips_;
|
||||
ARMXEmitter *emit_;
|
||||
u32 compilerPC_;
|
||||
MIPSComp::JitState *js_;
|
||||
MIPSComp::ArmJitOptions *jo_;
|
||||
|
||||
int numARMFpuReg_;
|
||||
int qTime_;
|
||||
|
||||
enum {
|
||||
MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON
|
||||
// With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers
|
||||
// are individually mappable though.
|
||||
MAX_ARMFPUREG = 32,
|
||||
MAX_ARMQUADS = 16,
|
||||
NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS,
|
||||
};
|
||||
|
||||
FPURegARM ar[MAX_ARMFPUREG];
|
||||
FPURegMIPS mr[NUM_MIPSFPUREG];
|
||||
FPURegQuad qr[MAX_ARMQUADS];
|
||||
FPURegMIPS *vr;
|
||||
|
||||
bool pendingFlush;
|
||||
|
||||
Reference in new issue
Block a user