Merge the RegCache changes from the old neon-vfpu branch

This commit is contained in:
Henrik Rydgard committed 2014-12-06 12:26:58 +01:00
1 parent 29dcc0a303
commit d98bde8e50
5 files changed
+603 -56

No files matched your search

+1 -2
View File
@@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() {
useNEONVFPU = false;
}
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips)
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips)
{
logBlocks = 0;
dontLogBlocks = 0;
@@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b)
while (js.compiling)
{
gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages
fpr.SetCompilerPC(js.compilerPC);
MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC);
js.downcountAmount += MIPSGetInstructionCycleEstimate(inst);
+4
View File
@@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) {
return mr[mipsReg].loc == ML_ARMREG_AS_PTR;
}
bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) {
return mr[mipsReg].loc == ML_ARMREG;
}
void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) {
// If we can do it with a simple Operand2, let's do that.
Operand2 op2;
+1
View File
@@ -103,6 +103,7 @@ public:
ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0);
ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty.
bool IsMapped(MIPSGPReg reg);
bool IsMappedAsPointer(MIPSGPReg reg);
void MapInIn(MIPSGPReg rd, MIPSGPReg rs);
+521 -42
View File
@@ -16,14 +16,17 @@
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
#include <cstring>
#include "base/logging.h"
#include "Common/CPUDetect.h"
#include "Core/MIPS/MIPS.h"
#include "Core/MIPS/ARM/ArmRegCacheFPU.h"
#include "Core/MIPS/ARM/ArmJit.h"
#include "Core/MIPS/MIPSTables.h"
using namespace ArmGen;
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) {
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) {
if (cpu_info.bNEON) {
numARMFpuReg_ = 32;
} else {
@@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init
}
}
void ArmRegCacheFPU::Init(ARMXEmitter *emitter) {
emit_ = emitter;
}
void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) {
if (!initialReady) {
SetupInitialRegs();
@@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() {
mrInitial[i].spillLock = false;
mrInitial[i].tempLock = false;
}
for (int i = 0; i < MAX_ARMQUADS; i++) {
qr[i].isDirty = false;
qr[i].mipsVec = -1;
qr[i].sz = V_Invalid;
qr[i].spillLock = false;
qr[i].isTemp = false;
memset(qr[i].vregs, 0xff, 4);
}
}
static const ARMReg *GetMIPSAllocationOrder(int &count) {
const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) {
// We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things.
static const ARMReg allocationOrder[] = {
S2, S3,
@@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
S12, S13, S14, S15
};
// With NEON, we have many more.
// In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU.
// VFPU will use NEON to do SIMD and it will be awkward to mix with FPU.
// VFP mapping
// VFPU registers and regular FP registers are mapped interchangably on top of the standard
// 16 FPU registers.
// NEON mapping
// We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers.
// VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON
// (or D16-D31 as doubles, but not relevant).
// Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads.
// We should attempt to map scalars to low Q registers and wider things to high registers,
// as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be
// able to use regular VFP instructions too.
@@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
S4, S5, S6, S7, // Q1
S8, S9, S10, S11, // Q2
S12, S13, S14, S15, // Q3
S16, S17, S18, S19, // Q4
S20, S21, S22, S23, // Q5
S24, S25, S26, S27, // Q6
S28, S29, S30, S31, // Q7
// Q8-Q15 free for NEON tricks
// Q4-Q15 free for VFPU
};
if (cpu_info.bNEON) {
if (jo_->useNEONVFPU) {
count = sizeof(allocationOrderNEON) / sizeof(const int);
return allocationOrderNEON;
} else {
@@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
}
}
bool ArmRegCacheFPU::IsMapped(MIPSReg r) {
return mr[r].loc == ML_ARMREG;
}
ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) {
// INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags);
if (jo_->useNEONVFPU && mipsReg >= 32) {
ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC);
return S0;
}
pendingFlush = true;
// Let's see if it's already mapped. If so we just need to update the dirty flag.
// We don't need to check for ML_NOINIT because we assume that anyone who maps
@@ -157,7 +176,7 @@ allocate:
}
// Uh oh, we have all them spilllocked....
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc);
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC);
return INVALID_REG;
}
@@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
}
void ArmRegCacheFPU::FlushArmReg(ARMReg r) {
int reg = r - S0;
if (ar[reg].mipsReg == -1) {
// Nothing to do, reg not mapped.
return;
}
if (ar[reg].mipsReg != -1) {
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) {
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
if (r >= S0 && r <= S31) {
int reg = r - S0;
if (ar[reg].mipsReg == -1) {
// Nothing to do, reg not mapped.
return;
}
// IMMs won't be in an ARM reg.
mr[ar[reg].mipsReg].loc = ML_MEM;
mr[ar[reg].mipsReg].reg = INVALID_REG;
} else {
ERROR_LOG(JIT, "Dirty but no mipsreg?");
if (ar[reg].mipsReg != -1) {
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG)
{
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
}
// IMMs won't be in an ARM reg.
mr[ar[reg].mipsReg].loc = ML_MEM;
mr[ar[reg].mipsReg].reg = INVALID_REG;
} else {
ERROR_LOG(JIT, "Dirty but no mipsreg?");
}
ar[reg].isDirty = false;
ar[reg].mipsReg = -1;
} else if (r >= D0 && r <= D31) {
// TODO: Convert to S regs and flush them individually.
} else if (r >= Q0 && r <= Q15) {
int quad = r - Q0;
QFlush(r);
}
ar[reg].isDirty = false;
ar[reg].mipsReg = -1;
}
void ArmRegCacheFPU::FlushV(MIPSReg r) {
FlushR(r + 32);
}
/*
void ArmRegCacheFPU::FlushQWithV(MIPSReg r) {
// Look for it in all the quads. If it's in any, flush that quad clean.
int flushCount = 0;
for (int i = 0; i < MAX_ARMQUADS; i++) {
if (qr[i].sz == V_Invalid)
continue;
int n = qr[i].sz;
bool flushThis = false;
for (int j = 0; j < n; j++) {
if (qr[i].vregs[j] == r) {
flushThis = true;
}
}
if (flushThis) {
QFlush(i);
flushCount++;
}
}
if (flushCount > 1) {
WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r);
}
}
*/
void ArmRegCacheFPU::FlushR(MIPSReg r) {
switch (mr[r].loc) {
case ML_IMM:
@@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) {
if (mr[r].reg == (int)INVALID_REG) {
ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg");
}
if (ar[mr[r].reg].isDirty) {
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
ar[mr[r].reg].isDirty = false;
if (mr[r].reg >= Q0 && mr[r].reg <= Q15) {
// This should happen rarely, but occasionally we need to flush a single stray
// mipsreg that's been part of a quad.
int quad = mr[r].reg - Q0;
if (qr[quad].isDirty) {
WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1);
emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true);
}
} else {
if (ar[mr[r].reg].isDirty) {
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
ar[mr[r].reg].isDirty = false;
}
ar[mr[r].reg].mipsReg = -1;
}
ar[mr[r].reg].mipsReg = -1;
break;
case ML_MEM:
@@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() {
return 16;
}
// Scalar only. Need a similar one for sequential Q vectors.
int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) {
int c = 1;
int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg);
@@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() {
DiscardR(i);
}
// Flush quads!
// These could also use sequential detection.
for (int i = 4; i < MAX_ARMQUADS; i++) {
QFlush(i);
}
// Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are
// sequential. This is necessary because we store VFPU registers in a staggered order to get
// columns sequential (most VFPU math in nearly all games is in columns, not rows).
@@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const {
}
int ArmRegCacheFPU::GetTempR() {
if (jo_->useNEONVFPU) {
ERROR_LOG(JIT, "VFP temps not allowed in NEON mode");
return 0;
}
pendingFlush = true;
for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) {
if (mr[r].loc == ML_MEM && !mr[r].tempLock) {
@@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) {
// This is actually pretty slow with all the 160 regs...
void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() {
for (int i = 0; i < NUM_MIPSFPUREG; i++)
for (int i = 0; i < NUM_MIPSFPUREG; i++) {
mr[i].spillLock = false;
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i)
}
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) {
DiscardR(i);
}
for (int i = 0; i < MAX_ARMQUADS; i++) {
qr[i].spillLock = false;
if (qr[i].isTemp) {
qr[i].isTemp = false;
qr[i].sz = V_Invalid;
}
}
}
ARMReg ArmRegCacheFPU::R(int mipsReg) {
@@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) {
return (ARMReg)(mr[mipsReg].reg + S0);
} else {
if (mipsReg < 32) {
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_));
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
} else if (mipsReg < 32 + 128) {
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
} else {
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
}
return INVALID_REG; // BAAAD
}
}
inline ARMReg QuadAsD(int quad) {
return (ARMReg)(D0 + quad * 2);
}
inline ARMReg QuadAsQ(int quad) {
return (ARMReg)(Q0 + quad);
}
bool MappableQ(int quad) {
return quad >= 4;
}
void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) {
ERROR_LOG(JIT, "QLoad4x4 not implemented");
// TODO
}
void ArmRegCacheFPU::QFlush(int quad) {
if (!MappableQ(quad)) {
ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad);
return;
}
if (qr[quad].isDirty && !qr[quad].isTemp) {
INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz));
ARMReg q = QuadAsQ(quad);
// Unlike reads, when writing to the register file we need to be careful to write the correct
// number of floats.
switch (qr[quad].sz) {
case V_Single:
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
// WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC);
break;
case V_Pair:
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) {
// Can combine, it's a column!
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
// WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
emit_->VST1_lane(F_32, q, R0, 1, true);
}
break;
case V_Triple:
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
emit_->VST1_lane(F_32, q, R0, 2, true);
} else {
// WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
emit_->VST1_lane(F_32, q, R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
emit_->VST1_lane(F_32, q, R0, 2, true);
}
break;
case V_Quad:
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
// WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
emit_->VST1_lane(F_32, q, R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
emit_->VST1_lane(F_32, q, R0, 2, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1);
emit_->VST1_lane(F_32, q, R0, 3, true);
}
break;
default:
ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz);
break;
}
qr[quad].isDirty = false;
int n = GetNumVectorElements(qr[quad].sz);
for (int i = 0; i < n; i++) {
int vr = qr[quad].vregs[i];
if (vr < 0 || vr > 128) {
ERROR_LOG(JIT, "Bad vr %i", vr);
}
FPURegMIPS &m = mr[32 + vr];
m.loc = ML_MEM;
m.lane = -1;
m.reg = -1;
}
} else {
if (qr[quad].isTemp) {
WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp);
}
}
qr[quad].isTemp = false;
qr[quad].mipsVec = -1;
qr[quad].sz = V_Invalid;
memset(qr[quad].vregs, 0xFF, 4);
}
int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) {
// Search for a free quad. A quad is free if the first register in it is free.
int quad = -1;
for (int i = 0; i < count; i++) {
int q = (i + start) & 15;
if (!MappableQ(q))
continue;
// Don't steal temp quads!
if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) {
// INFO_LOG(JIT, "Free quad: %i", q);
// Oh yeah! Free quad!
return q;
}
}
// Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some
// sort of age.
int bestQuad = -1;
int bestScore = -1;
for (int i = 0; i < count; i++) {
int q = (i + start) & 15;
if (!MappableQ(q))
continue;
if (qr[q].spillLock)
continue;
if (qr[q].isTemp)
continue;
int score = 0;
if (!qr[q].isDirty) {
score += 5;
}
if (score > bestScore) {
bestQuad = q;
bestScore = score;
}
}
if (bestQuad == -1) {
ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!");
return -1;
} else {
INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason");
QFlush(bestQuad);
return bestQuad;
}
}
ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) {
int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps
if (q < 0) {
ERROR_LOG(JIT, "Failed to allocate temp quad");
q = 0;
}
qr[q].spillLock = true;
qr[q].isTemp = true;
qr[q].sz = sz;
qr[q].isDirty = false; // doesn't matter
INFO_LOG(JIT, "Allocated temp quad %i", q);
if (sz == V_Single || sz == V_Pair) {
return D_0(ARMReg(Q0 + q));
} else {
return ARMReg(Q0 + q);
}
}
bool ArmRegCacheFPU::Consecutive(int v1, int v2) const {
return (voffset[v1] + 1) == voffset[v2];
}
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const {
return Consecutive(v1, v2) && Consecutive(v2, v3);
}
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const {
return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4);
}
void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) {
u8 vregs[4];
if (flags & MAP_MTX_TRANSPOSED) {
GetMatrixRows(matrix, mz, vregs);
} else {
GetMatrixColumns(matrix, mz, vregs);
}
// TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load.
int n = GetMatrixSide(mz);
VectorSize vsz = GetVectorSize(mz);
for (int i = 0; i < n; i++) {
regs[i] = QMapReg(vregs[i], vsz, flags);
}
}
ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) {
qTime_++;
int n = GetNumVectorElements(sz);
u8 vregs[4];
GetVectorRegs(vregs, sz, vreg);
// Range of registers to consider
int start = 0;
int count = 16;
if (flags & MAP_PREFER_HIGH) {
start = 8;
} else if (flags & MAP_PREFER_LOW) {
start = 4;
} else if (flags & MAP_FORCE_LOW) {
start = 4;
count = 4;
} else if (flags & MAP_FORCE_HIGH) {
start = 8;
count = 8;
}
// Let's check if they are all mapped in a quad somewhere.
// At the same time, check for the quad already being mapped.
// Later we can check for possible transposes as well.
// First just loop over all registers. If it's here and not in range, or overlapped, kick.
std::vector<int> quadsToFlush;
for (int i = 0; i < 16; i++) {
int q = (i + start) & 15;
if (!MappableQ(q))
continue;
// Skip unmapped quads.
if (qr[q].sz == V_Invalid)
continue;
// Check if completely there already. If so, set spill-lock, transfer dirty flag and exit.
if (vreg == qr[q].mipsVec && sz == qr[q].sz) {
if (i < count) {
INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz);
qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY);
qr[q].spillLock = true;
// Sanity check vregs
for (int i = 0; i < n; i++) {
if (vregs[i] != qr[q].vregs[i]) {
ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]);
}
}
return (ARMReg)(Q0 + q);
} else {
INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count);
quadsToFlush.push_back(q);
continue;
}
}
// Check for any overlap. Overlap == flush.
int origN = GetNumVectorElements(qr[q].sz);
for (int a = 0; a < n; a++) {
for (int b = 0; b < origN; b++) {
if (vregs[a] == qr[q].vregs[b]) {
quadsToFlush.push_back(q);
goto doubleBreak;
}
}
}
doubleBreak:
;
}
// We didn't find the extra register, but we got a list of regs to flush. Flush 'em.
// Here we can check for opportunities to do a "transpose-flush" of row vectors, etc.
if (!quadsToFlush.empty()) {
ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size());
}
for (size_t i = 0; i < quadsToFlush.size(); i++) {
QFlush(quadsToFlush[i]);
}
// Find where we want to map it, obeying the constraints we gave.
int quad = QGetFreeQuad(start, count, "mapping");
// If parts of our register are elsewhere, and we are dirty, we need to flush them
// before we reload in a new location.
// This may be problematic if inputs overlap irregularly with output, say:
// vdot S700, R000, C000
// It might still work by accident...
if (flags & MAP_DIRTY) {
for (int i = 0; i < n; i++) {
FlushV(vregs[i]);
}
}
qr[quad].sz = sz;
qr[quad].mipsVec = vreg;
if (!(flags & MAP_NOINIT)) {
// Okay, now we will try to load the whole thing in one go. This is possible
// if it's a row and easy if it's a single.
// Rows are rare, columns are common - but thanks to our register reordering,
// columns are actually in-order in memory.
switch (sz) {
case V_Single:
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
break;
case V_Pair:
if (Consecutive(vregs[0], vregs[1])) {
// Can combine, it's a column!
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
}
break;
case V_Triple:
if (Consecutive(vregs[0], vregs[1], vregs[2])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
} else {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
}
break;
case V_Quad:
if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true);
}
break;
default:
;
}
}
// OK, let's fill out the arrays to confirm that we have grabbed these registers.
for (int i = 0; i < n; i++) {
int mipsReg = 32 + vregs[i];
mr[mipsReg].loc = ML_ARMREG;
mr[mipsReg].reg = QuadAsQ(quad);
mr[mipsReg].lane = i;
qr[quad].vregs[i] = vregs[i];
}
qr[quad].isDirty = (flags & MAP_DIRTY) != 0;
qr[quad].spillLock = true;
INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty);
if (sz == V_Single || sz == V_Pair) {
return D_0(QuadAsQ(quad));
} else {
return QuadAsQ(quad);
}
}
+76 -12
View File
@@ -33,28 +33,55 @@ enum {
TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS,
};
enum {
MAP_MTX_TRANSPOSED = 16,
MAP_PREFER_LOW = 16,
MAP_PREFER_HIGH = 32,
// Force is not yet correctly implemented, if the reg is already mapped it will not move
MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15)
MAP_FORCE_HIGH = 128, // Only map Q8-Q15
};
struct FPURegARM {
int mipsReg; // if -1, no mipsreg attached.
bool isDirty; // Should the register be written back?
};
struct FPURegQuad {
int mipsVec;
VectorSize sz;
u8 vregs[4];
bool isDirty;
bool spillLock;
bool isTemp;
};
struct FPURegMIPS {
// Where is this MIPS register?
RegMIPSLoc loc;
// Data (only one of these is used, depending on loc. Could make a union).
int reg;
int lane;
bool spillLock; // if true, this register cannot be spilled.
bool tempLock;
// If loc == ML_MEM, it's back in its location in the CPU context struct.
};
namespace MIPSComp {
struct ArmJitOptions;
struct JitState;
}
class ArmRegCacheFPU
{
public:
ArmRegCacheFPU(MIPSState *mips);
ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo);
~ArmRegCacheFPU() {}
void Init(ARMXEmitter *emitter);
void Start(MIPSAnalyst::AnalysisResults &stats);
// Protect the arm register containing a MIPS register from spilling, to ensure that
@@ -80,9 +107,11 @@ public:
void MapDirty(MIPSReg rd);
void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true);
void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true);
bool IsMapped(MIPSReg r);
void FlushArmReg(ARMReg r);
void FlushR(MIPSReg r);
void DiscardR(MIPSReg r);
ARMReg R(int preg); // Returns a cached register
// VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode!
void MapRegV(int vreg, int flags = 0);
@@ -90,18 +119,41 @@ public:
void MapInInV(int rt, int rs);
void MapDirtyInV(int rd, int rs, bool avoidLoad = true);
void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true);
bool IsTempX(ARMReg r) const;
MIPSReg GetTempR();
bool IsTempX(ARMReg r) const;
MIPSReg GetTempV() { return GetTempR() - 32; }
// VFPU registers as single VFP registers.
ARMReg V(int vreg) { return R(vreg + 32); }
int FlushGetSequential(int a, int maxArmReg);
void FlushAll();
ARMReg R(int preg); // Returns a cached register
// This one is allowed at any point.
void FlushV(MIPSReg r);
// VFPU registers mapped to match NEON quads (and doubles, for pairs and singles)
// Here we return the ARM register directly instead of providing a "V" accessor
// and so on. Might switch to this model for the other regallocs later.
// Quad mapping does NOT look into the ar array. Instead we use the qr array to keep
// track of what's in each quad.
// Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types.
// Need to explicitly allow spilling to get spilling.
ARMReg QMapReg(int vreg, VectorSize sz, int flags);
// TODO
// Maps a matrix as a set of columns (yes, even transposed ones, always columns
// as those are faster to load/flush). When possible it will map into consecutive
// quad registers, enabling blazing-fast full-matrix loads, transposed or not.
void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags);
ARMReg QAllocTemp(VectorSize sz);
// VFPU registers as single VFP registers
ARMReg V(int vreg) { return R(vreg + 32); }
void QAllowSpill(int quad);
void QFlush(int quad);
void QLoad4x4(MIPSGPReg regPtr, int vquads[4]);
//void FlushQWithV(MIPSReg r);
// NOTE: These require you to release spill locks manually!
void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags);
@@ -112,31 +164,43 @@ public:
void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; }
// For better log output only.
void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; }
int GetMipsRegOffset(MIPSReg r);
private:
bool Consecutive(int v1, int v2) const;
bool Consecutive(int v1, int v2, int v3) const;
bool Consecutive(int v1, int v2, int v3, int v4) const;
MIPSReg GetTempR();
const ARMReg *GetMIPSAllocationOrder(int &count);
int GetMipsRegOffsetV(MIPSReg r) {
return GetMipsRegOffset(r + 32);
}
// This one WILL get a free quad as long as you haven't spill-locked them all.
int QGetFreeQuad(int start, int count, const char *reason);
int GetNumARMFPURegs();
private:
void SetupInitialRegs();
MIPSState *mips_;
ARMXEmitter *emit_;
u32 compilerPC_;
MIPSComp::JitState *js_;
MIPSComp::ArmJitOptions *jo_;
int numARMFpuReg_;
int qTime_;
enum {
MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON
// With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers
// are individually mappable though.
MAX_ARMFPUREG = 32,
MAX_ARMQUADS = 16,
NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS,
};
FPURegARM ar[MAX_ARMFPUREG];
FPURegMIPS mr[NUM_MIPSFPUREG];
FPURegQuad qr[MAX_ARMQUADS];
FPURegMIPS *vr;
bool pendingFlush;