diff --git a/Core/MIPS/x86/CompVFPU.cpp b/Core/MIPS/x86/CompVFPU.cpp index 9368bbf677..b9e70bb3f0 100644 --- a/Core/MIPS/x86/CompVFPU.cpp +++ b/Core/MIPS/x86/CompVFPU.cpp @@ -2390,7 +2390,7 @@ void Jit::Comp_Vmmov(MIPSOpcode op) { MatrixSize sz = GetMtxSize(op); int n = GetMatrixSide(sz); - if (false && jo.enableVFPUSIMD) { + if (jo.enableVFPUSIMD) { VectorSize vsz = GetVectorSize(sz); u8 dest[4][4]; MatrixOverlapType overlap = GetMatrixOverlap(_VD, _VS, sz); @@ -2413,7 +2413,6 @@ void Jit::Comp_Vmmov(MIPSOpcode op) { GetVectorRegs(vec, vsz, vecs[i]); fpr.MapRegsVS(vec, vsz, 0); fpr.MapRegsVS(dest[i], vsz, MAP_NOINIT); - fpr.SpillLockV(dest[i], vsz); MOVAPS(fpr.VSX(dest[i]), fpr.VS(vec)); fpr.ReleaseSpillLocks(); } @@ -2425,7 +2424,6 @@ void Jit::Comp_Vmmov(MIPSOpcode op) { u8 vec[4]; GetVectorRegs(vec, vsz, vecs[i]); fpr.MapRegsVS(vec, vsz, MAP_NOINIT); - fpr.SpillLockV(vec, vsz); fpr.MapRegsVS(dest[i], vsz, 0); MOVAPS(fpr.VSX(vec), fpr.VS(dest[i])); fpr.ReleaseSpillLocks(); diff --git a/Core/MIPS/x86/RegCacheFPU.cpp b/Core/MIPS/x86/RegCacheFPU.cpp index 6dc3193266..e6cbe416d9 100644 --- a/Core/MIPS/x86/RegCacheFPU.cpp +++ b/Core/MIPS/x86/RegCacheFPU.cpp @@ -25,7 +25,7 @@ #include "Core/MIPS/x86/RegCache.h" #include "Core/MIPS/x86/RegCacheFPU.h" -u32 FPURegCache::tempValues[NUM_TEMPS]; +float FPURegCache::tempValues[NUM_TEMPS]; FPURegCache::FPURegCache() : mips(0), initialReady(false), emit(0) { memset(regs, 0, sizeof(regs)); @@ -367,7 +367,7 @@ X64Reg FPURegCache::LoadRegsVS(const u8 *v, int n) { break; } } - const float *f = v[0] < 128 ? &mips->v[voffset[v[0]]] : &mips->v[v[0]]; + const float *f = v[0] < 128 ? &mips->v[voffset[v[0]]] : &tempValues[v[0] - 128]; if (((intptr_t)f & 0x7) == 0 && n == 2) { emit->MOVQ_xmm(res, vregs[v[0]].location); } else if (((intptr_t)f & 0xf) == 0) { @@ -607,30 +607,32 @@ void FPURegCache::StoreFromRegister(int i) { if (regs[i].lane != 0) { const int *mri = xregs[xr].mipsRegs; int seq = 1; - for (int i = 1; i < 4; ++i) { - if (mri[i] == -1) { + for (int j = 1; j < 4; ++j) { + if (mri[j] == -1) { break; } - if (mri[i] - 32 >= 128 && mri[i] == mri[i - 1] + 1) { + if (mri[j] - 32 >= 128 && mri[j] == mri[j - 1] + 1) { seq++; - } else if (mri[i] - 32 < 128 && voffset[mri[i] - 32] == voffset[mri[i - 1] - 32] + 1) { + } else if (mri[j] - 32 < 128 && voffset[mri[j] - 32] == voffset[mri[j - 1] - 32] + 1) { seq++; } else { break; } } + const float *f = mri[0] - 32 < 128 ? &mips->v[voffset[mri[0] - 32]] : &tempValues[mri[0] - 32 - 128]; + int align = (intptr_t)f & 0xf; + // If we can do a multistore... - if (seq == 2 || seq == 4) { + if ((seq == 2 && (align & 0x7) == 0) || seq == 4) { OpArg newLoc = GetDefaultLocation(mri[0]); if (xregs[xr].dirty) { - if (seq == 4) { - _assert_msg_(JIT, (newLoc.offset & 15) == 0, "Unaligned default location for movaps"); + if (seq == 4 && align == 0) emit->MOVAPS(newLoc, xr); - } else { - // _assert_msg_(JIT, (newLoc.offset & 7) == 0, "Unaligned default location for movq"); + else if (seq == 4) + emit->MOVUPS(newLoc, xr); + else emit->MOVQ_xmm(newLoc, xr); - } } for (int j = 0; j < seq; ++j) { int mr = xregs[xr].mipsRegs[j]; diff --git a/Core/MIPS/x86/RegCacheFPU.h b/Core/MIPS/x86/RegCacheFPU.h index c822b72ac3..5cd8e9ec06 100644 --- a/Core/MIPS/x86/RegCacheFPU.h +++ b/Core/MIPS/x86/RegCacheFPU.h @@ -241,7 +241,7 @@ private: X64CachedFPReg xregsInitial[NUM_X_FPREGS]; // TEMP0, etc. are swapped in here if necessary (e.g. on x86.) - static u32 tempValues[NUM_TEMPS]; + static float tempValues[NUM_TEMPS]; XEmitter *emit; MIPSComp::JitOptions *jo_;