Merge pull request #7141 from hrydgard/neon-vfpu-2

The old neon-vfpu branch, rebased on master
This commit is contained in:
Unknown W. Brackets committed 2014-12-06 16:06:50 -08:00
commit a893c213de
15 files changed
+2403 -97

No files matched your search

+1
View File
@@ -1022,6 +1022,7 @@ if(ARM)
Core/MIPS/ARM/ArmCompLoadStore.cpp
Core/MIPS/ARM/ArmCompVFPU.cpp
Core/MIPS/ARM/ArmCompVFPUNEON.cpp
Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp
Core/MIPS/ARM/ArmCompReplace.cpp
Core/MIPS/ARM/ArmJit.cpp
Core/MIPS/ARM/ArmJit.h
+14 -2
View File
@@ -317,6 +317,18 @@
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
</ClCompile>
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
</ClCompile>
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEONUtil.cpp">
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
</ClCompile>
<ClCompile Include="MIPS\ARM\ArmRegCacheFPU.cpp">
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
@@ -563,7 +575,7 @@
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|x64'">true</ExcludedFromBuild>
</ClInclude>
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEONUtil.h">
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Release|Win32'">true</ExcludedFromBuild>
<ExcludedFromBuild Condition="'$(Configuration)|$(Platform)'=='Debug|x64'">true</ExcludedFromBuild>
@@ -658,4 +670,4 @@
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
<ImportGroup Label="ExtensionTargets">
</ImportGroup>
</Project>
</Project>
+8 -2
View File
@@ -393,6 +393,12 @@
<ClCompile Include="MIPS\ARM\ArmCompVFPU.cpp">
<Filter>MIPS\ARM</Filter>
</ClCompile>
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
<Filter>MIPS\ARM</Filter>
</ClCompile>
<ClCompile Include="MIPS\ARM\ArmCompVFPUNEONUtil.cpp">
<Filter>MIPS\ARM</Filter>
</ClCompile>
<ClCompile Include="..\ext\disarm.cpp">
<Filter>Ext</Filter>
</ClCompile>
@@ -969,7 +975,7 @@
<ClInclude Include="MIPS\JitCommon\JitState.h">
<Filter>MIPS\JitCommon</Filter>
</ClInclude>
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEON.cpp">
<ClInclude Include="MIPS\ARM\ArmCompVFPUNEONUtil.h">
<Filter>MIPS\ARM</Filter>
</ClInclude>
<ClInclude Include="Debugger\DisassemblyManager.h">
@@ -1027,4 +1033,4 @@
<None Include="..\android\jni\Android.mk" />
<None Include="GameLogNotes.txt" />
</ItemGroup>
</Project>
</Project>
+7 -4
View File
@@ -357,9 +357,12 @@ void Jit::Comp_mxc1(MIPSOpcode op)
switch ((op >> 21) & 0x1f)
{
case 0: // R(rt) = FI(fs); break; //mfc1
fpr.MapReg(fs);
gpr.MapReg(rt, MAP_DIRTY | MAP_NOINIT);
VMOV(gpr.R(rt), fpr.R(fs));
if (fpr.IsMapped(fs)) {
VMOV(gpr.R(rt), fpr.R(fs));
} else {
LDR(gpr.R(rt), CTXREG, fpr.GetMipsRegOffset(fs));
}
return;
case 2: //cfc1
@@ -393,11 +396,11 @@ void Jit::Comp_mxc1(MIPSOpcode op)
case 4: //FI(fs) = R(rt); break; //mtc1
if (rt == MIPS_REG_ZERO) {
fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT);
fpr.MapReg(fs, MAP_NOINIT);
MOVI2F(fpr.R(fs), 0.0f, R0);
} else {
gpr.MapReg(rt);
fpr.MapReg(fs, MAP_DIRTY | MAP_NOINIT);
fpr.MapReg(fs, MAP_NOINIT);
VMOV(fpr.R(fs), gpr.R(rt));
}
return;
+4 -4
View File
@@ -324,8 +324,8 @@ namespace MIPSComp
void Jit::Comp_SVQ(MIPSOpcode op)
{
CONDITIONAL_DISABLE;
NEON_IF_AVAILABLE(CompNEON_SVQ);
CONDITIONAL_DISABLE;
int imm = (signed short)(op&0xFFFC);
int vt = (((op >> 16) & 0x1f)) | ((op&1) << 5);
@@ -506,6 +506,7 @@ namespace MIPSComp
void Jit::Comp_VIdt(MIPSOpcode op) {
NEON_IF_AVAILABLE(CompNEON_VIdt);
CONDITIONAL_DISABLE;
if (js.HasUnknownPrefix()) {
DISABLE;
@@ -1574,7 +1575,7 @@ namespace MIPSComp
VMUL(fpr.V(temp3), fpr.V(sregs[0]), fpr.V(tregs[1]));
VMLS(fpr.V(temp3), fpr.V(sregs[1]), fpr.V(tregs[0]));
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT);
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT);
VMOV(fpr.V(dregs[0]), S0);
VMOV(fpr.V(dregs[1]), S1);
VMOV(fpr.V(dregs[2]), fpr.V(temp3));
@@ -1609,7 +1610,7 @@ namespace MIPSComp
VMLS(fpr.V(temp4), fpr.V(sregs[2]), fpr.V(tregs[2]));
VMLA(fpr.V(temp4), fpr.V(sregs[3]), fpr.V(tregs[3]));
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_DIRTY | MAP_NOINIT);
fpr.MapRegsAndSpillLockV(dregs, sz, MAP_NOINIT);
VMOV(fpr.V(dregs[0]), S0);
VMOV(fpr.V(dregs[1]), S1);
VMOV(fpr.V(dregs[2]), fpr.V(temp3));
@@ -2118,5 +2119,4 @@ namespace MIPSComp
void Jit::Comp_Vbfy(MIPSOpcode op) {
DISABLE;
}
}
File diff suppressed because it is too large. Load diff
+419
View File
@@ -0,0 +1,419 @@
// Copyright (c) 2013- PPSSPP Project.
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU General Public License as published by
// the Free Software Foundation, version 2.0 or later versions.
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU General Public License 2.0 for more details.
// A copy of the GPL 2.0 should have been included with the program.
// If not, see http://www.gnu.org/licenses/
// Official git repository and contact information can be found at
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
// NEON VFPU
// This is where we will create an alternate implementation of the VFPU emulation
// that uses NEON Q registers to cache pairs/tris/quads, and so on.
// Will require major extensions to the reg cache and other things.
// ARM NEON can only do pairs and quads, not tris and scalars.
// We can do scalars, though, for many operations if all the operands
// are below Q8 (D16, S32) using regular VFP instructions but really not sure
// if it's worth it.
#include <cmath>
#include "base/logging.h"
#include "math/math_util.h"
#include "Common/CPUDetect.h"
#include "Core/MemMap.h"
#include "Core/MIPS/MIPS.h"
#include "Core/MIPS/MIPSAnalyst.h"
#include "Core/MIPS/MIPSCodeUtils.h"
#include "Core/MIPS/MIPSVFPUUtils.h"
#include "Core/Config.h"
#include "Core/Reporting.h"
#include "Core/MIPS/ARM/ArmJit.h"
#include "Core/MIPS/ARM/ArmRegCache.h"
#include "Core/MIPS/ARM/ArmCompVFPUNEONUtil.h"
// TODO: Somehow #ifdef away on ARMv5eabi, without breaking the linker.
// #define CONDITIONAL_DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
#define CONDITIONAL_DISABLE ;
#define DISABLE { fpr.ReleaseSpillLocksAndDiscardTemps(); Comp_Generic(op); return; }
#define _RS MIPS_GET_RS(op)
#define _RT MIPS_GET_RT(op)
#define _RD MIPS_GET_RD(op)
#define _FS MIPS_GET_FS(op)
#define _FT MIPS_GET_FT(op)
#define _FD MIPS_GET_FD(op)
#define _SA MIPS_GET_SA(op)
#define _POS ((op>> 6) & 0x1F)
#define _SIZE ((op>>11) & 0x1F)
#define _IMM16 (signed short)(op & 0xFFFF)
#define _IMM26 (op & 0x03FFFFFF)
namespace MIPSComp {
static const float minus_one = -1.0f;
static const float one = 1.0f;
static const float zero = 0.0f;
// On NEON, we map triples to Q registers and singles to D registers.
// Sometimes, as when doing dot products, it matters what's in that unused reg. This zeroes it.
void Jit::NEONMaskToSize(ARMReg vs, VectorSize sz) {
// TODO
}
ARMReg Jit::NEONMapPrefixST(int mipsReg, VectorSize sz, u32 prefix, int mapFlags) {
static const float constantArray[8] = { 0.f, 1.f, 2.f, 0.5f, 3.f, 1.f / 3.f, 0.25f, 1.f / 6.f };
static const float constantArrayNegated[8] = { -0.f, -1.f, -2.f, -0.5f, -3.f, -1.f / 3.f, -0.25f, -1.f / 6.f };
// Applying prefixes in SIMD fashion will actually be a lot easier than the old style.
if (prefix == 0xE4) {
return fpr.QMapReg(mipsReg, sz, mapFlags);
}
int n = GetNumVectorElements(sz);
int regnum[4] = { -1, -1, -1, -1 };
int abs[4] = { 0 };
int negate[4] = { 0 };
int constants[4] = { 0 };
int constNum[4] = { 0 };
int full_mask = (1 << n) - 1;
int abs_mask = (prefix >> 8) & full_mask;
int negate_mask = (prefix >> 16) & full_mask;
int constants_mask = (prefix >> 12) & full_mask;
// Decode prefix to keep the rest readable
int permuteMask = 0;
for (int i = 0; i < n; i++) {
permuteMask |= 3 << (i * 2);
regnum[i] = (prefix >> (i * 2)) & 3;
abs[i] = (prefix >> (8 + i)) & 1;
negate[i] = (prefix >> (16 + i)) & 1;
constants[i] = (prefix >> (12 + i)) & 1;
if (constants[i]) {
constNum[i] = regnum[i] + (abs[i] << 2);
abs[i] = 0;
}
}
abs_mask &= ~constants_mask;
bool anyPermute = (prefix & permuteMask) != (0xE4 & permuteMask);
if (constants_mask == full_mask) {
// It's all constants! Don't even bother mapping the input register,
// just allocate a temp one.
// If a single, this can sometimes be done cheaper. But meh.
ARMReg ar = fpr.QAllocTemp(sz);
for (int i = 0; i < n; i++) {
if ((i & 1) == 0) {
if (constNum[i] == constNum[i + 1]) {
// Replace two loads with a single immediate when easily possible.
ARMReg dest = i & 2 ? D_1(ar) : D_0(ar);
switch (constNum[i]) {
case 0:
case 1:
{
float c = constantArray[constNum[i]];
VMOV_immf(dest, negate[i] ? -c : c);
}
break;
// TODO: There are a few more that are doable.
default:
goto skip;
}
i++;
continue;
skip:
;
}
}
MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]);
VLD1_lane(F_32, ar, R0, i, true);
}
return ar;
}
// 1. Permute.
// 2. Abs
// If any constants:
// 3. Replace values with constants
// 4. Negate
ARMReg inputAR = fpr.QMapReg(mipsReg, sz, mapFlags);
ARMReg ar = fpr.QAllocTemp(sz);
if (!anyPermute) {
VMOV(ar, inputAR);
// No permutations!
} else {
bool allSame = false;
for (int i = 1; i < n; i++) {
if (regnum[0] == regnum[i])
allSame = true;
}
if (allSame) {
// Easy, someone is duplicating one value onto all the reg parts.
// If this is happening and QMapReg must load, we can combine these two actions
// into a VLD1_lane. TODO
VDUP(F_32, ar, inputAR, regnum[0]);
} else {
// Do some special cases
if (regnum[0] == 1 && regnum[1] == 0) {
INFO_LOG(HLE, "PREFIXST: Bottom swap!");
VREV64(I_32, ar, inputAR);
regnum[0] = 0;
regnum[1] = 1;
}
// TODO: Make a generic fallback using another temp register
bool match = true;
for (int i = 0; i < n; i++) {
if (regnum[i] != i)
match = false;
}
// TODO: Cannot do this permutation yet!
if (!match) {
ERROR_LOG(HLE, "PREFIXST: Unsupported permute! %i %i %i %i / %i", regnum[0], regnum[1], regnum[2], regnum[3], n);
VMOV(ar, inputAR);
}
}
}
// ABS
// Two methods: If all lanes are "absoluted", it's easy.
if (abs_mask == full_mask) {
// TODO: elide the above VMOV (in !anyPermute) when possible
VABS(F_32, ar, ar);
} else if (abs_mask != 0) {
// Partial ABS!
if (abs_mask == 3) {
VABS(F_32, D_0(ar), D_0(ar));
} else {
// Horrifying fallback: Mov to Q0, abs, move back.
// TODO: Optimize for lower quads where we don't need to move.
VMOV(MatchSize(Q0, ar), ar);
for (int i = 0; i < n; i++) {
if (abs_mask & (1 << i)) {
VABS((ARMReg)(S0 + i), (ARMReg)(S0 + i));
}
}
VMOV(ar, MatchSize(Q0, ar));
INFO_LOG(HLE, "PREFIXST: Partial ABS %i/%i! Slow fallback generated.", abs_mask, full_mask);
}
}
if (negate_mask == full_mask) {
// TODO: elide the above VMOV when possible
VNEG(F_32, ar, ar);
} else if (negate_mask != 0) {
// Partial negate! I guess we build sign bits in another register
// and simply XOR.
if (negate_mask == 3) {
VNEG(F_32, D_0(ar), D_0(ar));
} else {
// Horrifying fallback: Mov to Q0, negate, move back.
// TODO: Optimize for lower quads where we don't need to move.
VMOV(MatchSize(Q0, ar), ar);
for (int i = 0; i < n; i++) {
if (negate_mask & (1 << i)) {
VNEG((ARMReg)(S0 + i), (ARMReg)(S0 + i));
}
}
VMOV(ar, MatchSize(Q0, ar));
INFO_LOG(HLE, "PREFIXST: Partial Negate %i/%i! Slow fallback generated.", negate_mask, full_mask);
}
}
// Insert constants where requested, and check negate!
for (int i = 0; i < n; i++) {
if (constants[i]) {
MOVP2R(R0, (negate[i] ? constantArrayNegated : constantArray) + constNum[i]);
VLD1_lane(F_32, ar, R0, i, true);
}
}
return ar;
}
Jit::DestARMReg Jit::NEONMapPrefixD(int vreg, VectorSize sz, int mapFlags) {
// Inverted from the actual bits, easier to reason about 1 == write
int writeMask = (~(js.prefixD >> 8)) & 0xF;
int n = GetNumVectorElements(sz);
int full_mask = (1 << n) - 1;
DestARMReg dest;
dest.sz = sz;
if ((writeMask & full_mask) == full_mask) {
// No need to apply a write mask.
// Let's not make things complicated.
dest.rd = fpr.QMapReg(vreg, sz, mapFlags);
dest.backingRd = dest.rd;
} else {
// Allocate a temporary register.
ELOG("PREFIXD: Write mask allocated! %i/%i", writeMask, full_mask);
dest.rd = fpr.QAllocTemp(sz);
dest.backingRd = fpr.QMapReg(vreg, sz, mapFlags & ~MAP_NOINIT); // Force initialization of the backing reg.
}
return dest;
}
void Jit::NEONApplyPrefixD(DestARMReg dest) {
// Apply clamps to dest.rd
int n = GetNumVectorElements(dest.sz);
int sat1_mask = 0;
int sat3_mask = 0;
int full_mask = (1 << n) - 1;
for (int i = 0; i < n; i++) {
int sat = (js.prefixD >> (i * 2)) & 3;
if (sat == 1)
sat1_mask |= 1 << i;
if (sat == 3)
sat3_mask |= 1 << i;
}
if (sat1_mask && sat3_mask) {
// Why would anyone do this?
ELOG("PREFIXD: Can't have both sat[0-1] and sat[-1-1] at the same time yet");
}
if (sat1_mask) {
if (sat1_mask != full_mask) {
ELOG("PREFIXD: Can't have partial sat1 mask yet (%i vs %i)", sat1_mask, full_mask);
}
if (IsD(dest.rd)) {
VMOV_immf(D0, 0.0);
VMOV_immf(D1, 1.0);
VMAX(F_32, dest.rd, dest.rd, D0);
VMIN(F_32, dest.rd, dest.rd, D1);
} else {
VMOV_immf(Q0, 1.0);
VMIN(F_32, dest.rd, dest.rd, Q0);
VMOV_immf(Q0, 0.0);
VMAX(F_32, dest.rd, dest.rd, Q0);
}
}
if (sat3_mask && sat1_mask != full_mask) {
if (sat3_mask != full_mask) {
ELOG("PREFIXD: Can't have partial sat3 mask yet (%i vs %i)", sat3_mask, full_mask);
}
if (IsD(dest.rd)) {
VMOV_immf(D0, 0.0);
VMOV_immf(D1, 1.0);
VMAX(F_32, dest.rd, dest.rd, D0);
VMIN(F_32, dest.rd, dest.rd, D1);
} else {
VMOV_immf(Q0, 1.0);
VMIN(F_32, dest.rd, dest.rd, Q0);
VMOV_immf(Q0, -1.0);
VMAX(F_32, dest.rd, dest.rd, Q0);
}
}
// Check for actual mask operation (unrelated to the "masks" above).
if (dest.backingRd != dest.rd) {
// This means that we need to apply the write mask, from rd to backingRd.
// What a pain. We can at least shortcut easy cases like half the register.
// And we can generate the masks easily with some of the crazy vector imm modes. (bits2bytes for example).
// So no need to load them from RAM.
int writeMask = (~(js.prefixD >> 8)) & 0xF;
if (writeMask == 3) {
ILOG("Doing writemask = 3");
VMOV(D_0(dest.rd), D_0(dest.backingRd));
} else {
// TODO
ELOG("PREFIXD: Arbitrary write masks not supported (%i / %i)", writeMask, full_mask);
VMOV(dest.backingRd, dest.rd);
}
}
}
Jit::MappedRegs Jit::NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes) {
MappedRegs regs;
if (applyPrefixes) {
regs.vs = NEONMapPrefixS(_VS, ssize, 0);
regs.vt = NEONMapPrefixT(_VT, tsize, 0);
} else {
regs.vs = fpr.QMapReg(_VS, ssize, 0);
regs.vt = fpr.QMapReg(_VT, ssize, 0);
}
regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0 || GetVectorOverlap(_VD, dsize, _VT, ssize);
if (applyPrefixes) {
regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
} else {
regs.vd.rd = fpr.QMapReg(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
regs.vd.backingRd = regs.vd.rd;
regs.vd.sz = dsize;
}
return regs;
}
Jit::MappedRegs Jit::NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes) {
MappedRegs regs;
if (applyPrefixes) {
regs.vs = NEONMapPrefixS(_VS, ssize, 0);
regs.vt = NEONMapPrefixT(_VT, tsize, 0);
} else {
regs.vs = fpr.QMapReg(_VS, ssize, 0);
regs.vt = fpr.QMapReg(_VT, ssize, 0);
}
regs.vd.rd = INVALID_REG;
regs.vd.sz = V_Invalid;
return regs;
}
Jit::MappedRegs Jit::NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes) {
MappedRegs regs;
regs.vs = NEONMapPrefixS(_VS, ssize, 0);
regs.overlap = GetVectorOverlap(_VD, dsize, _VS, ssize) > 0;
regs.vd = NEONMapPrefixD(_VD, dsize, MAP_DIRTY | (regs.overlap ? 0 : MAP_NOINIT));
return regs;
}
// Requires quad registers.
void Jit::NEONTranspose4x4(ARMReg cols[4]) {
// 0123 _\ 0426
// 4567 / 1537
VTRN(F_32, cols[0], cols[1]);
// 89ab _\ 8cae
// cdef / 9dbf
VTRN(F_32, cols[2], cols[3]);
// 04[26] 048c
// 15 37 -> 1537
// [8c]ae 26ae
// 9d bf 9dbf
VSWP(D_1(cols[0]), D_0(cols[2]));
// 04 8c 048c
// 15[37] -> 159d
// 26 ae 26ae
// [9d]bf 37bf
VSWP(D_1(cols[1]), D_0(cols[3]));
}
} // namespace MIPSComp
+20
View File
@@ -0,0 +1,20 @@
#pragma once
#include "Core/MIPS/ARM/ArmJit.h"
#include "Core/MIPS/ARM/ArmRegCache.h"
namespace MIPSComp {
inline ARMReg MatchSize(ARMReg x, ARMReg target) {
if (IsQ(target) && IsQ(x))
return x;
if (IsD(target) && IsD(x))
return x;
if (IsD(target) && IsQ(x))
return D_0(x);
// if (IsQ(target) && IsD(x))
return (ARMReg)(D0 + (x - Q0) * 2);
}
}
+1 -2
View File
@@ -78,7 +78,7 @@ ArmJitOptions::ArmJitOptions() {
useNEONVFPU = false;
}
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_(mips)
Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips, &js, &jo), mips_(mips)
{
logBlocks = 0;
dontLogBlocks = 0;
@@ -297,7 +297,6 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b)
while (js.compiling)
{
gpr.SetCompilerPC(js.compilerPC); // Let it know for log messages
fpr.SetCompilerPC(js.compilerPC);
MIPSOpcode inst = Memory::Read_Opcode_JIT(js.compilerPC);
js.downcountAmount += MIPSGetInstructionCycleEstimate(inst);
+38
View File
@@ -17,6 +17,7 @@
#pragma once
#include "Common/CPUDetect.h"
#include "Core/MIPS/JitCommon/JitState.h"
#include "Core/MIPS/JitCommon/JitBlockCache.h"
#include "Core/MIPS/ARM/ArmRegCache.h"
@@ -240,6 +241,43 @@ private:
}
void GetVectorRegsPrefixD(u8 *regs, VectorSize sz, int vectorReg);
// For NEON mappings, it will be easier to deal directly in ARM registers.
ARMReg NEONMapPrefixST(int vfpuReg, VectorSize sz, u32 prefix, int mapFlags);
ARMReg NEONMapPrefixS(int vfpuReg, VectorSize sz, int mapFlags) {
return NEONMapPrefixST(vfpuReg, sz, js.prefixS, mapFlags);
}
ARMReg NEONMapPrefixT(int vfpuReg, VectorSize sz, int mapFlags) {
return NEONMapPrefixST(vfpuReg, sz, js.prefixT, mapFlags);
}
struct DestARMReg {
ARMReg rd;
ARMReg backingRd;
VectorSize sz;
operator ARMReg() const { return rd; }
};
struct MappedRegs {
ARMReg vs;
ARMReg vt;
DestARMReg vd;
bool overlap;
};
MappedRegs NEONMapDirtyInIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true);
MappedRegs NEONMapInIn(MIPSOpcode op, VectorSize ssize, VectorSize tsize, bool applyPrefixes = true);
MappedRegs NEONMapDirtyIn(MIPSOpcode op, VectorSize dsize, VectorSize ssize, bool applyPrefixes = true);
DestARMReg NEONMapPrefixD(int vfpuReg, VectorSize sz, int mapFlags);
void NEONApplyPrefixD(DestARMReg dest);
// NEON utils
void NEONMaskToSize(ARMReg vs, VectorSize sz);
void NEONTranspose4x4(ARMReg cols[4]);
// Utils
void SetR0ToEffectiveAddress(MIPSGPReg rs, s16 offset);
void SetCCAndR0ForSafeAddress(MIPSGPReg rs, s16 offset, ARMReg tempReg, bool reverse = false);
+4
View File
@@ -104,6 +104,10 @@ bool ArmRegCache::IsMappedAsPointer(MIPSGPReg mipsReg) {
return mr[mipsReg].loc == ML_ARMREG_AS_PTR;
}
bool ArmRegCache::IsMapped(MIPSGPReg mipsReg) {
return mr[mipsReg].loc == ML_ARMREG;
}
void ArmRegCache::SetRegImm(ARMReg reg, u32 imm) {
// If we can do it with a simple Operand2, let's do that.
Operand2 op2;
+1
View File
@@ -103,6 +103,7 @@ public:
ARMReg MapReg(MIPSGPReg reg, int mapFlags = 0);
ARMReg MapRegAsPointer(MIPSGPReg reg); // read-only, non-dirty.
bool IsMapped(MIPSGPReg reg);
bool IsMappedAsPointer(MIPSGPReg reg);
void MapInIn(MIPSGPReg rd, MIPSGPReg rs);
+521 -42
View File
@@ -16,14 +16,17 @@
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
#include <cstring>
#include "base/logging.h"
#include "Common/CPUDetect.h"
#include "Core/MIPS/MIPS.h"
#include "Core/MIPS/ARM/ArmRegCacheFPU.h"
#include "Core/MIPS/ARM/ArmJit.h"
#include "Core/MIPS/MIPSTables.h"
using namespace ArmGen;
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), initialReady(false) {
ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo) : mips_(mips), vr(mr + 32), js_(js), jo_(jo), initialReady(false) {
if (cpu_info.bNEON) {
numARMFpuReg_ = 32;
} else {
@@ -31,10 +34,6 @@ ArmRegCacheFPU::ArmRegCacheFPU(MIPSState *mips) : mips_(mips), vr(mr + 32), init
}
}
void ArmRegCacheFPU::Init(ARMXEmitter *emitter) {
emit_ = emitter;
}
void ArmRegCacheFPU::Start(MIPSAnalyst::AnalysisResults &stats) {
if (!initialReady) {
SetupInitialRegs();
@@ -57,9 +56,17 @@ void ArmRegCacheFPU::SetupInitialRegs() {
mrInitial[i].spillLock = false;
mrInitial[i].tempLock = false;
}
for (int i = 0; i < MAX_ARMQUADS; i++) {
qr[i].isDirty = false;
qr[i].mipsVec = -1;
qr[i].sz = V_Invalid;
qr[i].spillLock = false;
qr[i].isTemp = false;
memset(qr[i].vregs, 0xff, 4);
}
}
static const ARMReg *GetMIPSAllocationOrder(int &count) {
const ARMReg *ArmRegCacheFPU::GetMIPSAllocationOrder(int &count) {
// We reserve S0-S1 as scratch. Can afford two registers. Maybe even four, which could simplify some things.
static const ARMReg allocationOrder[] = {
S2, S3,
@@ -68,10 +75,16 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
S12, S13, S14, S15
};
// With NEON, we have many more.
// In the future I plan to use S0-S7 (Q0-Q1) for FPU and S8 forwards (Q2-Q15, yes, 15) for VFPU.
// VFPU will use NEON to do SIMD and it will be awkward to mix with FPU.
// VFP mapping
// VFPU registers and regular FP registers are mapped interchangably on top of the standard
// 16 FPU registers.
// NEON mapping
// We map FPU and VFPU registers entirely separately. FPU is mapped to 12 of the bottom 16 S registers.
// VFPU is mapped to the upper 48 regs, 32 of which can only be reached through NEON
// (or D16-D31 as doubles, but not relevant).
// Might consider shifting the split in the future, giving more regs to NEON allowing it to map more quads.
// We should attempt to map scalars to low Q registers and wider things to high registers,
// as the NEON instructions are all 2-vector or 4-vector, they don't do scalar, we want to be
// able to use regular VFP instructions too.
@@ -81,14 +94,10 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
S4, S5, S6, S7, // Q1
S8, S9, S10, S11, // Q2
S12, S13, S14, S15, // Q3
S16, S17, S18, S19, // Q4
S20, S21, S22, S23, // Q5
S24, S25, S26, S27, // Q6
S28, S29, S30, S31, // Q7
// Q8-Q15 free for NEON tricks
// Q4-Q15 free for VFPU
};
if (cpu_info.bNEON) {
if (jo_->useNEONVFPU) {
count = sizeof(allocationOrderNEON) / sizeof(const int);
return allocationOrderNEON;
} else {
@@ -97,7 +106,17 @@ static const ARMReg *GetMIPSAllocationOrder(int &count) {
}
}
bool ArmRegCacheFPU::IsMapped(MIPSReg r) {
return mr[r].loc == ML_ARMREG;
}
ARMReg ArmRegCacheFPU::MapReg(MIPSReg mipsReg, int mapFlags) {
// INFO_LOG(JIT, "FPR MapReg: %i flags=%i", mipsReg, mapFlags);
if (jo_->useNEONVFPU && mipsReg >= 32) {
ERROR_LOG(JIT, "Cannot map VFPU registers to ARM VFP registers in NEON mode. PC=%08x", js_->compilerPC);
return S0;
}
pendingFlush = true;
// Let's see if it's already mapped. If so we just need to update the dirty flag.
// We don't need to check for ML_NOINIT because we assume that anyone who maps
@@ -157,7 +176,7 @@ allocate:
}
// Uh oh, we have all them spilllocked....
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", mips_->pc);
ERROR_LOG(JIT, "Out of spillable registers at PC %08x!!!", js_->compilerPC);
return INVALID_REG;
}
@@ -263,26 +282,66 @@ void ArmRegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
}
void ArmRegCacheFPU::FlushArmReg(ARMReg r) {
int reg = r - S0;
if (ar[reg].mipsReg == -1) {
// Nothing to do, reg not mapped.
return;
}
if (ar[reg].mipsReg != -1) {
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG) {
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
if (r >= S0 && r <= S31) {
int reg = r - S0;
if (ar[reg].mipsReg == -1) {
// Nothing to do, reg not mapped.
return;
}
// IMMs won't be in an ARM reg.
mr[ar[reg].mipsReg].loc = ML_MEM;
mr[ar[reg].mipsReg].reg = INVALID_REG;
} else {
ERROR_LOG(JIT, "Dirty but no mipsreg?");
if (ar[reg].mipsReg != -1) {
if (ar[reg].isDirty && mr[ar[reg].mipsReg].loc == ML_ARMREG)
{
//INFO_LOG(JIT, "Flushing ARM reg %i", reg);
emit_->VSTR(r, CTXREG, GetMipsRegOffset(ar[reg].mipsReg));
}
// IMMs won't be in an ARM reg.
mr[ar[reg].mipsReg].loc = ML_MEM;
mr[ar[reg].mipsReg].reg = INVALID_REG;
} else {
ERROR_LOG(JIT, "Dirty but no mipsreg?");
}
ar[reg].isDirty = false;
ar[reg].mipsReg = -1;
} else if (r >= D0 && r <= D31) {
// TODO: Convert to S regs and flush them individually.
} else if (r >= Q0 && r <= Q15) {
int quad = r - Q0;
QFlush(r);
}
ar[reg].isDirty = false;
ar[reg].mipsReg = -1;
}
void ArmRegCacheFPU::FlushV(MIPSReg r) {
FlushR(r + 32);
}
/*
void ArmRegCacheFPU::FlushQWithV(MIPSReg r) {
// Look for it in all the quads. If it's in any, flush that quad clean.
int flushCount = 0;
for (int i = 0; i < MAX_ARMQUADS; i++) {
if (qr[i].sz == V_Invalid)
continue;
int n = qr[i].sz;
bool flushThis = false;
for (int j = 0; j < n; j++) {
if (qr[i].vregs[j] == r) {
flushThis = true;
}
}
if (flushThis) {
QFlush(i);
flushCount++;
}
}
if (flushCount > 1) {
WARN_LOG(JIT, "ERROR: More than one quad was flushed to flush reg %i", r);
}
}
*/
void ArmRegCacheFPU::FlushR(MIPSReg r) {
switch (mr[r].loc) {
case ML_IMM:
@@ -295,12 +354,24 @@ void ArmRegCacheFPU::FlushR(MIPSReg r) {
if (mr[r].reg == (int)INVALID_REG) {
ERROR_LOG(JIT, "FlushR: MipsReg had bad ArmReg");
}
if (ar[mr[r].reg].isDirty) {
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
ar[mr[r].reg].isDirty = false;
if (mr[r].reg >= Q0 && mr[r].reg <= Q15) {
// This should happen rarely, but occasionally we need to flush a single stray
// mipsreg that's been part of a quad.
int quad = mr[r].reg - Q0;
if (qr[quad].isDirty) {
WARN_LOG(JIT, "FlushR found quad register %i - PC=%08x", quad, js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffset(r), R1);
emit_->VST1_lane(F_32, (ARMReg)mr[r].reg, R0, mr[r].lane, true);
}
} else {
if (ar[mr[r].reg].isDirty) {
//INFO_LOG(JIT, "Flushing dirty reg %i", mr[r].reg);
emit_->VSTR((ARMReg)(mr[r].reg + S0), CTXREG, GetMipsRegOffset(r));
ar[mr[r].reg].isDirty = false;
}
ar[mr[r].reg].mipsReg = -1;
}
ar[mr[r].reg].mipsReg = -1;
break;
case ML_MEM:
@@ -322,6 +393,7 @@ int ArmRegCacheFPU::GetNumARMFPURegs() {
return 16;
}
// Scalar only. Need a similar one for sequential Q vectors.
int ArmRegCacheFPU::FlushGetSequential(int a, int maxArmReg) {
int c = 1;
int lastMipsOffset = GetMipsRegOffset(ar[a].mipsReg);
@@ -352,6 +424,12 @@ void ArmRegCacheFPU::FlushAll() {
DiscardR(i);
}
// Flush quads!
// These could also use sequential detection.
for (int i = 4; i < MAX_ARMQUADS; i++) {
QFlush(i);
}
// Loop through the ARM registers, then use GetMipsRegOffset to determine if MIPS registers are
// sequential. This is necessary because we store VFPU registers in a staggered order to get
// columns sequential (most VFPU math in nearly all games is in columns, not rows).
@@ -451,6 +529,10 @@ bool ArmRegCacheFPU::IsTempX(ARMReg r) const {
}
int ArmRegCacheFPU::GetTempR() {
if (jo_->useNEONVFPU) {
ERROR_LOG(JIT, "VFP temps not allowed in NEON mode");
return 0;
}
pendingFlush = true;
for (int r = TEMP0; r < TEMP0 + NUM_TEMPS; ++r) {
if (mr[r].loc == ML_MEM && !mr[r].tempLock) {
@@ -488,10 +570,19 @@ void ArmRegCacheFPU::SpillLock(MIPSReg r1, MIPSReg r2, MIPSReg r3, MIPSReg r4) {
// This is actually pretty slow with all the 160 regs...
void ArmRegCacheFPU::ReleaseSpillLocksAndDiscardTemps() {
for (int i = 0; i < NUM_MIPSFPUREG; i++)
for (int i = 0; i < NUM_MIPSFPUREG; i++) {
mr[i].spillLock = false;
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i)
}
for (int i = TEMP0; i < TEMP0 + NUM_TEMPS; ++i) {
DiscardR(i);
}
for (int i = 0; i < MAX_ARMQUADS; i++) {
qr[i].spillLock = false;
if (qr[i].isTemp) {
qr[i].isTemp = false;
qr[i].sz = V_Invalid;
}
}
}
ARMReg ArmRegCacheFPU::R(int mipsReg) {
@@ -499,12 +590,400 @@ ARMReg ArmRegCacheFPU::R(int mipsReg) {
return (ARMReg)(mr[mipsReg].reg + S0);
} else {
if (mipsReg < 32) {
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, compilerPC_, MIPSDisasmAt(compilerPC_));
ERROR_LOG(JIT, "FReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
} else if (mipsReg < 32 + 128) {
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
ERROR_LOG(JIT, "VReg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
} else {
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, compilerPC_, MIPSDisasmAt(compilerPC_));
ERROR_LOG(JIT, "Tempreg %i not in ARM reg. compilerPC = %08x : %s", mipsReg - 128 - 32, js_->compilerPC, MIPSDisasmAt(js_->compilerPC));
}
return INVALID_REG; // BAAAD
}
}
inline ARMReg QuadAsD(int quad) {
return (ARMReg)(D0 + quad * 2);
}
inline ARMReg QuadAsQ(int quad) {
return (ARMReg)(Q0 + quad);
}
bool MappableQ(int quad) {
return quad >= 4;
}
void ArmRegCacheFPU::QLoad4x4(MIPSGPReg regPtr, int vquads[4]) {
ERROR_LOG(JIT, "QLoad4x4 not implemented");
// TODO
}
void ArmRegCacheFPU::QFlush(int quad) {
if (!MappableQ(quad)) {
ERROR_LOG(JIT, "Cannot flush non-mappable quad %i", quad);
return;
}
if (qr[quad].isDirty && !qr[quad].isTemp) {
INFO_LOG(JIT, "Flushing Q%i (%s)", quad, GetVectorNotation(qr[quad].mipsVec, qr[quad].sz));
ARMReg q = QuadAsQ(quad);
// Unlike reads, when writing to the register file we need to be careful to write the correct
// number of floats.
switch (qr[quad].sz) {
case V_Single:
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
// WARN_LOG(JIT, "S: Falling back to individual flush: pc=%08x", js_->compilerPC);
break;
case V_Pair:
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1])) {
// Can combine, it's a column!
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1(F_32, q, R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
// WARN_LOG(JIT, "P: Falling back to individual flush: pc=%08x", js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
emit_->VST1_lane(F_32, q, R0, 1, true);
}
break;
case V_Triple:
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
emit_->VST1_lane(F_32, q, R0, 2, true);
} else {
// WARN_LOG(JIT, "T: Falling back to individual flush: pc=%08x", js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
emit_->VST1_lane(F_32, q, R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
emit_->VST1_lane(F_32, q, R0, 2, true);
}
break;
case V_Quad:
if (Consecutive(qr[quad].vregs[0], qr[quad].vregs[1], qr[quad].vregs[2], qr[quad].vregs[3])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
// WARN_LOG(JIT, "Q: Falling back to individual flush: pc=%08x", js_->compilerPC);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[0]), R1);
emit_->VST1_lane(F_32, q, R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[1]), R1);
emit_->VST1_lane(F_32, q, R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[2]), R1);
emit_->VST1_lane(F_32, q, R0, 2, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(qr[quad].vregs[3]), R1);
emit_->VST1_lane(F_32, q, R0, 3, true);
}
break;
default:
ERROR_LOG(JIT, "Unknown quad size %i", qr[quad].sz);
break;
}
qr[quad].isDirty = false;
int n = GetNumVectorElements(qr[quad].sz);
for (int i = 0; i < n; i++) {
int vr = qr[quad].vregs[i];
if (vr < 0 || vr > 128) {
ERROR_LOG(JIT, "Bad vr %i", vr);
}
FPURegMIPS &m = mr[32 + vr];
m.loc = ML_MEM;
m.lane = -1;
m.reg = -1;
}
} else {
if (qr[quad].isTemp) {
WARN_LOG(JIT, "Not flushing quad %i; dirty = %i, isTemp = %i", quad, qr[quad].isDirty, qr[quad].isTemp);
}
}
qr[quad].isTemp = false;
qr[quad].mipsVec = -1;
qr[quad].sz = V_Invalid;
memset(qr[quad].vregs, 0xFF, 4);
}
int ArmRegCacheFPU::QGetFreeQuad(int start, int count, const char *reason) {
// Search for a free quad. A quad is free if the first register in it is free.
int quad = -1;
for (int i = 0; i < count; i++) {
int q = (i + start) & 15;
if (!MappableQ(q))
continue;
// Don't steal temp quads!
if (qr[q].mipsVec == (int)INVALID_REG && !qr[q].isTemp) {
// INFO_LOG(JIT, "Free quad: %i", q);
// Oh yeah! Free quad!
return q;
}
}
// Okay, find the "best scoring" reg to replace. Scoring algorithm TBD but may include some
// sort of age.
int bestQuad = -1;
int bestScore = -1;
for (int i = 0; i < count; i++) {
int q = (i + start) & 15;
if (!MappableQ(q))
continue;
if (qr[q].spillLock)
continue;
if (qr[q].isTemp)
continue;
int score = 0;
if (!qr[q].isDirty) {
score += 5;
}
if (score > bestScore) {
bestQuad = q;
bestScore = score;
}
}
if (bestQuad == -1) {
ERROR_LOG(JIT, "Failed finding a free quad. Things will now go haywire!");
return -1;
} else {
INFO_LOG(JIT, "No register found in %i and the next %i, kicked out %i (%s)", start, count, bestQuad, reason ? reason : "no reason");
QFlush(bestQuad);
return bestQuad;
}
}
ARMReg ArmRegCacheFPU::QAllocTemp(VectorSize sz) {
int q = QGetFreeQuad(8, 16, "allocating temporary"); // Prefer high quads as temps
if (q < 0) {
ERROR_LOG(JIT, "Failed to allocate temp quad");
q = 0;
}
qr[q].spillLock = true;
qr[q].isTemp = true;
qr[q].sz = sz;
qr[q].isDirty = false; // doesn't matter
INFO_LOG(JIT, "Allocated temp quad %i", q);
if (sz == V_Single || sz == V_Pair) {
return D_0(ARMReg(Q0 + q));
} else {
return ARMReg(Q0 + q);
}
}
bool ArmRegCacheFPU::Consecutive(int v1, int v2) const {
return (voffset[v1] + 1) == voffset[v2];
}
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3) const {
return Consecutive(v1, v2) && Consecutive(v2, v3);
}
bool ArmRegCacheFPU::Consecutive(int v1, int v2, int v3, int v4) const {
return Consecutive(v1, v2) && Consecutive(v2, v3) && Consecutive(v3, v4);
}
void ArmRegCacheFPU::QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags) {
u8 vregs[4];
if (flags & MAP_MTX_TRANSPOSED) {
GetMatrixRows(matrix, mz, vregs);
} else {
GetMatrixColumns(matrix, mz, vregs);
}
// TODO: Zap existing mappings, reserve 4 consecutive regs, then do a fast load.
int n = GetMatrixSide(mz);
VectorSize vsz = GetVectorSize(mz);
for (int i = 0; i < n; i++) {
regs[i] = QMapReg(vregs[i], vsz, flags);
}
}
ARMReg ArmRegCacheFPU::QMapReg(int vreg, VectorSize sz, int flags) {
qTime_++;
int n = GetNumVectorElements(sz);
u8 vregs[4];
GetVectorRegs(vregs, sz, vreg);
// Range of registers to consider
int start = 0;
int count = 16;
if (flags & MAP_PREFER_HIGH) {
start = 8;
} else if (flags & MAP_PREFER_LOW) {
start = 4;
} else if (flags & MAP_FORCE_LOW) {
start = 4;
count = 4;
} else if (flags & MAP_FORCE_HIGH) {
start = 8;
count = 8;
}
// Let's check if they are all mapped in a quad somewhere.
// At the same time, check for the quad already being mapped.
// Later we can check for possible transposes as well.
// First just loop over all registers. If it's here and not in range, or overlapped, kick.
std::vector<int> quadsToFlush;
for (int i = 0; i < 16; i++) {
int q = (i + start) & 15;
if (!MappableQ(q))
continue;
// Skip unmapped quads.
if (qr[q].sz == V_Invalid)
continue;
// Check if completely there already. If so, set spill-lock, transfer dirty flag and exit.
if (vreg == qr[q].mipsVec && sz == qr[q].sz) {
if (i < count) {
INFO_LOG(JIT, "Quad already mapped: %i : %i (size %i)", q, vreg, sz);
qr[q].isDirty = qr[q].isDirty || (flags & MAP_DIRTY);
qr[q].spillLock = true;
// Sanity check vregs
for (int i = 0; i < n; i++) {
if (vregs[i] != qr[q].vregs[i]) {
ERROR_LOG(JIT, "Sanity check failed: %i vs %i", vregs[i], qr[q].vregs[i]);
}
}
return (ARMReg)(Q0 + q);
} else {
INFO_LOG(JIT, "Quad out of range %i (count = %i), needs moving. For now we flush.", start, count);
quadsToFlush.push_back(q);
continue;
}
}
// Check for any overlap. Overlap == flush.
int origN = GetNumVectorElements(qr[q].sz);
for (int a = 0; a < n; a++) {
for (int b = 0; b < origN; b++) {
if (vregs[a] == qr[q].vregs[b]) {
quadsToFlush.push_back(q);
goto doubleBreak;
}
}
}
doubleBreak:
;
}
// We didn't find the extra register, but we got a list of regs to flush. Flush 'em.
// Here we can check for opportunities to do a "transpose-flush" of row vectors, etc.
if (!quadsToFlush.empty()) {
ILOG("New mapping %s collided with %i quads, flushing them.", GetVectorNotation(vreg, sz), (int)quadsToFlush.size());
}
for (size_t i = 0; i < quadsToFlush.size(); i++) {
QFlush(quadsToFlush[i]);
}
// Find where we want to map it, obeying the constraints we gave.
int quad = QGetFreeQuad(start, count, "mapping");
// If parts of our register are elsewhere, and we are dirty, we need to flush them
// before we reload in a new location.
// This may be problematic if inputs overlap irregularly with output, say:
// vdot S700, R000, C000
// It might still work by accident...
if (flags & MAP_DIRTY) {
for (int i = 0; i < n; i++) {
FlushV(vregs[i]);
}
}
qr[quad].sz = sz;
qr[quad].mipsVec = vreg;
if (!(flags & MAP_NOINIT)) {
// Okay, now we will try to load the whole thing in one go. This is possible
// if it's a row and easy if it's a single.
// Rows are rare, columns are common - but thanks to our register reordering,
// columns are actually in-order in memory.
switch (sz) {
case V_Single:
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
break;
case V_Pair:
if (Consecutive(vregs[0], vregs[1])) {
// Can combine, it's a column!
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
}
break;
case V_Triple:
if (Consecutive(vregs[0], vregs[1], vregs[2])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1(F_32, QuadAsD(quad), R0, 1, ALIGN_NONE, REG_UPDATE); // TODO: Allow ALIGN_64 when applicable
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
} else {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
}
break;
case V_Quad:
if (Consecutive(vregs[0], vregs[1], vregs[2], vregs[3])) {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1(F_32, QuadAsD(quad), R0, 2, ALIGN_NONE); // TODO: Allow ALIGN_64 when applicable
} else {
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[0]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 0, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[1]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 1, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[2]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 2, true);
emit_->ADDI2R(R0, CTXREG, GetMipsRegOffsetV(vregs[3]), R1);
emit_->VLD1_lane(F_32, QuadAsQ(quad), R0, 3, true);
}
break;
default:
;
}
}
// OK, let's fill out the arrays to confirm that we have grabbed these registers.
for (int i = 0; i < n; i++) {
int mipsReg = 32 + vregs[i];
mr[mipsReg].loc = ML_ARMREG;
mr[mipsReg].reg = QuadAsQ(quad);
mr[mipsReg].lane = i;
qr[quad].vregs[i] = vregs[i];
}
qr[quad].isDirty = (flags & MAP_DIRTY) != 0;
qr[quad].spillLock = true;
INFO_LOG(JIT, "Mapped Q%i to vfpu %i (%s), sz=%i, dirty=%i", quad, vreg, GetVectorNotation(vreg, sz), (int)sz, qr[quad].isDirty);
if (sz == V_Single || sz == V_Pair) {
return D_0(QuadAsQ(quad));
} else {
return QuadAsQ(quad);
}
}
+77 -12
View File
@@ -33,28 +33,56 @@ enum {
TOTAL_MAPPABLE_MIPSFPUREGS = 32 + 128 + NUM_TEMPS,
};
enum {
MAP_READ = 0,
MAP_MTX_TRANSPOSED = 16,
MAP_PREFER_LOW = 16,
MAP_PREFER_HIGH = 32,
// Force is not yet correctly implemented, if the reg is already mapped it will not move
MAP_FORCE_LOW = 64, // Only map Q0-Q7 (and probably not Q0-Q3 as they are S registers so that leaves Q8-Q15)
MAP_FORCE_HIGH = 128, // Only map Q8-Q15
};
struct FPURegARM {
int mipsReg; // if -1, no mipsreg attached.
bool isDirty; // Should the register be written back?
};
struct FPURegQuad {
int mipsVec;
VectorSize sz;
u8 vregs[4];
bool isDirty;
bool spillLock;
bool isTemp;
};
struct FPURegMIPS {
// Where is this MIPS register?
RegMIPSLoc loc;
// Data (only one of these is used, depending on loc. Could make a union).
int reg;
int lane;
bool spillLock; // if true, this register cannot be spilled.
bool tempLock;
// If loc == ML_MEM, it's back in its location in the CPU context struct.
};
namespace MIPSComp {
struct ArmJitOptions;
struct JitState;
}
class ArmRegCacheFPU
{
public:
ArmRegCacheFPU(MIPSState *mips);
ArmRegCacheFPU(MIPSState *mips, MIPSComp::JitState *js, MIPSComp::ArmJitOptions *jo);
~ArmRegCacheFPU() {}
void Init(ARMXEmitter *emitter);
void Start(MIPSAnalyst::AnalysisResults &stats);
// Protect the arm register containing a MIPS register from spilling, to ensure that
@@ -80,9 +108,11 @@ public:
void MapDirty(MIPSReg rd);
void MapDirtyIn(MIPSReg rd, MIPSReg rs, bool avoidLoad = true);
void MapDirtyInIn(MIPSReg rd, MIPSReg rs, MIPSReg rt, bool avoidLoad = true);
bool IsMapped(MIPSReg r);
void FlushArmReg(ARMReg r);
void FlushR(MIPSReg r);
void DiscardR(MIPSReg r);
ARMReg R(int preg); // Returns a cached register
// VFPU register as single ARM VFP registers. Must not be used in the upcoming NEON mode!
void MapRegV(int vreg, int flags = 0);
@@ -90,18 +120,41 @@ public:
void MapInInV(int rt, int rs);
void MapDirtyInV(int rd, int rs, bool avoidLoad = true);
void MapDirtyInInV(int rd, int rs, int rt, bool avoidLoad = true);
bool IsTempX(ARMReg r) const;
MIPSReg GetTempR();
bool IsTempX(ARMReg r) const;
MIPSReg GetTempV() { return GetTempR() - 32; }
// VFPU registers as single VFP registers.
ARMReg V(int vreg) { return R(vreg + 32); }
int FlushGetSequential(int a, int maxArmReg);
void FlushAll();
ARMReg R(int preg); // Returns a cached register
// This one is allowed at any point.
void FlushV(MIPSReg r);
// VFPU registers mapped to match NEON quads (and doubles, for pairs and singles)
// Here we return the ARM register directly instead of providing a "V" accessor
// and so on. Might switch to this model for the other regallocs later.
// Quad mapping does NOT look into the ar array. Instead we use the qr array to keep
// track of what's in each quad.
// Note that we automatically spill-lock EVERY Q REGISTER we map, unlike other types.
// Need to explicitly allow spilling to get spilling.
ARMReg QMapReg(int vreg, VectorSize sz, int flags);
// TODO
// Maps a matrix as a set of columns (yes, even transposed ones, always columns
// as those are faster to load/flush). When possible it will map into consecutive
// quad registers, enabling blazing-fast full-matrix loads, transposed or not.
void QMapMatrix(ARMReg *regs, int matrix, MatrixSize mz, int flags);
ARMReg QAllocTemp(VectorSize sz);
// VFPU registers as single VFP registers
ARMReg V(int vreg) { return R(vreg + 32); }
void QAllowSpill(int quad);
void QFlush(int quad);
void QLoad4x4(MIPSGPReg regPtr, int vquads[4]);
//void FlushQWithV(MIPSReg r);
// NOTE: These require you to release spill locks manually!
void MapRegsAndSpillLockV(int vec, VectorSize vsz, int flags);
@@ -112,31 +165,43 @@ public:
void SetEmitter(ARMXEmitter *emitter) { emit_ = emitter; }
// For better log output only.
void SetCompilerPC(u32 compilerPC) { compilerPC_ = compilerPC; }
int GetMipsRegOffset(MIPSReg r);
private:
bool Consecutive(int v1, int v2) const;
bool Consecutive(int v1, int v2, int v3) const;
bool Consecutive(int v1, int v2, int v3, int v4) const;
MIPSReg GetTempR();
const ARMReg *GetMIPSAllocationOrder(int &count);
int GetMipsRegOffsetV(MIPSReg r) {
return GetMipsRegOffset(r + 32);
}
// This one WILL get a free quad as long as you haven't spill-locked them all.
int QGetFreeQuad(int start, int count, const char *reason);
int GetNumARMFPURegs();
private:
void SetupInitialRegs();
MIPSState *mips_;
ARMXEmitter *emit_;
u32 compilerPC_;
MIPSComp::JitState *js_;
MIPSComp::ArmJitOptions *jo_;
int numARMFpuReg_;
int qTime_;
enum {
MAX_ARMFPUREG = 32, // TODO: Support 32, which you have with NEON
// With NEON, we have 64 S = 32 D = 16 Q registers. Only the first 32 S registers
// are individually mappable though.
MAX_ARMFPUREG = 32,
MAX_ARMQUADS = 16,
NUM_MIPSFPUREG = TOTAL_MAPPABLE_MIPSFPUREGS,
};
FPURegARM ar[MAX_ARMFPUREG];
FPURegMIPS mr[NUM_MIPSFPUREG];
FPURegQuad qr[MAX_ARMQUADS];
FPURegMIPS *vr;
bool pendingFlush;
+2
View File
@@ -64,6 +64,7 @@ ARCH_FILES := \
$(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \
$(SRC)/Core/MIPS/ARM/ArmAsm.cpp \
$(SRC)/Core/MIPS/ARM/ArmJit.cpp \
@@ -84,6 +85,7 @@ ARCH_FILES := \
$(SRC)/Core/MIPS/ARM/ArmCompLoadStore.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPU.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEON.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompVFPUNEONUtil.cpp \
$(SRC)/Core/MIPS/ARM/ArmCompReplace.cpp \
$(SRC)/Core/MIPS/ARM/ArmAsm.cpp \
$(SRC)/Core/MIPS/ARM/ArmJit.cpp \