diff --git a/GPU/Common/VertexDecoderArm.cpp b/GPU/Common/VertexDecoderArm.cpp index 52cefc0d71..c2f6ae46fd 100644 --- a/GPU/Common/VertexDecoderArm.cpp +++ b/GPU/Common/VertexDecoderArm.cpp @@ -136,7 +136,6 @@ static const JitLookup jitLookup[] = { {&VertexDecoder::Step_PosS8Through, &VertexDecoderJitCache::Jit_PosS8Through}, {&VertexDecoder::Step_PosS16Through, &VertexDecoderJitCache::Jit_PosS16Through}, - {&VertexDecoder::Step_PosFloatThrough, &VertexDecoderJitCache::Jit_PosFloat}, {&VertexDecoder::Step_PosS8, &VertexDecoderJitCache::Jit_PosS8}, {&VertexDecoder::Step_PosS16, &VertexDecoderJitCache::Jit_PosS16}, @@ -882,7 +881,7 @@ void VertexDecoderJitCache::Jit_PosS8Through() { // TODO: SIMD LDRSB(tempReg1, srcReg, dec_->posoff); LDRSB(tempReg2, srcReg, dec_->posoff + 1); - LDRSB(tempReg3, srcReg, dec_->posoff + 2); // signed? + LDRB(tempReg3, srcReg, dec_->posoff + 2); static const ARMReg tr[3] = { tempReg1, tempReg2, tempReg3 }; static const ARMReg fr[3] = { fpScratchReg, fpScratchReg2, fpScratchReg3 }; ADD(scratchReg, dstReg, dec_->decFmt.posoff); diff --git a/GPU/Common/VertexDecoderArm64.cpp b/GPU/Common/VertexDecoderArm64.cpp index 305ba3ed90..32cc5bc291 100644 --- a/GPU/Common/VertexDecoderArm64.cpp +++ b/GPU/Common/VertexDecoderArm64.cpp @@ -113,7 +113,7 @@ static const JitLookup jitLookup[] = { {&VertexDecoder::Step_PosS8Through, &VertexDecoderJitCache::Jit_PosS8Through}, {&VertexDecoder::Step_PosS16Through, &VertexDecoderJitCache::Jit_PosS16Through}, - {&VertexDecoder::Step_PosFloatThrough, &VertexDecoderJitCache::Jit_PosFloat}, + {&VertexDecoder::Step_PosFloatThrough, &VertexDecoderJitCache::Jit_PosFloatThrough}, {&VertexDecoder::Step_PosS8, &VertexDecoderJitCache::Jit_PosS8}, {&VertexDecoder::Step_PosS16, &VertexDecoderJitCache::Jit_PosS16}, @@ -670,7 +670,7 @@ void VertexDecoderJitCache::Jit_PosFloat() { void VertexDecoderJitCache::Jit_PosS8Through() { LDRSB(INDEX_UNSIGNED, tempReg1, srcReg, dec_->posoff); LDRSB(INDEX_UNSIGNED, tempReg2, srcReg, dec_->posoff + 1); - LDRSB(INDEX_UNSIGNED, tempReg3, srcReg, dec_->posoff + 2); // signed? + LDRB(INDEX_UNSIGNED, tempReg3, srcReg, dec_->posoff + 2); fp.SCVTF(fpScratchReg, tempReg1); fp.SCVTF(fpScratchReg2, tempReg2); fp.SCVTF(fpScratchReg3, tempReg3); @@ -691,6 +691,25 @@ void VertexDecoderJitCache::Jit_PosS16Through() { STR(INDEX_UNSIGNED, src[1], dstReg, dec_->decFmt.posoff + 8); } +void VertexDecoderJitCache::Jit_PosFloatThrough() { + // Instead of just copying 12 bytes, we copy 8 and clamp Z. + if ((dec_->posoff & 7) == 0 && (dec_->decFmt.posoff & 7) == 0) { + LDR(INDEX_UNSIGNED, EncodeRegTo64(tempReg1), srcReg, dec_->posoff); + STR(INDEX_UNSIGNED, EncodeRegTo64(tempReg1), dstReg, dec_->decFmt.posoff); + } else { + LDP(INDEX_SIGNED, tempReg1, tempReg2, srcReg, dec_->posoff); + STP(INDEX_SIGNED, tempReg1, tempReg2, dstReg, dec_->decFmt.posoff); + } + + fp.LDUR(32, neonScratchRegD, srcReg, dec_->posoff + 8); + fp.FCVTZU(32, neonScratchRegD, neonScratchRegD); + // Narrow to 16 bit, saturating meanwhile. + fp.UQXTN(16, neonScratchRegD, neonScratchRegD); + fp.UXTL(16, neonScratchRegD, neonScratchRegD); + fp.UCVTF(32, neonScratchRegD, neonScratchRegD); + fp.STUR(32, neonScratchRegD, dstReg, dec_->decFmt.posoff + 8); +} + void VertexDecoderJitCache::Jit_NormalS8() { LDURH(tempReg1, srcReg, dec_->nrmoff); LDRB(INDEX_UNSIGNED, tempReg3, srcReg, dec_->nrmoff + 2); diff --git a/GPU/Common/VertexDecoderCommon.cpp b/GPU/Common/VertexDecoderCommon.cpp index c6a4631636..64ef34aa80 100644 --- a/GPU/Common/VertexDecoderCommon.cpp +++ b/GPU/Common/VertexDecoderCommon.cpp @@ -776,10 +776,11 @@ void VertexDecoder::Step_PosFloatSkin() const void VertexDecoder::Step_PosS8Through() const { float *v = (float *)(decoded_ + decFmt.posoff); - const s8 *sv = (const s8*)(ptr_ + posoff); + const s8 *sv = (const s8 *)(ptr_ + posoff); + const u8 *uv = (const u8 *)(ptr_ + posoff); v[0] = sv[0]; v[1] = sv[1]; - v[2] = sv[2]; + v[2] = uv[2]; } void VertexDecoder::Step_PosS16Through() const @@ -794,9 +795,10 @@ void VertexDecoder::Step_PosS16Through() const void VertexDecoder::Step_PosFloatThrough() const { - u8 *v = (u8 *)(decoded_ + decFmt.posoff); - const u8 *fv = (const u8 *)(ptr_ + posoff); - memcpy(v, fv, 12); + float *v = (float *)(decoded_ + decFmt.posoff); + const float *fv = (const float *)(ptr_ + posoff); + memcpy(v, fv, 8); + v[2] = fv[2] > 65535.0f ? 65535.0f : (fv[2] < 0.0f ? 0.0f : fv[2]); } void VertexDecoder::Step_PosS8Morph() const diff --git a/GPU/Common/VertexDecoderCommon.h b/GPU/Common/VertexDecoderCommon.h index b80c18c2dc..257e7751ba 100644 --- a/GPU/Common/VertexDecoderCommon.h +++ b/GPU/Common/VertexDecoderCommon.h @@ -131,8 +131,7 @@ public: pos[2] = f[2]; } else { // Integer value passed in a float. Clamped to 0, 65535. - const float z = (int)f[2] * (1.0f / 65535.0f); - pos[2] = z > 1.0f ? 1.0f : (z < 0.0f ? 0.0f : z); + pos[2] = (int)f[2] * (1.0f / 65535.0f); } } break; @@ -179,11 +178,6 @@ public: { const float *f = (const float *)(data_ + decFmt_.posoff); memcpy(pos, f, 12); - if (isThrough()) { - // Integer value passed in a float. Clamped to 0, 65535. - const float z = (int)pos[2]; - pos[2] = z > 65535.0f ? 65535.0f : (z < 0.0f ? 0.0f : z); - } } break; case DEC_S16_3: @@ -666,6 +660,7 @@ public: void Jit_PosFloat(); void Jit_PosS8Through(); void Jit_PosS16Through(); + void Jit_PosFloatThrough(); void Jit_PosS8Skin(); void Jit_PosS16Skin(); diff --git a/GPU/Common/VertexDecoderX86.cpp b/GPU/Common/VertexDecoderX86.cpp index d16c3fd047..611bf56c21 100644 --- a/GPU/Common/VertexDecoderX86.cpp +++ b/GPU/Common/VertexDecoderX86.cpp @@ -138,7 +138,7 @@ static const JitLookup jitLookup[] = { {&VertexDecoder::Step_PosS8Through, &VertexDecoderJitCache::Jit_PosS8Through}, {&VertexDecoder::Step_PosS16Through, &VertexDecoderJitCache::Jit_PosS16Through}, - {&VertexDecoder::Step_PosFloatThrough, &VertexDecoderJitCache::Jit_PosFloat}, + {&VertexDecoder::Step_PosFloatThrough, &VertexDecoderJitCache::Jit_PosFloatThrough}, {&VertexDecoder::Step_PosS8, &VertexDecoderJitCache::Jit_PosS8}, {&VertexDecoder::Step_PosS16, &VertexDecoderJitCache::Jit_PosS16}, @@ -1348,7 +1348,10 @@ void VertexDecoderJitCache::Jit_PosS8Through() { DEBUG_LOG_REPORT_ONCE(vertexS8Through, G3D, "Using S8 positions in throughmode"); // SIMD doesn't really matter since this isn't useful on hardware. for (int i = 0; i < 3; i++) { - MOVSX(32, 8, tempReg1, MDisp(srcReg, dec_->posoff + i)); + if (i == 2) + MOVZX(32, 8, tempReg1, MDisp(srcReg, dec_->posoff + i)); + else + MOVSX(32, 8, tempReg1, MDisp(srcReg, dec_->posoff + i)); CVTSI2SS(fpScratchReg, R(tempReg1)); MOVSS(MDisp(dstReg, dec_->decFmt.posoff + i * 4), fpScratchReg); } @@ -1377,6 +1380,29 @@ void VertexDecoderJitCache::Jit_PosS16Through() { } } +void VertexDecoderJitCache::Jit_PosFloatThrough() { + PXOR(fpScratchReg2, R(fpScratchReg2)); + if (cpu_info.Mode64bit) { + MOV(64, R(tempReg1), MDisp(srcReg, dec_->posoff)); + MOVSS(fpScratchReg, MDisp(srcReg, dec_->posoff + 8)); + MOV(64, MDisp(dstReg, dec_->decFmt.posoff), R(tempReg1)); + } else { + MOV(32, R(tempReg1), MDisp(srcReg, dec_->posoff)); + MOV(32, R(tempReg2), MDisp(srcReg, dec_->posoff + 4)); + MOVSS(fpScratchReg, MDisp(srcReg, dec_->posoff + 8)); + MOV(32, MDisp(dstReg, dec_->decFmt.posoff), R(tempReg1)); + MOV(32, MDisp(dstReg, dec_->decFmt.posoff + 4), R(tempReg2)); + } + + CVTTPS2DQ(fpScratchReg, R(fpScratchReg)); + // Use pack to saturate to 0,65535. + PACKUSDW(fpScratchReg, R(fpScratchReg)); + PUNPCKLWD(fpScratchReg, R(fpScratchReg2)); + CVTDQ2PS(fpScratchReg, R(fpScratchReg)); + + MOVSS(MDisp(dstReg, dec_->decFmt.posoff + 8), fpScratchReg); +} + void VertexDecoderJitCache::Jit_PosS8() { Jit_AnyS8ToFloat(dec_->posoff); MOVUPS(MDisp(dstReg, dec_->decFmt.posoff), XMM3);