diff --git a/GPU/GPUState.h b/GPU/GPUState.h index e91be6d1ab..c6fb03c4d3 100644 --- a/GPU/GPUState.h +++ b/GPU/GPUState.h @@ -298,9 +298,7 @@ struct GPUgstate { bool isTextureAlphaUsed() const { return (texfunc & 0x100) != 0; } GETextureFormat getTextureFormat() const { return static_cast(texformat & 0xF); } bool isTextureFormatIndexed() const { return (texformat & 4) != 0; } // GE_TFMT_CLUT4 - GE_TFMT_CLUT32 are 0b1xx. - int getTextureEnvColR() const { return texenvcolor&0xFF; } - int getTextureEnvColG() const { return (texenvcolor>>8)&0xFF; } - int getTextureEnvColB() const { return (texenvcolor>>16)&0xFF; } + int getTextureEnvColRGB() const { return texenvcolor & 0x00FFFFFF; } u32 getClutAddress() const { return (clutaddr & 0x00FFFFF0) | ((clutaddrupper << 8) & 0x0F000000); } int getClutLoadBytes() const { return (loadclut & 0x3F) * 32; } int getClutLoadBlocks() const { return (loadclut & 0x3F); } diff --git a/GPU/Software/BinManager.cpp b/GPU/Software/BinManager.cpp index 993bf8712e..19339cdd77 100644 --- a/GPU/Software/BinManager.cpp +++ b/GPU/Software/BinManager.cpp @@ -136,10 +136,10 @@ void BinManager::UpdateState() { stateIndex_ = (int)states_.Push(RasterizerState()); ComputeRasterizerState(&states_[stateIndex_]); - DrawingCoords scissorTL(gstate.getScissorX1(), gstate.getScissorY1(), 0); - DrawingCoords scissorBR(gstate.getScissorX2(), gstate.getScissorY2(), 0); - ScreenCoords screenScissorTL = TransformUnit::DrawingToScreen(scissorTL); - ScreenCoords screenScissorBR = TransformUnit::DrawingToScreen(scissorBR); + DrawingCoords scissorTL(gstate.getScissorX1(), gstate.getScissorY1()); + DrawingCoords scissorBR(gstate.getScissorX2(), gstate.getScissorY2()); + ScreenCoords screenScissorTL = TransformUnit::DrawingToScreen(scissorTL, 0); + ScreenCoords screenScissorBR = TransformUnit::DrawingToScreen(scissorBR, 0); scissor_.x1 = screenScissorTL.x; scissor_.y1 = screenScissorTL.y; @@ -237,8 +237,8 @@ void BinManager::Drain() { int h2 = (queueRange_.y2 - queueRange_.y1 + 31) / 32; // Always bin the entire possible range, but focus on the drawn area. - ScreenCoords tl = TransformUnit::DrawingToScreen(DrawingCoords(0, 0, 0)); - ScreenCoords br = TransformUnit::DrawingToScreen(DrawingCoords(1024, 1024, 0)); + ScreenCoords tl = TransformUnit::DrawingToScreen(DrawingCoords(0, 0), 0); + ScreenCoords br = TransformUnit::DrawingToScreen(DrawingCoords(1024, 1024), 0); taskRanges_.clear(); if (h2 >= 18 && w2 >= h2 * 4) { diff --git a/GPU/Software/DrawPixel.cpp b/GPU/Software/DrawPixel.cpp index f345059f70..54eb519bdc 100644 --- a/GPU/Software/DrawPixel.cpp +++ b/GPU/Software/DrawPixel.cpp @@ -460,7 +460,7 @@ void SOFTRAST_CALL DrawSinglePixel(int x, int y, int z, int fog, Vec4IntArg colo const Vec4 dst = Vec4::FromRGBA(old_color); Vec3 blended = AlphaBlendingResult(pixelID, prim_color, dst); if (pixelID.dithering) { - blended += Vec3::AssignToAll(pixelID.cached.ditherMatrix[y * 4 + x]); + blended += Vec3::AssignToAll(pixelID.cached.ditherMatrix[(y & 3) * 4 + (x & 3)]); } // ToRGB() always automatically clamps. @@ -469,7 +469,7 @@ void SOFTRAST_CALL DrawSinglePixel(int x, int y, int z, int fog, Vec4IntArg colo } else { if (pixelID.dithering) { // We'll discard alpha anyway. - prim_color += Vec4::AssignToAll(pixelID.cached.ditherMatrix[y * 4 + x]); + prim_color += Vec4::AssignToAll(pixelID.cached.ditherMatrix[(y & 3) * 4 + (x & 3)]); } #if defined(_M_SSE) diff --git a/GPU/Software/FuncId.cpp b/GPU/Software/FuncId.cpp index 5d0898265a..c6c0d37c5d 100644 --- a/GPU/Software/FuncId.cpp +++ b/GPU/Software/FuncId.cpp @@ -22,7 +22,7 @@ #include "GPU/GPUState.h" #include "GPU/Software/FuncId.h" -static_assert(sizeof(SamplerID) == sizeof(SamplerID::fullKey), "Bad sampler ID size"); +static_assert(sizeof(SamplerID) == sizeof(SamplerID::fullKey) + sizeof(SamplerID::cached), "Bad sampler ID size"); static_assert(sizeof(PixelFuncID) == sizeof(PixelFuncID::fullKey) + sizeof(PixelFuncID::cached), "Bad pixel func ID size"); static inline GEComparison OptimizeRefByteCompare(GEComparison func, u8 ref) { @@ -199,6 +199,8 @@ void ComputePixelFuncID(PixelFuncID *id) { if (id->applyFog) { id->cached.fogColor = gstate.fogcolor & 0x00FFFFFF; } + if (id->applyLogicOp) + id->cached.logicOp = gstate.getLogicOp(); id->cached.minz = gstate.getDepthRangeMin(); id->cached.maxz = gstate.getDepthRangeMax(); id->cached.framebufStride = gstate.FrameBufStride(); @@ -399,6 +401,9 @@ void ComputeSamplerID(SamplerID *id_out) { int bytes = h * (bufw * bitspp) / 8; if (bitspp < 32 && !Memory::IsValidAddress(addr + bytes + (32 - bitspp) / 8)) id.overReadSafe = false; + + id.cached.sizes[i].w = w; + id.cached.sizes[i].h = h; } id.hasAnyMips = maxLevel != 0; @@ -411,6 +416,7 @@ void ComputeSamplerID(SamplerID *id_out) { id.hasClutMask = gstate.getClutIndexMask() != 0xFF; id.hasClutShift = gstate.getClutIndexShift() != 0; id.hasClutOffset = gstate.getClutIndexStartPos() != 0; + id.cached.clutFormat = gstate.clutformat; } id.clampS = gstate.isTexCoordClampedS(); @@ -424,6 +430,9 @@ void ComputeSamplerID(SamplerID *id_out) { if (id.texFunc > GE_TEXFUNC_ADD) id.texFunc = GE_TEXFUNC_ADD; + if (id.texFunc == GE_TEXFUNC_BLEND) + id.cached.texBlendColor = gstate.getTextureEnvColRGB(); + *id_out = id; } diff --git a/GPU/Software/FuncId.h b/GPU/Software/FuncId.h index ab8f749928..22a531fe7b 100644 --- a/GPU/Software/FuncId.h +++ b/GPU/Software/FuncId.h @@ -166,6 +166,15 @@ struct SamplerID { SamplerID() : fullKey(0) { } + struct { + struct { + uint16_t w; + uint16_t h; + } sizes[8]; + uint32_t texBlendColor; + uint32_t clutFormat; + } cached; + union { uint32_t fullKey; struct { diff --git a/GPU/Software/Rasterizer.cpp b/GPU/Software/Rasterizer.cpp index 55b5fb569e..46fc4c93c3 100644 --- a/GPU/Software/Rasterizer.cpp +++ b/GPU/Software/Rasterizer.cpp @@ -127,6 +127,8 @@ void ComputeRasterizerState(RasterizerState *state) { else state->texptr[i] = nullptr; } + + state->textureLodSlope = gstate.getTextureLodSlope(); } state->texLevelMode = gstate.getTexLevelMode(); @@ -138,6 +140,9 @@ void ComputeRasterizerState(RasterizerState *state) { state->magFilt = gstate.isMagnifyFilteringEnabled(); state->antialiasLines = gstate.isAntiAliasEnabled(); + state->screenOffsetX = gstate.getOffsetX16(); + state->screenOffsetY = gstate.getOffsetY16(); + #if defined(SOFTGPU_MEMORY_TAGGING_DETAILED) || defined(SOFTGPU_MEMORY_TAGGING_BASIC) DisplayList currentList{}; if (gpuDebug) @@ -224,114 +229,6 @@ static inline bool IsRightSideOrFlatBottomLine(const Vec2& vertex, const Ve } } -Vec4IntResult SOFTRAST_CALL GetTextureFunctionOutput(Vec4IntArg prim_color_in, Vec4IntArg texcolor_in) { - const Vec4 prim_color = prim_color_in; - const Vec4 texcolor = texcolor_in; - - Vec3 out_rgb; - int out_a; - - bool rgba = gstate.isTextureAlphaUsed(); - - switch (gstate.getTextureFunction()) { - case GE_TEXFUNC_MODULATE: - { -#if defined(_M_SSE) - // Modulate weights slightly on the tex color, by adding one to prim and dividing by 256. - const __m128i p = _mm_slli_epi16(_mm_packs_epi32(prim_color.ivec, prim_color.ivec), 4); - const __m128i pboost = _mm_add_epi16(p, _mm_set1_epi16(1 << 4)); - __m128i t = _mm_slli_epi16(_mm_packs_epi32(texcolor.ivec, texcolor.ivec), 4); - if (gstate.isColorDoublingEnabled()) { - const __m128i amask = _mm_set_epi16(-1, 0, 0, 0, -1, 0, 0, 0); - const __m128i a = _mm_and_si128(t, amask); - const __m128i rgb = _mm_andnot_si128(amask, t); - t = _mm_or_si128(_mm_slli_epi16(rgb, 1), a); - } - const __m128i b = _mm_mulhi_epi16(pboost, t); - out_rgb.ivec = _mm_unpacklo_epi16(b, _mm_setzero_si128()); - - if (rgba) { - return ToVec4IntResult(Vec4(out_rgb.ivec)); - } else { - out_a = prim_color.a(); - } -#else - if (gstate.isColorDoublingEnabled()) { - out_rgb = ((prim_color.rgb() + Vec3::AssignToAll(1)) * texcolor.rgb() * 2) / 256; - } else { - out_rgb = (prim_color.rgb() + Vec3::AssignToAll(1)) * texcolor.rgb() / 256; - } - out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a(); -#endif - break; - } - - case GE_TEXFUNC_DECAL: - { - if (rgba) { - int t = texcolor.a(); - int invt = 255 - t; - // Both colors are boosted here, making the alpha have more weight. - Vec3 one = Vec3::AssignToAll(1); - out_rgb = ((prim_color.rgb() + one) * invt + (texcolor.rgb() + one) * t); - // Keep the bits of accuracy when doubling. - if (gstate.isColorDoublingEnabled()) - out_rgb /= 128; - else - out_rgb /= 256; - } else { - if (gstate.isColorDoublingEnabled()) - out_rgb = texcolor.rgb() * 2; - else - out_rgb = texcolor.rgb(); - } - out_a = prim_color.a(); - break; - } - - case GE_TEXFUNC_BLEND: - { - const Vec3 const255(255, 255, 255); - const Vec3 texenv(gstate.getTextureEnvColR(), gstate.getTextureEnvColG(), gstate.getTextureEnvColB()); - - // Unlike the others (and even alpha), this one simply always rounds up. - const Vec3 roundup = Vec3::AssignToAll(255); - out_rgb = ((const255 - texcolor.rgb()) * prim_color.rgb() + texcolor.rgb() * texenv + roundup); - // Must divide by less to keep the precision for doubling to be accurate. - if (gstate.isColorDoublingEnabled()) - out_rgb /= 128; - else - out_rgb /= 256; - - out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a(); - break; - } - - case GE_TEXFUNC_REPLACE: - out_rgb = texcolor.rgb(); - // Doubling even happens for replace. - if (gstate.isColorDoublingEnabled()) - out_rgb *= 2; - out_a = (rgba) ? texcolor.a() : prim_color.a(); - break; - - case GE_TEXFUNC_ADD: - case GE_TEXFUNC_UNKNOWN1: - case GE_TEXFUNC_UNKNOWN2: - case GE_TEXFUNC_UNKNOWN3: - // Don't need to clamp afterward, we always clamp before tests. - out_rgb = prim_color.rgb() + texcolor.rgb(); - if (gstate.isColorDoublingEnabled()) - out_rgb *= 2; - - // Alpha is still blended the common way. - out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a(); - break; - } - - return ToVec4IntResult(Vec4(out_rgb, out_a)); -} - static inline Vec3 GetSourceFactor(GEBlendSrcFactor factor, const Vec4 &source, const Vec4 &dst, uint32_t fix) { switch (factor) { case GE_SRCBLEND_DSTCOLOR: @@ -525,9 +422,9 @@ static inline Vec4IntResult SOFTRAST_CALL ApplyTexturing(float s, float t, int x const int *bufw0 = &state.texbufw[texlevel]; if (!bilinear) { - return state.nearest(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel); + return state.nearest(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel, state.samplerID); } - return state.linear(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel); + return state.linear(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel, state.samplerID); } static inline Vec4IntResult SOFTRAST_CALL ApplyTexturingSingle(float s, float t, int x, int y, Vec4IntArg prim_color, int texlevel, int frac_texlevel, bool bilinear, const RasterizerState &state) { @@ -561,7 +458,7 @@ static inline void CalculateSamplingParams(const float ds, const float dt, const break; case GE_TEXLEVEL_MODE_SLOPE: // This is always offset by an extra texlevel. - detail = 1 * 16 + TexLog2(gstate.getTextureLodSlope()); + detail = 1 * 16 + TexLog2(state.textureLodSlope); break; case GE_TEXLEVEL_MODE_CONST: default: @@ -823,7 +720,7 @@ void DrawTriangleSlice( Vec4 w1 = w1_base; Vec4 w2 = w2_base; - DrawingCoords p = TransformUnit::ScreenToDrawing(ScreenCoords(minX, curY, 0)); + DrawingCoords p = TransformUnit::ScreenToDrawing(minX, curY, state.screenOffsetX, state.screenOffsetY); int64_t rowMinX = minX, rowMaxX = maxX; e0.NarrowMinMaxX(w0, minX, rowMinX, rowMaxX); @@ -950,8 +847,8 @@ void DrawTriangleSlice( #if !defined(SOFTGPU_MEMORY_TAGGING_DETAILED) && defined(SOFTGPU_MEMORY_TAGGING_BASIC) for (int y = minY; y <= maxY; y += 16) { - DrawingCoords p = TransformUnit::ScreenToDrawing(ScreenCoords(minX, y, 0)); - DrawingCoords pend = TransformUnit::ScreenToDrawing(ScreenCoords(maxX, y, 0)); + DrawingCoords p = TransformUnit::ScreenToDrawing(minX, y, state.screenOffsetX, state.screenOffsetY); + DrawingCoords pend = TransformUnit::ScreenToDrawing(maxX, y, state.screenOffsetX, state.screenOffsetY); uint32_t row = gstate.getFrameBufAddress() + p.y * pixelID.cached.framebufStride * bpp; NotifyMemInfo(MemBlockFlags::WRITE, row + p.x * bpp, (pend.x - p.x) * bpp, tag.c_str(), tag.size()); @@ -1004,7 +901,7 @@ void DrawPoint(const VertexData &v0, const BinCoords &range, const RasterizerSta if (!pixelID.clearMode) prim_color += Vec4(sec_color, 0); - DrawingCoords p = TransformUnit::ScreenToDrawing(pos); + DrawingCoords p = TransformUnit::ScreenToDrawing(pos, state.screenOffsetX, state.screenOffsetY); u16 z = pos.z; u8 fog = 255; @@ -1031,8 +928,8 @@ void DrawPoint(const VertexData &v0, const BinCoords &range, const RasterizerSta } void ClearRectangle(const VertexData &v0, const VertexData &v1, const BinCoords &range, const RasterizerState &state) { - DrawingCoords pprime = TransformUnit::ScreenToDrawing(ScreenCoords(range.x1, range.y1, 0)); - DrawingCoords pend = TransformUnit::ScreenToDrawing(ScreenCoords(range.x2, range.y2, 0)); + DrawingCoords pprime = TransformUnit::ScreenToDrawing(range.x1, range.y1, state.screenOffsetX, state.screenOffsetY); + DrawingCoords pend = TransformUnit::ScreenToDrawing(range.x2, range.y2, state.screenOffsetX, state.screenOffsetY); auto &pixelID = state.pixelID; auto &samplerID = state.samplerID; @@ -1296,8 +1193,8 @@ void DrawLine(const VertexData &v0, const VertexData &v1, const BinCoords &range CalculateSamplingParams(ds, dt, state, texLevel, texLevelFrac, texBilinear); if (state.antialiasLines) { - // TODO: This is a niave and wrong implementation. - DrawingCoords p0 = TransformUnit::ScreenToDrawing(ScreenCoords((int)x, (int)y, (int)z)); + // TODO: This is a naive and wrong implementation. + DrawingCoords p0 = TransformUnit::ScreenToDrawing(x, y, state.screenOffsetX, state.screenOffsetY); s = ((float)p0.x + xinc / 32.0f) / 512.0f; t = ((float)p0.y + yinc / 32.0f) / 512.0f; @@ -1311,10 +1208,8 @@ void DrawLine(const VertexData &v0, const VertexData &v1, const BinCoords &range if (!pixelID.clearMode) prim_color += Vec4(sec_color, 0); - ScreenCoords pprime = ScreenCoords((int)x, (int)y, (int)z); - PROFILE_THIS_SCOPE("draw_px"); - DrawingCoords p = TransformUnit::ScreenToDrawing(pprime); + DrawingCoords p = TransformUnit::ScreenToDrawing(x, y, state.screenOffsetX, state.screenOffsetY); state.drawPixel(p.x, p.y, z, fog, ToVec4IntArg(prim_color), pixelID); #if defined(SOFTGPU_MEMORY_TAGGING_DETAILED) || defined(SOFTGPU_MEMORY_TAGGING_BASIC) @@ -1376,7 +1271,7 @@ bool GetCurrentTexture(GPUDebugBuffer &buffer, int level) u32 *row = (u32 *)buffer.GetData(); for (int y = 0; y < h; ++y) { for (int x = 0; x < w; ++x) { - row[x] = Vec4(sampler(x, y, texptr, texbufw, level)).ToRGBA(); + row[x] = Vec4(sampler(x, y, texptr, texbufw, level, id)).ToRGBA(); } row += w; } diff --git a/GPU/Software/Rasterizer.h b/GPU/Software/Rasterizer.h index fb92d42cfa..537d35ee5a 100644 --- a/GPU/Software/Rasterizer.h +++ b/GPU/Software/Rasterizer.h @@ -41,6 +41,9 @@ struct RasterizerState { Sampler::NearestFunc nearest; int texbufw[8]{}; u8 *texptr[8]{}; + float textureLodSlope; + int screenOffsetX; + int screenOffsetY; struct { uint8_t maxTexLevel : 3; @@ -77,6 +80,5 @@ bool GetCurrentTexture(GPUDebugBuffer &buffer, int level); // Shared functions with RasterizerRectangle.cpp Vec3 AlphaBlendingResult(const PixelFuncID &pixelID, const Vec4 &source, const Vec4 &dst); -Vec4IntResult SOFTRAST_CALL GetTextureFunctionOutput(Vec4IntArg prim_color, Vec4IntArg texcolor); } // namespace Rasterizer diff --git a/GPU/Software/RasterizerRectangle.cpp b/GPU/Software/RasterizerRectangle.cpp index 4be56b13ff..363ca6a3fb 100644 --- a/GPU/Software/RasterizerRectangle.cpp +++ b/GPU/Software/RasterizerRectangle.cpp @@ -87,14 +87,14 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran auto &pixelID = state.pixelID; auto &samplerID = state.samplerID; - DrawingCoords pos0 = TransformUnit::ScreenToDrawing(v0.screenpos); + DrawingCoords pos0 = TransformUnit::ScreenToDrawing(v0.screenpos, state.screenOffsetX, state.screenOffsetY); // Include the ending pixel based on its center, not start. - DrawingCoords pos1 = TransformUnit::ScreenToDrawing(v1.screenpos + ScreenCoords(7, 7, 0)); + DrawingCoords pos1 = TransformUnit::ScreenToDrawing(v1.screenpos + ScreenCoords(7, 7, 0), state.screenOffsetX, state.screenOffsetY); - DrawingCoords scissorTL = TransformUnit::ScreenToDrawing(ScreenCoords(range.x1, range.y1, 0)); - DrawingCoords scissorBR = TransformUnit::ScreenToDrawing(ScreenCoords(range.x2, range.y2, 0)); + DrawingCoords scissorTL = TransformUnit::ScreenToDrawing(range.x1, range.y1, state.screenOffsetX, state.screenOffsetY); + DrawingCoords scissorBR = TransformUnit::ScreenToDrawing(range.x2, range.y2, state.screenOffsetX, state.screenOffsetY); - int z = pos0.z; + int z = v1.screenpos.z; int fog = 255; bool isWhite = v1.color0 == Vec4(255, 255, 255, 255); @@ -146,7 +146,7 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran int s = s_start; u16 *pixel = fb.Get16Ptr(pos0.x, y, pixelID.cached.framebufStride); for (int x = pos0.x; x < pos1.x; x++) { - u32 tex_color = Vec4(fetchFunc(s, t, texptr, texbufw, 0)).ToRGBA(); + u32 tex_color = Vec4(fetchFunc(s, t, texptr, texbufw, 0, state.samplerID)).ToRGBA(); if (tex_color & 0xFF000000) { DrawSinglePixel5551(pixel, tex_color, pixelID); } @@ -162,7 +162,7 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran u16 *pixel = fb.Get16Ptr(pos0.x, y, pixelID.cached.framebufStride); for (int x = pos0.x; x < pos1.x; x++) { Vec4 prim_color = v1.color0; - Vec4 tex_color = fetchFunc(s, t, texptr, texbufw, 0); + Vec4 tex_color = fetchFunc(s, t, texptr, texbufw, 0, state.samplerID); prim_color = Vec4(ModulateRGBA(ToVec4IntArg(prim_color), ToVec4IntArg(tex_color), state.samplerID)); if (prim_color.a() > 0) { DrawSinglePixel5551(pixel, prim_color.ToRGBA(), pixelID); @@ -187,7 +187,7 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran float s = sf_start; // Not really that fast but faster than triangle. for (int x = pos0.x; x < pos1.x; x++) { - Vec4 prim_color = state.nearest(s, t, xoff, yoff, ToVec4IntArg(v1.color0), &texptr, &texbufw, 0, 0); + Vec4 prim_color = state.nearest(s, t, xoff, yoff, ToVec4IntArg(v1.color0), &texptr, &texbufw, 0, 0, state.samplerID); state.drawPixel(x, y, z, 255, ToVec4IntArg(prim_color), pixelID); s += dsf; } diff --git a/GPU/Software/RasterizerRegCache.h b/GPU/Software/RasterizerRegCache.h index 067f8f16c6..81e4a91a85 100644 --- a/GPU/Software/RasterizerRegCache.h +++ b/GPU/Software/RasterizerRegCache.h @@ -106,14 +106,13 @@ struct RegCache { VEC_INDEX = 0x0005, GEN_SRC_ALPHA = 0x0100, - GEN_GSTATE = 0x0101, - GEN_ID = 0x0102, - GEN_CONST_BASE = 0x0103, - GEN_STENCIL = 0x0104, - GEN_COLOR_OFF = 0x0105, - GEN_DEPTH_OFF = 0x0106, - GEN_RESULT = 0x0107, - GEN_SHIFTVAL = 0x0108, + GEN_ID = 0x0101, + GEN_CONST_BASE = 0x0102, + GEN_STENCIL = 0x0103, + GEN_COLOR_OFF = 0x0104, + GEN_DEPTH_OFF = 0x0105, + GEN_RESULT = 0x0106, + GEN_SHIFTVAL = 0x0107, GEN_ARG_X = 0x0180, GEN_ARG_Y = 0x0181, diff --git a/GPU/Software/Sampler.cpp b/GPU/Software/Sampler.cpp index c9214001f9..0ee39e61b8 100644 --- a/GPU/Software/Sampler.cpp +++ b/GPU/Software/Sampler.cpp @@ -39,9 +39,9 @@ extern u32 clut[4096]; namespace Sampler { -static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac); -static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac); -static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level); +static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID); +static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID); +static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level, const SamplerID &samplerID); std::mutex jitCacheLock; SamplerJitCache *jitCache = nullptr; @@ -242,10 +242,9 @@ FetchFunc SamplerJitCache::GetFetch(const SamplerID &id) { return nullptr; } -template -static inline int GetPixelDataOffset(unsigned int row_pitch_pixels, unsigned int u, unsigned int v) -{ - if (!gstate.isTextureSwizzled()) +template +static inline int GetPixelDataOffset(uint32_t row_pitch_pixels, uint32_t u, uint32_t v, bool swizzled) { + if (!swizzled) return (v * (row_pitch_pixels * texel_size_bits >> 3)) + (u * texel_size_bits >> 3); const int tile_size_bits = 32; @@ -263,12 +262,10 @@ static inline int GetPixelDataOffset(unsigned int row_pitch_pixels, unsigned int return tile_idx * (tile_size_bits / 8) + ((u % texels_per_tile) * texel_size_bits) / 8; } -static inline u32 LookupColor(unsigned int index, unsigned int level) -{ - const bool mipmapShareClut = gstate.isClutSharedForMipmaps(); - const int clutSharingOffset = mipmapShareClut ? 0 : level * 16; +static inline u32 LookupColor(unsigned int index, unsigned int level, const SamplerID &samplerID) { + const int clutSharingOffset = samplerID.useSharedClut ? 0 : level * 16; - switch (gstate.getClutPaletteFormat()) { + switch (samplerID.ClutFmt()) { case GE_CMODE_16BIT_BGR5650: return RGB565ToRGBA8888(reinterpret_cast(clut)[index + clutSharingOffset]); @@ -282,11 +279,24 @@ static inline u32 LookupColor(unsigned int index, unsigned int level) return clut[index + clutSharingOffset]; default: - ERROR_LOG_REPORT(G3D, "Software: Unsupported palette format: %x", gstate.getClutPaletteFormat()); + ERROR_LOG_REPORT(G3D, "Software: Unsupported palette format: %x", samplerID.ClutFmt()); return 0; } } +uint32_t TransformClutIndex(uint32_t index, const SamplerID &samplerID) { + if (samplerID.hasClutShift || samplerID.hasClutMask || samplerID.hasClutOffset) { + const uint8_t shift = (samplerID.cached.clutFormat >> 2) & 0x1F; + const uint8_t mask = (samplerID.cached.clutFormat >> 8) & 0xFF; + const uint16_t offset = ((samplerID.cached.clutFormat >> 16) & 0x1F) << 4; + // We need to wrap any entries beyond the first 1024 bytes. + const uint16_t offsetMask = samplerID.ClutFmt() == GE_CMODE_32BIT_ABGR8888 ? 0xFF : 0x1FF; + + return ((index >> shift) & mask) | (offset & offsetMask); + } + return index & 0xFF; +} + struct Nearest4 { alignas(16) u32 v[4]; @@ -296,76 +306,74 @@ struct Nearest4 { }; template -inline static Nearest4 SOFTRAST_CALL SampleNearest(const int u[N], const int v[N], const u8 *srcptr, int texbufw, int level) { +inline static Nearest4 SOFTRAST_CALL SampleNearest(const int u[N], const int v[N], const u8 *srcptr, int texbufw, int level, const SamplerID &samplerID) { Nearest4 res; if (!srcptr) { memset(res.v, 0, sizeof(res.v)); return res; } - GETextureFormat texfmt = gstate.getTextureFormat(); - // TODO: Should probably check if textures are aligned properly... - switch (texfmt) { + switch (samplerID.TexFmt()) { case GE_TFMT_4444: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle); res.v[i] = RGBA4444ToRGBA8888(*(const u16 *)src); } return res; case GE_TFMT_5551: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle); res.v[i] = RGBA5551ToRGBA8888(*(const u16 *)src); } return res; case GE_TFMT_5650: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle); res.v[i] = RGB565ToRGBA8888(*(const u16 *)src); } return res; case GE_TFMT_8888: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i], samplerID.swizzle); res.v[i] = *(const u32 *)src; } return res; case GE_TFMT_CLUT32: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i], samplerID.swizzle); u32 val = src[0] + (src[1] << 8) + (src[2] << 16) + (src[3] << 24); - res.v[i] = LookupColor(gstate.transformClutIndex(val), 0); + res.v[i] = LookupColor(TransformClutIndex(val, samplerID), 0, samplerID); } return res; case GE_TFMT_CLUT16: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle); u16 val = src[0] + (src[1] << 8); - res.v[i] = LookupColor(gstate.transformClutIndex(val), 0); + res.v[i] = LookupColor(TransformClutIndex(val, samplerID), 0, samplerID); } return res; case GE_TFMT_CLUT8: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<8>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<8>(texbufw, u[i], v[i], samplerID.swizzle); u8 val = *src; - res.v[i] = LookupColor(gstate.transformClutIndex(val), 0); + res.v[i] = LookupColor(TransformClutIndex(val, samplerID), 0, samplerID); } return res; case GE_TFMT_CLUT4: for (int i = 0; i < N; ++i) { - const u8 *src = srcptr + GetPixelDataOffset<4>(texbufw, u[i], v[i]); + const u8 *src = srcptr + GetPixelDataOffset<4>(texbufw, u[i], v[i], samplerID.swizzle); u8 val = (u[i] & 1) ? (src[0] >> 4) : (src[0] & 0xF); // Only CLUT4 uses separate mipmap palettes. - res.v[i] = LookupColor(gstate.transformClutIndex(val), level); + res.v[i] = LookupColor(TransformClutIndex(val, samplerID), level, samplerID); } return res; @@ -391,7 +399,7 @@ inline static Nearest4 SOFTRAST_CALL SampleNearest(const int u[N], const int v[N return res; default: - ERROR_LOG_REPORT(G3D, "Software: Unsupported texture format: %x", texfmt); + ERROR_LOG_REPORT(G3D, "Software: Unsupported texture format: %x", samplerID.TexFmt()); memset(res.v, 0, sizeof(res.v)); return res; } @@ -410,8 +418,8 @@ static inline int WrapUV(int v, int height) { } template -static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], const int v[N], int width, int height) { - if (gstate.isTexCoordClampedS()) { +static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], const int v[N], int width, int height, const SamplerID &samplerID) { + if (samplerID.clampS) { for (int i = 0; i < N; ++i) { out_u[i] = ClampUV(u[i], width); } @@ -420,7 +428,7 @@ static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], c out_u[i] = WrapUV(u[i], width); } } - if (gstate.isTexCoordClampedT()) { + if (samplerID.clampT) { for (int i = 0; i < N; ++i) { out_v[i] = ClampUV(v[i], height); } @@ -431,9 +439,9 @@ static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], c } } -static inline void GetTexelCoordinates(int level, float s, float t, int &out_u, int &out_v, int x, int y) { - int width = gstate.getTextureWidth(level); - int height = gstate.getTextureHeight(level); +static inline void GetTexelCoordinates(int level, float s, float t, int &out_u, int &out_v, int x, int y, const SamplerID &samplerID) { + int width = samplerID.cached.sizes[level].w; + int height = samplerID.cached.sizes[level].h; int base_u = (int)(s * width * 256.0f) + 12 - x; int base_v = (int)(t * height * 256.0f) + 12 - y; @@ -441,28 +449,134 @@ static inline void GetTexelCoordinates(int level, float s, float t, int &out_u, base_u >>= 8; base_v >>= 8; - ApplyTexelClamp<1>(&out_u, &out_v, &base_u, &base_v, width, height); + ApplyTexelClamp<1>(&out_u, &out_v, &base_u, &base_v, width, height, samplerID); } -static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac) { +Vec4IntResult SOFTRAST_CALL GetTextureFunctionOutput(Vec4IntArg prim_color_in, Vec4IntArg texcolor_in, const SamplerID &samplerID) { + const Vec4 prim_color = prim_color_in; + const Vec4 texcolor = texcolor_in; + + Vec3 out_rgb; + int out_a; + + bool rgba = samplerID.useTextureAlpha; + + switch (samplerID.TexFunc()) { + case GE_TEXFUNC_MODULATE: + { +#if defined(_M_SSE) + // Modulate weights slightly on the tex color, by adding one to prim and dividing by 256. + const __m128i p = _mm_slli_epi16(_mm_packs_epi32(prim_color.ivec, prim_color.ivec), 4); + const __m128i pboost = _mm_add_epi16(p, _mm_set1_epi16(1 << 4)); + __m128i t = _mm_slli_epi16(_mm_packs_epi32(texcolor.ivec, texcolor.ivec), 4); + if (samplerID.useColorDoubling) { + const __m128i amask = _mm_set_epi16(-1, 0, 0, 0, -1, 0, 0, 0); + const __m128i a = _mm_and_si128(t, amask); + const __m128i rgb = _mm_andnot_si128(amask, t); + t = _mm_or_si128(_mm_slli_epi16(rgb, 1), a); + } + const __m128i b = _mm_mulhi_epi16(pboost, t); + out_rgb.ivec = _mm_unpacklo_epi16(b, _mm_setzero_si128()); + + if (rgba) { + return ToVec4IntResult(Vec4(out_rgb.ivec)); + } else { + out_a = prim_color.a(); + } +#else + if (samplerID.useColorDoubling) { + out_rgb = ((prim_color.rgb() + Vec3::AssignToAll(1)) * texcolor.rgb() * 2) / 256; + } else { + out_rgb = (prim_color.rgb() + Vec3::AssignToAll(1)) * texcolor.rgb() / 256; + } + out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a(); +#endif + break; + } + + case GE_TEXFUNC_DECAL: + if (rgba) { + int t = texcolor.a(); + int invt = 255 - t; + // Both colors are boosted here, making the alpha have more weight. + Vec3 one = Vec3::AssignToAll(1); + out_rgb = ((prim_color.rgb() + one) * invt + (texcolor.rgb() + one) * t); + // Keep the bits of accuracy when doubling. + if (samplerID.useColorDoubling) + out_rgb /= 128; + else + out_rgb /= 256; + } else { + if (samplerID.useColorDoubling) + out_rgb = texcolor.rgb() * 2; + else + out_rgb = texcolor.rgb(); + } + out_a = prim_color.a(); + break; + + case GE_TEXFUNC_BLEND: + { + const Vec3 const255(255, 255, 255); + const Vec3 texenv = Vec3::FromRGB(samplerID.cached.texBlendColor); + + // Unlike the others (and even alpha), this one simply always rounds up. + const Vec3 roundup = Vec3::AssignToAll(255); + out_rgb = ((const255 - texcolor.rgb()) * prim_color.rgb() + texcolor.rgb() * texenv + roundup); + // Must divide by less to keep the precision for doubling to be accurate. + if (samplerID.useColorDoubling) + out_rgb /= 128; + else + out_rgb /= 256; + + out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a(); + break; + } + + case GE_TEXFUNC_REPLACE: + out_rgb = texcolor.rgb(); + // Doubling even happens for replace. + if (samplerID.useColorDoubling) + out_rgb *= 2; + out_a = (rgba) ? texcolor.a() : prim_color.a(); + break; + + case GE_TEXFUNC_ADD: + case GE_TEXFUNC_UNKNOWN1: + case GE_TEXFUNC_UNKNOWN2: + case GE_TEXFUNC_UNKNOWN3: + // Don't need to clamp afterward, we always clamp before tests. + out_rgb = prim_color.rgb() + texcolor.rgb(); + if (samplerID.useColorDoubling) + out_rgb *= 2; + + // Alpha is still blended the common way. + out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a(); + break; + } + + return ToVec4IntResult(Vec4(out_rgb, out_a)); +} + +static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID) { int u, v; // Nearest filtering only. Round texcoords. - GetTexelCoordinates(level, s, t, u, v, x, y); - Vec4 c0 = Vec4::FromRGBA(SampleNearest<1>(&u, &v, tptr[0], bufw[0], level).v[0]); + GetTexelCoordinates(level, s, t, u, v, x, y, samplerID); + Vec4 c0 = Vec4::FromRGBA(SampleNearest<1>(&u, &v, tptr[0], bufw[0], level, samplerID).v[0]); if (levelFrac) { - GetTexelCoordinates(level + 1, s, t, u, v, x, y); - Vec4 c1 = Vec4::FromRGBA(SampleNearest<1>(&u, &v, tptr[1], bufw[1], level + 1).v[0]); + GetTexelCoordinates(level + 1, s, t, u, v, x, y, samplerID); + Vec4 c1 = Vec4::FromRGBA(SampleNearest<1>(&u, &v, tptr[1], bufw[1], level + 1, samplerID).v[0]); c0 = (c1 * levelFrac + c0 * (16 - levelFrac)) / 16; } - return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0)); + return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0), samplerID); } -static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level) { - Nearest4 c = SampleNearest<1>(&u, &v, tptr, bufw, level); +static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level, const SamplerID &samplerID) { + Nearest4 c = SampleNearest<1>(&u, &v, tptr, bufw, level, samplerID); return ToVec4IntResult(Vec4::FromRGBA(c.v[0])); } @@ -518,33 +632,33 @@ static inline Vec4IntResult SOFTRAST_CALL ApplyTexelClampQuadT(bool clamp, int v #endif } -static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadS(int level, float in_s, int &frac_u, int x) { - int width = gstate.getTextureWidth(level); +static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadS(int level, float in_s, int &frac_u, int x, const SamplerID &samplerID) { + int width = samplerID.cached.sizes[level].w; int base_u = (int)(in_s * width * 256) + 12 - x - 128; frac_u = (int)(base_u >> 4) & 0x0F; base_u >>= 8; // Need to generate and individually wrap/clamp the four sample coordinates. Ugh. - return ApplyTexelClampQuadS(gstate.isTexCoordClampedS(), base_u, width); + return ApplyTexelClampQuadS(samplerID.clampS, base_u, width); } -static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadT(int level, float in_t, int &frac_v, int y) { - int height = gstate.getTextureHeight(level); +static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadT(int level, float in_t, int &frac_v, int y, const SamplerID &samplerID) { + int height = samplerID.cached.sizes[level].h; int base_v = (int)(in_t * height * 256) + 12 - y - 128; frac_v = (int)(base_v >> 4) & 0x0F; base_v >>= 8; // Need to generate and individually wrap/clamp the four sample coordinates. Ugh. - return ApplyTexelClampQuadT(gstate.isTexCoordClampedT(), base_v, height); + return ApplyTexelClampQuadT(samplerID.clampT, base_v, height); } -static Vec4IntResult SOFTRAST_CALL SampleLinearLevel(float s, float t, int x, int y, const u8 *const *tptr, const int *bufw, int texlevel) { +static Vec4IntResult SOFTRAST_CALL SampleLinearLevel(float s, float t, int x, int y, const u8 *const *tptr, const int *bufw, int texlevel, const SamplerID &samplerID) { int frac_u, frac_v; - const Vec4 u = GetTexelCoordinatesQuadS(texlevel, s, frac_u, x); - const Vec4 v = GetTexelCoordinatesQuadT(texlevel, t, frac_v, y); - Nearest4 c = SampleNearest<4>(u.AsArray(), v.AsArray(), tptr[0], bufw[0], texlevel); + const Vec4 u = GetTexelCoordinatesQuadS(texlevel, s, frac_u, x, samplerID); + const Vec4 v = GetTexelCoordinatesQuadT(texlevel, t, frac_v, y, samplerID); + Nearest4 c = SampleNearest<4>(u.AsArray(), v.AsArray(), tptr[0], bufw[0], texlevel, samplerID); Vec4 texcolor_tl = Vec4::FromRGBA(c.v[0]); Vec4 texcolor_tr = Vec4::FromRGBA(c.v[1]); @@ -555,13 +669,13 @@ static Vec4IntResult SOFTRAST_CALL SampleLinearLevel(float s, float t, int x, in return ToVec4IntResult((top * (0x10 - frac_v) + bot * frac_v) / (16 * 16)); } -static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int texlevel, int levelFrac) { - Vec4 c0 = SampleLinearLevel(s, t, x, y, tptr, bufw, texlevel); +static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int texlevel, int levelFrac, const SamplerID &samplerID) { + Vec4 c0 = SampleLinearLevel(s, t, x, y, tptr, bufw, texlevel, samplerID); if (levelFrac) { - const Vec4 c1 = SampleLinearLevel(s, t, x, y, tptr + 1, bufw + 1, texlevel + 1); + const Vec4 c1 = SampleLinearLevel(s, t, x, y, tptr + 1, bufw + 1, texlevel + 1, samplerID); c0 = (c1 * levelFrac + c0 * (16 - levelFrac)) / 16; } - return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0)); + return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0), samplerID); } }; diff --git a/GPU/Software/Sampler.h b/GPU/Software/Sampler.h index 4ae9182407..b154454cd2 100644 --- a/GPU/Software/Sampler.h +++ b/GPU/Software/Sampler.h @@ -33,13 +33,13 @@ namespace Sampler { #pragma GCC diagnostic ignored "-Wignored-attributes" #endif -typedef Rasterizer::Vec4IntResult(SOFTRAST_CALL *FetchFunc)(int u, int v, const u8 *tptr, int bufw, int level); +typedef Rasterizer::Vec4IntResult(SOFTRAST_CALL *FetchFunc)(int u, int v, const u8 *tptr, int bufw, int level, const SamplerID &samplerID); FetchFunc GetFetchFunc(SamplerID id); -typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *NearestFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac); +typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *NearestFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID); NearestFunc GetNearestFunc(SamplerID id); -typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *LinearFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac); +typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *LinearFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID); LinearFunc GetLinearFunc(SamplerID id); void Init(); @@ -67,7 +67,8 @@ private: void Describe(const std::string &message); Rasterizer::RegCache::Reg GetZeroVec(); - Rasterizer::RegCache::Reg GetGState(); + Rasterizer::RegCache::Reg GetSamplerID(); + void UnlockSamplerID(Rasterizer::RegCache::Reg &r); void WriteConstantPool(const SamplerID &id); @@ -104,6 +105,7 @@ private: Arm64Gen::ARM64FloatEmitter fp; #elif PPSSPP_ARCH(AMD64) || PPSSPP_ARCH(X86) int stackArgPos_ = 0; + int stackIDOffset_ = -1; int stackFracUV1Offset_ = 0; #endif diff --git a/GPU/Software/SamplerX86.cpp b/GPU/Software/SamplerX86.cpp index 56a22ba491..ebdbd8e11f 100644 --- a/GPU/Software/SamplerX86.cpp +++ b/GPU/Software/SamplerX86.cpp @@ -41,6 +41,7 @@ FetchFunc SamplerJitCache::CompileFetch(const SamplerID &id) { RegCache::GEN_ARG_TEXPTR, RegCache::GEN_ARG_BUFW, RegCache::GEN_ARG_LEVEL, + RegCache::GEN_ARG_ID, }); regCache_.ChangeReg(RAX, RegCache::GEN_RESULT); regCache_.ChangeReg(XMM0, RegCache::VEC_RESULT); @@ -49,6 +50,15 @@ FetchFunc SamplerJitCache::CompileFetch(const SamplerID &id) { Describe("Init"); const u8 *start = AlignCode16(); +#if PPSSPP_PLATFORM(WINDOWS) + // RET and shadow space. + stackArgPos_ = 8 + 32; + stackIDOffset_ = 8; +#else + stackArgPos_ = 0; + stackIDOffset_ = -1; +#endif + // Early exit on !srcPtr. FixupBranch zeroSrc; if (id.hasInvalidPtr) { @@ -123,6 +133,7 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) { RegCache::GEN_ARG_BUFW_PTR, RegCache::GEN_ARG_LEVEL, RegCache::GEN_ARG_LEVELFRAC, + RegCache::GEN_ARG_ID, }); #if PPSSPP_PLATFORM(WINDOWS) @@ -130,11 +141,14 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) { stackArgPos_ = 8 + 32 + 8; // Positions: stackArgPos_+0=src, stackArgPos_+8=bufw, stackArgPos_+16=level, stackArgPos_+24=levelFrac + stackIDOffset_ = 32; // Use the shadow space to save U1/V1. We also use this var for frac U1/V1 in linear. stackFracUV1Offset_ = -8; #else stackArgPos_ = 0; + // This is the only arg that went to the stack. + stackIDOffset_ = 8; // Use the red zone. stackFracUV1Offset_ = -8; #endif @@ -283,7 +297,7 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) { regCache_.Unlock(vReg, RegCache::GEN_ARG_V); regCache_.ForceRetain(RegCache::GEN_ARG_V); - bool hadGState = regCache_.Has(RegCache::GEN_GSTATE); + bool hadId = regCache_.Has(RegCache::GEN_ID); bool hadZero = regCache_.Has(RegCache::VEC_ZERO); success = success && Jit_ReadTextureFormat(id); @@ -299,8 +313,8 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) { regCache_.Unlock(resultReg, RegCache::GEN_RESULT); // Since we're inside a conditional, make sure these go away if we allocated them. - if (!hadGState && regCache_.Has(RegCache::GEN_GSTATE)) - regCache_.ForceRelease(RegCache::GEN_GSTATE); + if (!hadId && regCache_.Has(RegCache::GEN_ID)) + regCache_.ForceRelease(RegCache::GEN_ID); if (!hadZero && regCache_.Has(RegCache::VEC_ZERO)) regCache_.ForceRelease(RegCache::VEC_ZERO); @@ -493,6 +507,7 @@ LinearFunc SamplerJitCache::CompileLinear(const SamplerID &id) { RegCache::GEN_ARG_BUFW_PTR, RegCache::GEN_ARG_LEVEL, RegCache::GEN_ARG_LEVELFRAC, + RegCache::GEN_ARG_ID, }); #if PPSSPP_PLATFORM(WINDOWS) @@ -500,8 +515,10 @@ LinearFunc SamplerJitCache::CompileLinear(const SamplerID &id) { stackArgPos_ = 8 + 32 + 8; // Positions: stackArgPos_+0=src, stackArgPos_+8=bufw, stackArgPos_+16=level, stackArgPos_+24=levelFrac + stackIDOffset_ = 32; #else stackArgPos_ = 0; + stackIDOffset_ = 8; #endif // Start out by saving some registers, since we'll need more. @@ -931,13 +948,23 @@ RegCache::Reg SamplerJitCache::GetZeroVec() { return regCache_.Find(RegCache::VEC_ZERO); } -RegCache::Reg SamplerJitCache::GetGState() { - if (!regCache_.Has(RegCache::GEN_GSTATE)) { - X64Reg r = regCache_.Alloc(RegCache::GEN_GSTATE); - MOV(PTRBITS, R(r), ImmPtr(&gstate.nop)); +RegCache::Reg SamplerJitCache::GetSamplerID() { + if (regCache_.Has(RegCache::GEN_ARG_ID)) + return regCache_.Find(RegCache::GEN_ARG_ID); + if (!regCache_.Has(RegCache::GEN_ID)) { + X64Reg r = regCache_.Alloc(RegCache::GEN_ID); + _assert_(stackIDOffset_ != -1); + MOV(PTRBITS, R(r), MDisp(RSP, stackArgPos_ + stackIDOffset_)); return r; } - return regCache_.Find(RegCache::GEN_GSTATE); + return regCache_.Find(RegCache::GEN_ID); +} + +void SamplerJitCache::UnlockSamplerID(RegCache::Reg &r) { + if (regCache_.Has(RegCache::GEN_ARG_ID)) + regCache_.Unlock(r, RegCache::GEN_ARG_ID); + else + regCache_.Unlock(r, RegCache::GEN_ID); } bool SamplerJitCache::Jit_FetchQuad(const SamplerID &id, bool level1) { @@ -1123,14 +1150,14 @@ bool SamplerJitCache::Jit_TransformClutIndexQuad(const SamplerID &id, int bitsPe X64Reg indexReg = regCache_.Find(RegCache::VEC_INDEX); bool maskedIndex = false; - // Okay, first load the actual gstate clutformat bits we'll use. + // Okay, first load the actual samplerID clutformat bits we'll use. X64Reg formatReg = regCache_.Alloc(RegCache::VEC_TEMP0); - X64Reg gstateReg = GetGState(); + X64Reg idReg = GetSamplerID(); if (cpu_info.bAVX2 && !id.hasClutShift) - VPBROADCASTD(128, formatReg, MDisp(gstateReg, offsetof(GPUgstate, clutformat))); + VPBROADCASTD(128, formatReg, MDisp(idReg, offsetof(SamplerID, cached.clutFormat))); else - MOVD_xmm(formatReg, MDisp(gstateReg, offsetof(GPUgstate, clutformat))); - regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE); + MOVD_xmm(formatReg, MDisp(idReg, offsetof(SamplerID, cached.clutFormat))); + UnlockSamplerID(idReg); // Shift = (clutformat >> 2) & 0x1F if (id.hasClutShift) { @@ -1610,19 +1637,19 @@ bool SamplerJitCache::Jit_ApplyTextureFunc(const SamplerID &id) { // We divide later. } - X64Reg gstateReg = GetGState(); + X64Reg idReg = GetSamplerID(); X64Reg texEnvReg = regCache_.Alloc(RegCache::VEC_TEMP1); if (cpu_info.bSSE4_1) { - PMOVZXBW(texEnvReg, MDisp(gstateReg, offsetof(GPUgstate, texenvcolor))); + PMOVZXBW(texEnvReg, MDisp(idReg, offsetof(SamplerID, cached.texBlendColor))); } else { - MOVD_xmm(texEnvReg, MDisp(gstateReg, offsetof(GPUgstate, texenvcolor))); + MOVD_xmm(texEnvReg, MDisp(idReg, offsetof(SamplerID, cached.texBlendColor))); X64Reg zeroReg = GetZeroVec(); PUNPCKLBW(texEnvReg, R(zeroReg)); regCache_.Unlock(zeroReg, RegCache::VEC_ZERO); } PMULLW(resultReg, R(texEnvReg)); regCache_.Release(texEnvReg, RegCache::VEC_TEMP1); - regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE); + UnlockSamplerID(idReg); // Add in the prim color side and divide. PADDW(resultReg, R(tempReg)); @@ -2418,8 +2445,7 @@ bool SamplerJitCache::Jit_GetTexelCoords(const SamplerID &id) { X64Reg tReg = regCache_.Find(RegCache::VEC_ARG_T); if (constWidth256f_ == nullptr) { // We have to figure out levels and the proper width, ugh. - X64Reg shiftReg = regCache_.Find(RegCache::GEN_SHIFTVAL); - X64Reg gstateReg = GetGState(); + X64Reg idReg = GetSamplerID(); X64Reg tempReg = regCache_.Alloc(RegCache::GEN_TEMP0); X64Reg levelReg = INVALID_REG; @@ -2432,12 +2458,9 @@ bool SamplerJitCache::Jit_GetTexelCoords(const SamplerID &id) { X64Reg tempVecReg = regCache_.Alloc(RegCache::VEC_TEMP0); auto loadSizeAndMul = [&](X64Reg dest, bool clamp, bool isY, bool isLevel1) { - int offset = offsetof(GPUgstate, texsize) + (isY ? 1 : 0) + (isLevel1 ? 4 : 0); - // Grab the size, and shift. - MOVZX(32, 8, shiftReg, MComplex(gstateReg, levelReg, SCALE_4, offset)); - AND(32, R(shiftReg), Imm8(0x0F)); - MOV(32, R(tempReg), Imm32(1)); - SHL(32, R(tempReg), R(shiftReg)); + int offset = offsetof(SamplerID, cached.sizes[0].w) + (isY ? 2 : 0) + (isLevel1 ? 4 : 0); + // Grab the size, already pre-shifted for us. + MOVZX(32, 16, tempReg, MComplex(idReg, levelReg, SCALE_4, offset)); // Now move to a float, multiplying by 256 meanwhile with a shift. MOVD_xmm(tempVecReg, R(tempReg)); @@ -2477,8 +2500,7 @@ bool SamplerJitCache::Jit_GetTexelCoords(const SamplerID &id) { loadSizeAndMul(uReg, id.clampS, false, false); loadSizeAndMul(vReg, id.clampT, true, false); - regCache_.Unlock(shiftReg, RegCache::GEN_SHIFTVAL); - regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE); + UnlockSamplerID(idReg); regCache_.Release(tempReg, RegCache::GEN_TEMP0); regCache_.Unlock(levelReg, RegCache::GEN_ARG_LEVEL); @@ -2552,7 +2574,7 @@ bool SamplerJitCache::Jit_GetTexelCoordsQuad(const SamplerID &id) { if (constWidth256f_ == nullptr) { // We have to figure out levels and the proper width, ugh. - X64Reg gstateReg = GetGState(); + X64Reg idReg = GetSamplerID(); X64Reg tempReg = regCache_.Alloc(RegCache::GEN_TEMP0); X64Reg levelReg = INVALID_REG; @@ -2567,18 +2589,22 @@ bool SamplerJitCache::Jit_GetTexelCoordsQuad(const SamplerID &id) { } // This will load the current and next level's sizes. - MOV(64, R(tempReg), MComplex(gstateReg, levelReg, SCALE_4, offsetof(GPUgstate, texsize))); - regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE); + MOV(64, R(tempReg), MComplex(idReg, levelReg, SCALE_4, offsetof(SamplerID, cached.sizes[0].w))); + UnlockSamplerID(idReg); X64Reg tempVecReg = regCache_.Alloc(RegCache::VEC_TEMP0); auto loadSizeAndMul = [&](X64Reg dest, X64Reg size, bool isY, bool isLevel1) { - // Grab the size and mask out 4 bits using walls. - MOVQ_xmm(tempVecReg, R(tempReg)); - PSLLQ(tempVecReg, 60 - (isY ? 8 : 0) - (isLevel1 ? 32 : 0)); - PSRLQ(tempVecReg, 60); - - MOVDQA(size, M(constOnes32_)); - PSLLD(size, R(tempVecReg)); + // Grab the size and mask out 16 bits using walls. + MOVQ_xmm(size, R(tempReg)); + if (cpu_info.bSSE4_1) { + int lane = (isY ? 1 : 0) + (isLevel1 ? 2 : 0); + PMOVZXWD(size, R(size)); + PSHUFD(size, R(size), _MM_SHUFFLE(lane, lane, lane, lane)); + } else { + PSLLQ(size, 48 - (isY ? 16 : 0) - (isLevel1 ? 32 : 0)); + PSRLQ(size, 48); + PSHUFD(size, R(size), _MM_SHUFFLE(0, 0, 0, 0)); + } if (cpu_info.bAVX) { VPSLLD(128, tempVecReg, size, 8); @@ -3354,8 +3380,9 @@ bool SamplerJitCache::Jit_TransformClutIndex(const SamplerID &id, int bitsPerInd _assert_msg_(hasRCX, "Could not obtain RCX, locked?"); X64Reg temp1Reg = regCache_.Alloc(RegCache::GEN_TEMP1); - MOV(PTRBITS, R(temp1Reg), ImmPtr(&gstate.clutformat)); - MOV(32, R(temp1Reg), MatR(temp1Reg)); + X64Reg idReg = GetSamplerID(); + MOV(32, R(temp1Reg), MDisp(idReg, offsetof(SamplerID, cached.clutFormat))); + UnlockSamplerID(idReg); X64Reg resultReg = regCache_.Find(RegCache::GEN_RESULT); @@ -3418,7 +3445,7 @@ bool SamplerJitCache::Jit_ReadClutColor(const SamplerID &id) { } else { #if PPSSPP_PLATFORM(WINDOWS) // The argument was saved on the stack. - MOV(32, R(temp2Reg), MDisp(RSP, 40)); + MOV(32, R(temp2Reg), MDisp(RSP, stackArgPos_)); #else _assert_(false); #endif diff --git a/GPU/Software/TransformUnit.cpp b/GPU/Software/TransformUnit.cpp index 63d6b0d26c..8ab3c1f789 100644 --- a/GPU/Software/TransformUnit.cpp +++ b/GPU/Software/TransformUnit.cpp @@ -133,22 +133,11 @@ ScreenCoords TransformUnit::ClipToScreen(const ClipCoords& coords) return ClipToScreenInternal(coords, nullptr); } -DrawingCoords TransformUnit::ScreenToDrawing(const ScreenCoords& coords) -{ - DrawingCoords ret; - // TODO: What to do when offset > coord? - ret.x = ((s32)coords.x - gstate.getOffsetX16()) / 16; - ret.y = ((s32)coords.y - gstate.getOffsetY16()) / 16; - ret.z = coords.z; - return ret; -} - -ScreenCoords TransformUnit::DrawingToScreen(const DrawingCoords& coords) -{ +ScreenCoords TransformUnit::DrawingToScreen(const DrawingCoords &coords, u16 z) { ScreenCoords ret; ret.x = (u32)coords.x * 16 + gstate.getOffsetX16(); ret.y = (u32)coords.y * 16 + gstate.getOffsetY16(); - ret.z = coords.z; + ret.z = z; return ret; } @@ -708,7 +697,7 @@ bool TransformUnit::GetCurrentSimpleVertices(int count, std::vector xy() const { return Vec2(x, y); } - - DrawingCoords operator * (const float t) const - { - return DrawingCoords((s16)(x * t), (s16)(y * t), (u16)(z * t)); - } - - DrawingCoords operator + (const DrawingCoords& oth) const - { - return DrawingCoords(x + oth.x, y + oth.y, z + oth.z); - } }; struct VertexData { @@ -116,8 +102,17 @@ public: static ViewCoords WorldToView(const WorldCoords& coords); static ClipCoords ViewToClip(const ViewCoords& coords); static ScreenCoords ClipToScreen(const ClipCoords& coords); - static DrawingCoords ScreenToDrawing(const ScreenCoords& coords); - static ScreenCoords DrawingToScreen(const DrawingCoords& coords); + static inline DrawingCoords ScreenToDrawing(int x, int y, int offsetX, int offsetY) { + DrawingCoords ret; + // When offset > coord, it correctly goes negative and force-scissors. + ret.x = (x - offsetX) / 16; + ret.y = (y - offsetY) / 16; + return ret; + } + static inline DrawingCoords ScreenToDrawing(const ScreenCoords &coords, int offsetX, int offsetY) { + return ScreenToDrawing(coords.x, coords.y, offsetX, offsetY); + } + static ScreenCoords DrawingToScreen(const DrawingCoords &coords, u16 z); void SubmitPrimitive(void* vertices, void* indices, GEPrimitiveType prim_type, int vertex_count, u32 vertex_type, int *bytesRead, SoftwareDrawEngine *drawEngine);