Merge pull request #15311 from unknownbrackets/softgpu-state

Avoid gstate references in rasterizerization
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2022-01-16 09:40:25 +01:00
commit 9bef900cd7
14 files changed
+327 -288

No files matched your search

+1 -3
View File
@@ -298,9 +298,7 @@ struct GPUgstate {
bool isTextureAlphaUsed() const { return (texfunc & 0x100) != 0; }
GETextureFormat getTextureFormat() const { return static_cast<GETextureFormat>(texformat & 0xF); }
bool isTextureFormatIndexed() const { return (texformat & 4) != 0; } // GE_TFMT_CLUT4 - GE_TFMT_CLUT32 are 0b1xx.
int getTextureEnvColR() const { return texenvcolor&0xFF; }
int getTextureEnvColG() const { return (texenvcolor>>8)&0xFF; }
int getTextureEnvColB() const { return (texenvcolor>>16)&0xFF; }
int getTextureEnvColRGB() const { return texenvcolor & 0x00FFFFFF; }
u32 getClutAddress() const { return (clutaddr & 0x00FFFFF0) | ((clutaddrupper << 8) & 0x0F000000); }
int getClutLoadBytes() const { return (loadclut & 0x3F) * 32; }
int getClutLoadBlocks() const { return (loadclut & 0x3F); }
+6 -6
View File
@@ -136,10 +136,10 @@ void BinManager::UpdateState() {
stateIndex_ = (int)states_.Push(RasterizerState());
ComputeRasterizerState(&states_[stateIndex_]);
DrawingCoords scissorTL(gstate.getScissorX1(), gstate.getScissorY1(), 0);
DrawingCoords scissorBR(gstate.getScissorX2(), gstate.getScissorY2(), 0);
ScreenCoords screenScissorTL = TransformUnit::DrawingToScreen(scissorTL);
ScreenCoords screenScissorBR = TransformUnit::DrawingToScreen(scissorBR);
DrawingCoords scissorTL(gstate.getScissorX1(), gstate.getScissorY1());
DrawingCoords scissorBR(gstate.getScissorX2(), gstate.getScissorY2());
ScreenCoords screenScissorTL = TransformUnit::DrawingToScreen(scissorTL, 0);
ScreenCoords screenScissorBR = TransformUnit::DrawingToScreen(scissorBR, 0);
scissor_.x1 = screenScissorTL.x;
scissor_.y1 = screenScissorTL.y;
@@ -237,8 +237,8 @@ void BinManager::Drain() {
int h2 = (queueRange_.y2 - queueRange_.y1 + 31) / 32;
// Always bin the entire possible range, but focus on the drawn area.
ScreenCoords tl = TransformUnit::DrawingToScreen(DrawingCoords(0, 0, 0));
ScreenCoords br = TransformUnit::DrawingToScreen(DrawingCoords(1024, 1024, 0));
ScreenCoords tl = TransformUnit::DrawingToScreen(DrawingCoords(0, 0), 0);
ScreenCoords br = TransformUnit::DrawingToScreen(DrawingCoords(1024, 1024), 0);
taskRanges_.clear();
if (h2 >= 18 && w2 >= h2 * 4) {
+2 -2
View File
@@ -460,7 +460,7 @@ void SOFTRAST_CALL DrawSinglePixel(int x, int y, int z, int fog, Vec4IntArg colo
const Vec4<int> dst = Vec4<int>::FromRGBA(old_color);
Vec3<int> blended = AlphaBlendingResult(pixelID, prim_color, dst);
if (pixelID.dithering) {
blended += Vec3<int>::AssignToAll(pixelID.cached.ditherMatrix[y * 4 + x]);
blended += Vec3<int>::AssignToAll(pixelID.cached.ditherMatrix[(y & 3) * 4 + (x & 3)]);
}
// ToRGB() always automatically clamps.
@@ -469,7 +469,7 @@ void SOFTRAST_CALL DrawSinglePixel(int x, int y, int z, int fog, Vec4IntArg colo
} else {
if (pixelID.dithering) {
// We'll discard alpha anyway.
prim_color += Vec4<int>::AssignToAll(pixelID.cached.ditherMatrix[y * 4 + x]);
prim_color += Vec4<int>::AssignToAll(pixelID.cached.ditherMatrix[(y & 3) * 4 + (x & 3)]);
}
#if defined(_M_SSE)
+10 -1
View File
@@ -22,7 +22,7 @@
#include "GPU/GPUState.h"
#include "GPU/Software/FuncId.h"
static_assert(sizeof(SamplerID) == sizeof(SamplerID::fullKey), "Bad sampler ID size");
static_assert(sizeof(SamplerID) == sizeof(SamplerID::fullKey) + sizeof(SamplerID::cached), "Bad sampler ID size");
static_assert(sizeof(PixelFuncID) == sizeof(PixelFuncID::fullKey) + sizeof(PixelFuncID::cached), "Bad pixel func ID size");
static inline GEComparison OptimizeRefByteCompare(GEComparison func, u8 ref) {
@@ -199,6 +199,8 @@ void ComputePixelFuncID(PixelFuncID *id) {
if (id->applyFog) {
id->cached.fogColor = gstate.fogcolor & 0x00FFFFFF;
}
if (id->applyLogicOp)
id->cached.logicOp = gstate.getLogicOp();
id->cached.minz = gstate.getDepthRangeMin();
id->cached.maxz = gstate.getDepthRangeMax();
id->cached.framebufStride = gstate.FrameBufStride();
@@ -399,6 +401,9 @@ void ComputeSamplerID(SamplerID *id_out) {
int bytes = h * (bufw * bitspp) / 8;
if (bitspp < 32 && !Memory::IsValidAddress(addr + bytes + (32 - bitspp) / 8))
id.overReadSafe = false;
id.cached.sizes[i].w = w;
id.cached.sizes[i].h = h;
}
id.hasAnyMips = maxLevel != 0;
@@ -411,6 +416,7 @@ void ComputeSamplerID(SamplerID *id_out) {
id.hasClutMask = gstate.getClutIndexMask() != 0xFF;
id.hasClutShift = gstate.getClutIndexShift() != 0;
id.hasClutOffset = gstate.getClutIndexStartPos() != 0;
id.cached.clutFormat = gstate.clutformat;
}
id.clampS = gstate.isTexCoordClampedS();
@@ -424,6 +430,9 @@ void ComputeSamplerID(SamplerID *id_out) {
if (id.texFunc > GE_TEXFUNC_ADD)
id.texFunc = GE_TEXFUNC_ADD;
if (id.texFunc == GE_TEXFUNC_BLEND)
id.cached.texBlendColor = gstate.getTextureEnvColRGB();
*id_out = id;
}
+9
View File
@@ -166,6 +166,15 @@ struct SamplerID {
SamplerID() : fullKey(0) {
}
struct {
struct {
uint16_t w;
uint16_t h;
} sizes[8];
uint32_t texBlendColor;
uint32_t clutFormat;
} cached;
union {
uint32_t fullKey;
struct {
+18 -123
View File
@@ -127,6 +127,8 @@ void ComputeRasterizerState(RasterizerState *state) {
else
state->texptr[i] = nullptr;
}
state->textureLodSlope = gstate.getTextureLodSlope();
}
state->texLevelMode = gstate.getTexLevelMode();
@@ -138,6 +140,9 @@ void ComputeRasterizerState(RasterizerState *state) {
state->magFilt = gstate.isMagnifyFilteringEnabled();
state->antialiasLines = gstate.isAntiAliasEnabled();
state->screenOffsetX = gstate.getOffsetX16();
state->screenOffsetY = gstate.getOffsetY16();
#if defined(SOFTGPU_MEMORY_TAGGING_DETAILED) || defined(SOFTGPU_MEMORY_TAGGING_BASIC)
DisplayList currentList{};
if (gpuDebug)
@@ -224,114 +229,6 @@ static inline bool IsRightSideOrFlatBottomLine(const Vec2<int>& vertex, const Ve
}
}
Vec4IntResult SOFTRAST_CALL GetTextureFunctionOutput(Vec4IntArg prim_color_in, Vec4IntArg texcolor_in) {
const Vec4<int> prim_color = prim_color_in;
const Vec4<int> texcolor = texcolor_in;
Vec3<int> out_rgb;
int out_a;
bool rgba = gstate.isTextureAlphaUsed();
switch (gstate.getTextureFunction()) {
case GE_TEXFUNC_MODULATE:
{
#if defined(_M_SSE)
// Modulate weights slightly on the tex color, by adding one to prim and dividing by 256.
const __m128i p = _mm_slli_epi16(_mm_packs_epi32(prim_color.ivec, prim_color.ivec), 4);
const __m128i pboost = _mm_add_epi16(p, _mm_set1_epi16(1 << 4));
__m128i t = _mm_slli_epi16(_mm_packs_epi32(texcolor.ivec, texcolor.ivec), 4);
if (gstate.isColorDoublingEnabled()) {
const __m128i amask = _mm_set_epi16(-1, 0, 0, 0, -1, 0, 0, 0);
const __m128i a = _mm_and_si128(t, amask);
const __m128i rgb = _mm_andnot_si128(amask, t);
t = _mm_or_si128(_mm_slli_epi16(rgb, 1), a);
}
const __m128i b = _mm_mulhi_epi16(pboost, t);
out_rgb.ivec = _mm_unpacklo_epi16(b, _mm_setzero_si128());
if (rgba) {
return ToVec4IntResult(Vec4<int>(out_rgb.ivec));
} else {
out_a = prim_color.a();
}
#else
if (gstate.isColorDoublingEnabled()) {
out_rgb = ((prim_color.rgb() + Vec3<int>::AssignToAll(1)) * texcolor.rgb() * 2) / 256;
} else {
out_rgb = (prim_color.rgb() + Vec3<int>::AssignToAll(1)) * texcolor.rgb() / 256;
}
out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a();
#endif
break;
}
case GE_TEXFUNC_DECAL:
{
if (rgba) {
int t = texcolor.a();
int invt = 255 - t;
// Both colors are boosted here, making the alpha have more weight.
Vec3<int> one = Vec3<int>::AssignToAll(1);
out_rgb = ((prim_color.rgb() + one) * invt + (texcolor.rgb() + one) * t);
// Keep the bits of accuracy when doubling.
if (gstate.isColorDoublingEnabled())
out_rgb /= 128;
else
out_rgb /= 256;
} else {
if (gstate.isColorDoublingEnabled())
out_rgb = texcolor.rgb() * 2;
else
out_rgb = texcolor.rgb();
}
out_a = prim_color.a();
break;
}
case GE_TEXFUNC_BLEND:
{
const Vec3<int> const255(255, 255, 255);
const Vec3<int> texenv(gstate.getTextureEnvColR(), gstate.getTextureEnvColG(), gstate.getTextureEnvColB());
// Unlike the others (and even alpha), this one simply always rounds up.
const Vec3<int> roundup = Vec3<int>::AssignToAll(255);
out_rgb = ((const255 - texcolor.rgb()) * prim_color.rgb() + texcolor.rgb() * texenv + roundup);
// Must divide by less to keep the precision for doubling to be accurate.
if (gstate.isColorDoublingEnabled())
out_rgb /= 128;
else
out_rgb /= 256;
out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a();
break;
}
case GE_TEXFUNC_REPLACE:
out_rgb = texcolor.rgb();
// Doubling even happens for replace.
if (gstate.isColorDoublingEnabled())
out_rgb *= 2;
out_a = (rgba) ? texcolor.a() : prim_color.a();
break;
case GE_TEXFUNC_ADD:
case GE_TEXFUNC_UNKNOWN1:
case GE_TEXFUNC_UNKNOWN2:
case GE_TEXFUNC_UNKNOWN3:
// Don't need to clamp afterward, we always clamp before tests.
out_rgb = prim_color.rgb() + texcolor.rgb();
if (gstate.isColorDoublingEnabled())
out_rgb *= 2;
// Alpha is still blended the common way.
out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a();
break;
}
return ToVec4IntResult(Vec4<int>(out_rgb, out_a));
}
static inline Vec3<int> GetSourceFactor(GEBlendSrcFactor factor, const Vec4<int> &source, const Vec4<int> &dst, uint32_t fix) {
switch (factor) {
case GE_SRCBLEND_DSTCOLOR:
@@ -525,9 +422,9 @@ static inline Vec4IntResult SOFTRAST_CALL ApplyTexturing(float s, float t, int x
const int *bufw0 = &state.texbufw[texlevel];
if (!bilinear) {
return state.nearest(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel);
return state.nearest(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel, state.samplerID);
}
return state.linear(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel);
return state.linear(s, t, x, y, prim_color, tptr0, bufw0, texlevel, frac_texlevel, state.samplerID);
}
static inline Vec4IntResult SOFTRAST_CALL ApplyTexturingSingle(float s, float t, int x, int y, Vec4IntArg prim_color, int texlevel, int frac_texlevel, bool bilinear, const RasterizerState &state) {
@@ -561,7 +458,7 @@ static inline void CalculateSamplingParams(const float ds, const float dt, const
break;
case GE_TEXLEVEL_MODE_SLOPE:
// This is always offset by an extra texlevel.
detail = 1 * 16 + TexLog2(gstate.getTextureLodSlope());
detail = 1 * 16 + TexLog2(state.textureLodSlope);
break;
case GE_TEXLEVEL_MODE_CONST:
default:
@@ -823,7 +720,7 @@ void DrawTriangleSlice(
Vec4<int> w1 = w1_base;
Vec4<int> w2 = w2_base;
DrawingCoords p = TransformUnit::ScreenToDrawing(ScreenCoords(minX, curY, 0));
DrawingCoords p = TransformUnit::ScreenToDrawing(minX, curY, state.screenOffsetX, state.screenOffsetY);
int64_t rowMinX = minX, rowMaxX = maxX;
e0.NarrowMinMaxX(w0, minX, rowMinX, rowMaxX);
@@ -950,8 +847,8 @@ void DrawTriangleSlice(
#if !defined(SOFTGPU_MEMORY_TAGGING_DETAILED) && defined(SOFTGPU_MEMORY_TAGGING_BASIC)
for (int y = minY; y <= maxY; y += 16) {
DrawingCoords p = TransformUnit::ScreenToDrawing(ScreenCoords(minX, y, 0));
DrawingCoords pend = TransformUnit::ScreenToDrawing(ScreenCoords(maxX, y, 0));
DrawingCoords p = TransformUnit::ScreenToDrawing(minX, y, state.screenOffsetX, state.screenOffsetY);
DrawingCoords pend = TransformUnit::ScreenToDrawing(maxX, y, state.screenOffsetX, state.screenOffsetY);
uint32_t row = gstate.getFrameBufAddress() + p.y * pixelID.cached.framebufStride * bpp;
NotifyMemInfo(MemBlockFlags::WRITE, row + p.x * bpp, (pend.x - p.x) * bpp, tag.c_str(), tag.size());
@@ -1004,7 +901,7 @@ void DrawPoint(const VertexData &v0, const BinCoords &range, const RasterizerSta
if (!pixelID.clearMode)
prim_color += Vec4<int>(sec_color, 0);
DrawingCoords p = TransformUnit::ScreenToDrawing(pos);
DrawingCoords p = TransformUnit::ScreenToDrawing(pos, state.screenOffsetX, state.screenOffsetY);
u16 z = pos.z;
u8 fog = 255;
@@ -1031,8 +928,8 @@ void DrawPoint(const VertexData &v0, const BinCoords &range, const RasterizerSta
}
void ClearRectangle(const VertexData &v0, const VertexData &v1, const BinCoords &range, const RasterizerState &state) {
DrawingCoords pprime = TransformUnit::ScreenToDrawing(ScreenCoords(range.x1, range.y1, 0));
DrawingCoords pend = TransformUnit::ScreenToDrawing(ScreenCoords(range.x2, range.y2, 0));
DrawingCoords pprime = TransformUnit::ScreenToDrawing(range.x1, range.y1, state.screenOffsetX, state.screenOffsetY);
DrawingCoords pend = TransformUnit::ScreenToDrawing(range.x2, range.y2, state.screenOffsetX, state.screenOffsetY);
auto &pixelID = state.pixelID;
auto &samplerID = state.samplerID;
@@ -1296,8 +1193,8 @@ void DrawLine(const VertexData &v0, const VertexData &v1, const BinCoords &range
CalculateSamplingParams(ds, dt, state, texLevel, texLevelFrac, texBilinear);
if (state.antialiasLines) {
// TODO: This is a niave and wrong implementation.
DrawingCoords p0 = TransformUnit::ScreenToDrawing(ScreenCoords((int)x, (int)y, (int)z));
// TODO: This is a naive and wrong implementation.
DrawingCoords p0 = TransformUnit::ScreenToDrawing(x, y, state.screenOffsetX, state.screenOffsetY);
s = ((float)p0.x + xinc / 32.0f) / 512.0f;
t = ((float)p0.y + yinc / 32.0f) / 512.0f;
@@ -1311,10 +1208,8 @@ void DrawLine(const VertexData &v0, const VertexData &v1, const BinCoords &range
if (!pixelID.clearMode)
prim_color += Vec4<int>(sec_color, 0);
ScreenCoords pprime = ScreenCoords((int)x, (int)y, (int)z);
PROFILE_THIS_SCOPE("draw_px");
DrawingCoords p = TransformUnit::ScreenToDrawing(pprime);
DrawingCoords p = TransformUnit::ScreenToDrawing(x, y, state.screenOffsetX, state.screenOffsetY);
state.drawPixel(p.x, p.y, z, fog, ToVec4IntArg(prim_color), pixelID);
#if defined(SOFTGPU_MEMORY_TAGGING_DETAILED) || defined(SOFTGPU_MEMORY_TAGGING_BASIC)
@@ -1376,7 +1271,7 @@ bool GetCurrentTexture(GPUDebugBuffer &buffer, int level)
u32 *row = (u32 *)buffer.GetData();
for (int y = 0; y < h; ++y) {
for (int x = 0; x < w; ++x) {
row[x] = Vec4<int>(sampler(x, y, texptr, texbufw, level)).ToRGBA();
row[x] = Vec4<int>(sampler(x, y, texptr, texbufw, level, id)).ToRGBA();
}
row += w;
}
+3 -1
View File
@@ -41,6 +41,9 @@ struct RasterizerState {
Sampler::NearestFunc nearest;
int texbufw[8]{};
u8 *texptr[8]{};
float textureLodSlope;
int screenOffsetX;
int screenOffsetY;
struct {
uint8_t maxTexLevel : 3;
@@ -77,6 +80,5 @@ bool GetCurrentTexture(GPUDebugBuffer &buffer, int level);
// Shared functions with RasterizerRectangle.cpp
Vec3<int> AlphaBlendingResult(const PixelFuncID &pixelID, const Vec4<int> &source, const Vec4<int> &dst);
Vec4IntResult SOFTRAST_CALL GetTextureFunctionOutput(Vec4IntArg prim_color, Vec4IntArg texcolor);
} // namespace Rasterizer
+8 -8
View File
@@ -87,14 +87,14 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran
auto &pixelID = state.pixelID;
auto &samplerID = state.samplerID;
DrawingCoords pos0 = TransformUnit::ScreenToDrawing(v0.screenpos);
DrawingCoords pos0 = TransformUnit::ScreenToDrawing(v0.screenpos, state.screenOffsetX, state.screenOffsetY);
// Include the ending pixel based on its center, not start.
DrawingCoords pos1 = TransformUnit::ScreenToDrawing(v1.screenpos + ScreenCoords(7, 7, 0));
DrawingCoords pos1 = TransformUnit::ScreenToDrawing(v1.screenpos + ScreenCoords(7, 7, 0), state.screenOffsetX, state.screenOffsetY);
DrawingCoords scissorTL = TransformUnit::ScreenToDrawing(ScreenCoords(range.x1, range.y1, 0));
DrawingCoords scissorBR = TransformUnit::ScreenToDrawing(ScreenCoords(range.x2, range.y2, 0));
DrawingCoords scissorTL = TransformUnit::ScreenToDrawing(range.x1, range.y1, state.screenOffsetX, state.screenOffsetY);
DrawingCoords scissorBR = TransformUnit::ScreenToDrawing(range.x2, range.y2, state.screenOffsetX, state.screenOffsetY);
int z = pos0.z;
int z = v1.screenpos.z;
int fog = 255;
bool isWhite = v1.color0 == Vec4<int>(255, 255, 255, 255);
@@ -146,7 +146,7 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran
int s = s_start;
u16 *pixel = fb.Get16Ptr(pos0.x, y, pixelID.cached.framebufStride);
for (int x = pos0.x; x < pos1.x; x++) {
u32 tex_color = Vec4<int>(fetchFunc(s, t, texptr, texbufw, 0)).ToRGBA();
u32 tex_color = Vec4<int>(fetchFunc(s, t, texptr, texbufw, 0, state.samplerID)).ToRGBA();
if (tex_color & 0xFF000000) {
DrawSinglePixel5551(pixel, tex_color, pixelID);
}
@@ -162,7 +162,7 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran
u16 *pixel = fb.Get16Ptr(pos0.x, y, pixelID.cached.framebufStride);
for (int x = pos0.x; x < pos1.x; x++) {
Vec4<int> prim_color = v1.color0;
Vec4<int> tex_color = fetchFunc(s, t, texptr, texbufw, 0);
Vec4<int> tex_color = fetchFunc(s, t, texptr, texbufw, 0, state.samplerID);
prim_color = Vec4<int>(ModulateRGBA(ToVec4IntArg(prim_color), ToVec4IntArg(tex_color), state.samplerID));
if (prim_color.a() > 0) {
DrawSinglePixel5551(pixel, prim_color.ToRGBA(), pixelID);
@@ -187,7 +187,7 @@ void DrawSprite(const VertexData &v0, const VertexData &v1, const BinCoords &ran
float s = sf_start;
// Not really that fast but faster than triangle.
for (int x = pos0.x; x < pos1.x; x++) {
Vec4<int> prim_color = state.nearest(s, t, xoff, yoff, ToVec4IntArg(v1.color0), &texptr, &texbufw, 0, 0);
Vec4<int> prim_color = state.nearest(s, t, xoff, yoff, ToVec4IntArg(v1.color0), &texptr, &texbufw, 0, 0, state.samplerID);
state.drawPixel(x, y, z, 255, ToVec4IntArg(prim_color), pixelID);
s += dsf;
}
+7 -8
View File
@@ -106,14 +106,13 @@ struct RegCache {
VEC_INDEX = 0x0005,
GEN_SRC_ALPHA = 0x0100,
GEN_GSTATE = 0x0101,
GEN_ID = 0x0102,
GEN_CONST_BASE = 0x0103,
GEN_STENCIL = 0x0104,
GEN_COLOR_OFF = 0x0105,
GEN_DEPTH_OFF = 0x0106,
GEN_RESULT = 0x0107,
GEN_SHIFTVAL = 0x0108,
GEN_ID = 0x0101,
GEN_CONST_BASE = 0x0102,
GEN_STENCIL = 0x0103,
GEN_COLOR_OFF = 0x0104,
GEN_DEPTH_OFF = 0x0105,
GEN_RESULT = 0x0106,
GEN_SHIFTVAL = 0x0107,
GEN_ARG_X = 0x0180,
GEN_ARG_Y = 0x0181,
+173 -59
View File
@@ -39,9 +39,9 @@ extern u32 clut[4096];
namespace Sampler {
static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac);
static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac);
static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level);
static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID);
static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID);
static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level, const SamplerID &samplerID);
std::mutex jitCacheLock;
SamplerJitCache *jitCache = nullptr;
@@ -242,10 +242,9 @@ FetchFunc SamplerJitCache::GetFetch(const SamplerID &id) {
return nullptr;
}
template <unsigned int texel_size_bits>
static inline int GetPixelDataOffset(unsigned int row_pitch_pixels, unsigned int u, unsigned int v)
{
if (!gstate.isTextureSwizzled())
template <uint32_t texel_size_bits>
static inline int GetPixelDataOffset(uint32_t row_pitch_pixels, uint32_t u, uint32_t v, bool swizzled) {
if (!swizzled)
return (v * (row_pitch_pixels * texel_size_bits >> 3)) + (u * texel_size_bits >> 3);
const int tile_size_bits = 32;
@@ -263,12 +262,10 @@ static inline int GetPixelDataOffset(unsigned int row_pitch_pixels, unsigned int
return tile_idx * (tile_size_bits / 8) + ((u % texels_per_tile) * texel_size_bits) / 8;
}
static inline u32 LookupColor(unsigned int index, unsigned int level)
{
const bool mipmapShareClut = gstate.isClutSharedForMipmaps();
const int clutSharingOffset = mipmapShareClut ? 0 : level * 16;
static inline u32 LookupColor(unsigned int index, unsigned int level, const SamplerID &samplerID) {
const int clutSharingOffset = samplerID.useSharedClut ? 0 : level * 16;
switch (gstate.getClutPaletteFormat()) {
switch (samplerID.ClutFmt()) {
case GE_CMODE_16BIT_BGR5650:
return RGB565ToRGBA8888(reinterpret_cast<u16*>(clut)[index + clutSharingOffset]);
@@ -282,11 +279,24 @@ static inline u32 LookupColor(unsigned int index, unsigned int level)
return clut[index + clutSharingOffset];
default:
ERROR_LOG_REPORT(G3D, "Software: Unsupported palette format: %x", gstate.getClutPaletteFormat());
ERROR_LOG_REPORT(G3D, "Software: Unsupported palette format: %x", samplerID.ClutFmt());
return 0;
}
}
uint32_t TransformClutIndex(uint32_t index, const SamplerID &samplerID) {
if (samplerID.hasClutShift || samplerID.hasClutMask || samplerID.hasClutOffset) {
const uint8_t shift = (samplerID.cached.clutFormat >> 2) & 0x1F;
const uint8_t mask = (samplerID.cached.clutFormat >> 8) & 0xFF;
const uint16_t offset = ((samplerID.cached.clutFormat >> 16) & 0x1F) << 4;
// We need to wrap any entries beyond the first 1024 bytes.
const uint16_t offsetMask = samplerID.ClutFmt() == GE_CMODE_32BIT_ABGR8888 ? 0xFF : 0x1FF;
return ((index >> shift) & mask) | (offset & offsetMask);
}
return index & 0xFF;
}
struct Nearest4 {
alignas(16) u32 v[4];
@@ -296,76 +306,74 @@ struct Nearest4 {
};
template <int N>
inline static Nearest4 SOFTRAST_CALL SampleNearest(const int u[N], const int v[N], const u8 *srcptr, int texbufw, int level) {
inline static Nearest4 SOFTRAST_CALL SampleNearest(const int u[N], const int v[N], const u8 *srcptr, int texbufw, int level, const SamplerID &samplerID) {
Nearest4 res;
if (!srcptr) {
memset(res.v, 0, sizeof(res.v));
return res;
}
GETextureFormat texfmt = gstate.getTextureFormat();
// TODO: Should probably check if textures are aligned properly...
switch (texfmt) {
switch (samplerID.TexFmt()) {
case GE_TFMT_4444:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle);
res.v[i] = RGBA4444ToRGBA8888(*(const u16 *)src);
}
return res;
case GE_TFMT_5551:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle);
res.v[i] = RGBA5551ToRGBA8888(*(const u16 *)src);
}
return res;
case GE_TFMT_5650:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle);
res.v[i] = RGB565ToRGBA8888(*(const u16 *)src);
}
return res;
case GE_TFMT_8888:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i], samplerID.swizzle);
res.v[i] = *(const u32 *)src;
}
return res;
case GE_TFMT_CLUT32:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<32>(texbufw, u[i], v[i], samplerID.swizzle);
u32 val = src[0] + (src[1] << 8) + (src[2] << 16) + (src[3] << 24);
res.v[i] = LookupColor(gstate.transformClutIndex(val), 0);
res.v[i] = LookupColor(TransformClutIndex(val, samplerID), 0, samplerID);
}
return res;
case GE_TFMT_CLUT16:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<16>(texbufw, u[i], v[i], samplerID.swizzle);
u16 val = src[0] + (src[1] << 8);
res.v[i] = LookupColor(gstate.transformClutIndex(val), 0);
res.v[i] = LookupColor(TransformClutIndex(val, samplerID), 0, samplerID);
}
return res;
case GE_TFMT_CLUT8:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<8>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<8>(texbufw, u[i], v[i], samplerID.swizzle);
u8 val = *src;
res.v[i] = LookupColor(gstate.transformClutIndex(val), 0);
res.v[i] = LookupColor(TransformClutIndex(val, samplerID), 0, samplerID);
}
return res;
case GE_TFMT_CLUT4:
for (int i = 0; i < N; ++i) {
const u8 *src = srcptr + GetPixelDataOffset<4>(texbufw, u[i], v[i]);
const u8 *src = srcptr + GetPixelDataOffset<4>(texbufw, u[i], v[i], samplerID.swizzle);
u8 val = (u[i] & 1) ? (src[0] >> 4) : (src[0] & 0xF);
// Only CLUT4 uses separate mipmap palettes.
res.v[i] = LookupColor(gstate.transformClutIndex(val), level);
res.v[i] = LookupColor(TransformClutIndex(val, samplerID), level, samplerID);
}
return res;
@@ -391,7 +399,7 @@ inline static Nearest4 SOFTRAST_CALL SampleNearest(const int u[N], const int v[N
return res;
default:
ERROR_LOG_REPORT(G3D, "Software: Unsupported texture format: %x", texfmt);
ERROR_LOG_REPORT(G3D, "Software: Unsupported texture format: %x", samplerID.TexFmt());
memset(res.v, 0, sizeof(res.v));
return res;
}
@@ -410,8 +418,8 @@ static inline int WrapUV(int v, int height) {
}
template <int N>
static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], const int v[N], int width, int height) {
if (gstate.isTexCoordClampedS()) {
static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], const int v[N], int width, int height, const SamplerID &samplerID) {
if (samplerID.clampS) {
for (int i = 0; i < N; ++i) {
out_u[i] = ClampUV(u[i], width);
}
@@ -420,7 +428,7 @@ static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], c
out_u[i] = WrapUV(u[i], width);
}
}
if (gstate.isTexCoordClampedT()) {
if (samplerID.clampT) {
for (int i = 0; i < N; ++i) {
out_v[i] = ClampUV(v[i], height);
}
@@ -431,9 +439,9 @@ static inline void ApplyTexelClamp(int out_u[N], int out_v[N], const int u[N], c
}
}
static inline void GetTexelCoordinates(int level, float s, float t, int &out_u, int &out_v, int x, int y) {
int width = gstate.getTextureWidth(level);
int height = gstate.getTextureHeight(level);
static inline void GetTexelCoordinates(int level, float s, float t, int &out_u, int &out_v, int x, int y, const SamplerID &samplerID) {
int width = samplerID.cached.sizes[level].w;
int height = samplerID.cached.sizes[level].h;
int base_u = (int)(s * width * 256.0f) + 12 - x;
int base_v = (int)(t * height * 256.0f) + 12 - y;
@@ -441,28 +449,134 @@ static inline void GetTexelCoordinates(int level, float s, float t, int &out_u,
base_u >>= 8;
base_v >>= 8;
ApplyTexelClamp<1>(&out_u, &out_v, &base_u, &base_v, width, height);
ApplyTexelClamp<1>(&out_u, &out_v, &base_u, &base_v, width, height, samplerID);
}
static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac) {
Vec4IntResult SOFTRAST_CALL GetTextureFunctionOutput(Vec4IntArg prim_color_in, Vec4IntArg texcolor_in, const SamplerID &samplerID) {
const Vec4<int> prim_color = prim_color_in;
const Vec4<int> texcolor = texcolor_in;
Vec3<int> out_rgb;
int out_a;
bool rgba = samplerID.useTextureAlpha;
switch (samplerID.TexFunc()) {
case GE_TEXFUNC_MODULATE:
{
#if defined(_M_SSE)
// Modulate weights slightly on the tex color, by adding one to prim and dividing by 256.
const __m128i p = _mm_slli_epi16(_mm_packs_epi32(prim_color.ivec, prim_color.ivec), 4);
const __m128i pboost = _mm_add_epi16(p, _mm_set1_epi16(1 << 4));
__m128i t = _mm_slli_epi16(_mm_packs_epi32(texcolor.ivec, texcolor.ivec), 4);
if (samplerID.useColorDoubling) {
const __m128i amask = _mm_set_epi16(-1, 0, 0, 0, -1, 0, 0, 0);
const __m128i a = _mm_and_si128(t, amask);
const __m128i rgb = _mm_andnot_si128(amask, t);
t = _mm_or_si128(_mm_slli_epi16(rgb, 1), a);
}
const __m128i b = _mm_mulhi_epi16(pboost, t);
out_rgb.ivec = _mm_unpacklo_epi16(b, _mm_setzero_si128());
if (rgba) {
return ToVec4IntResult(Vec4<int>(out_rgb.ivec));
} else {
out_a = prim_color.a();
}
#else
if (samplerID.useColorDoubling) {
out_rgb = ((prim_color.rgb() + Vec3<int>::AssignToAll(1)) * texcolor.rgb() * 2) / 256;
} else {
out_rgb = (prim_color.rgb() + Vec3<int>::AssignToAll(1)) * texcolor.rgb() / 256;
}
out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a();
#endif
break;
}
case GE_TEXFUNC_DECAL:
if (rgba) {
int t = texcolor.a();
int invt = 255 - t;
// Both colors are boosted here, making the alpha have more weight.
Vec3<int> one = Vec3<int>::AssignToAll(1);
out_rgb = ((prim_color.rgb() + one) * invt + (texcolor.rgb() + one) * t);
// Keep the bits of accuracy when doubling.
if (samplerID.useColorDoubling)
out_rgb /= 128;
else
out_rgb /= 256;
} else {
if (samplerID.useColorDoubling)
out_rgb = texcolor.rgb() * 2;
else
out_rgb = texcolor.rgb();
}
out_a = prim_color.a();
break;
case GE_TEXFUNC_BLEND:
{
const Vec3<int> const255(255, 255, 255);
const Vec3<int> texenv = Vec3<int>::FromRGB(samplerID.cached.texBlendColor);
// Unlike the others (and even alpha), this one simply always rounds up.
const Vec3<int> roundup = Vec3<int>::AssignToAll(255);
out_rgb = ((const255 - texcolor.rgb()) * prim_color.rgb() + texcolor.rgb() * texenv + roundup);
// Must divide by less to keep the precision for doubling to be accurate.
if (samplerID.useColorDoubling)
out_rgb /= 128;
else
out_rgb /= 256;
out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a();
break;
}
case GE_TEXFUNC_REPLACE:
out_rgb = texcolor.rgb();
// Doubling even happens for replace.
if (samplerID.useColorDoubling)
out_rgb *= 2;
out_a = (rgba) ? texcolor.a() : prim_color.a();
break;
case GE_TEXFUNC_ADD:
case GE_TEXFUNC_UNKNOWN1:
case GE_TEXFUNC_UNKNOWN2:
case GE_TEXFUNC_UNKNOWN3:
// Don't need to clamp afterward, we always clamp before tests.
out_rgb = prim_color.rgb() + texcolor.rgb();
if (samplerID.useColorDoubling)
out_rgb *= 2;
// Alpha is still blended the common way.
out_a = (rgba) ? ((prim_color.a() + 1) * texcolor.a() / 256) : prim_color.a();
break;
}
return ToVec4IntResult(Vec4<int>(out_rgb, out_a));
}
static Vec4IntResult SOFTRAST_CALL SampleNearest(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID) {
int u, v;
// Nearest filtering only. Round texcoords.
GetTexelCoordinates(level, s, t, u, v, x, y);
Vec4<int> c0 = Vec4<int>::FromRGBA(SampleNearest<1>(&u, &v, tptr[0], bufw[0], level).v[0]);
GetTexelCoordinates(level, s, t, u, v, x, y, samplerID);
Vec4<int> c0 = Vec4<int>::FromRGBA(SampleNearest<1>(&u, &v, tptr[0], bufw[0], level, samplerID).v[0]);
if (levelFrac) {
GetTexelCoordinates(level + 1, s, t, u, v, x, y);
Vec4<int> c1 = Vec4<int>::FromRGBA(SampleNearest<1>(&u, &v, tptr[1], bufw[1], level + 1).v[0]);
GetTexelCoordinates(level + 1, s, t, u, v, x, y, samplerID);
Vec4<int> c1 = Vec4<int>::FromRGBA(SampleNearest<1>(&u, &v, tptr[1], bufw[1], level + 1, samplerID).v[0]);
c0 = (c1 * levelFrac + c0 * (16 - levelFrac)) / 16;
}
return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0));
return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0), samplerID);
}
static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level) {
Nearest4 c = SampleNearest<1>(&u, &v, tptr, bufw, level);
static Vec4IntResult SOFTRAST_CALL SampleFetch(int u, int v, const u8 *tptr, int bufw, int level, const SamplerID &samplerID) {
Nearest4 c = SampleNearest<1>(&u, &v, tptr, bufw, level, samplerID);
return ToVec4IntResult(Vec4<int>::FromRGBA(c.v[0]));
}
@@ -518,33 +632,33 @@ static inline Vec4IntResult SOFTRAST_CALL ApplyTexelClampQuadT(bool clamp, int v
#endif
}
static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadS(int level, float in_s, int &frac_u, int x) {
int width = gstate.getTextureWidth(level);
static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadS(int level, float in_s, int &frac_u, int x, const SamplerID &samplerID) {
int width = samplerID.cached.sizes[level].w;
int base_u = (int)(in_s * width * 256) + 12 - x - 128;
frac_u = (int)(base_u >> 4) & 0x0F;
base_u >>= 8;
// Need to generate and individually wrap/clamp the four sample coordinates. Ugh.
return ApplyTexelClampQuadS(gstate.isTexCoordClampedS(), base_u, width);
return ApplyTexelClampQuadS(samplerID.clampS, base_u, width);
}
static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadT(int level, float in_t, int &frac_v, int y) {
int height = gstate.getTextureHeight(level);
static inline Vec4IntResult SOFTRAST_CALL GetTexelCoordinatesQuadT(int level, float in_t, int &frac_v, int y, const SamplerID &samplerID) {
int height = samplerID.cached.sizes[level].h;
int base_v = (int)(in_t * height * 256) + 12 - y - 128;
frac_v = (int)(base_v >> 4) & 0x0F;
base_v >>= 8;
// Need to generate and individually wrap/clamp the four sample coordinates. Ugh.
return ApplyTexelClampQuadT(gstate.isTexCoordClampedT(), base_v, height);
return ApplyTexelClampQuadT(samplerID.clampT, base_v, height);
}
static Vec4IntResult SOFTRAST_CALL SampleLinearLevel(float s, float t, int x, int y, const u8 *const *tptr, const int *bufw, int texlevel) {
static Vec4IntResult SOFTRAST_CALL SampleLinearLevel(float s, float t, int x, int y, const u8 *const *tptr, const int *bufw, int texlevel, const SamplerID &samplerID) {
int frac_u, frac_v;
const Vec4<int> u = GetTexelCoordinatesQuadS(texlevel, s, frac_u, x);
const Vec4<int> v = GetTexelCoordinatesQuadT(texlevel, t, frac_v, y);
Nearest4 c = SampleNearest<4>(u.AsArray(), v.AsArray(), tptr[0], bufw[0], texlevel);
const Vec4<int> u = GetTexelCoordinatesQuadS(texlevel, s, frac_u, x, samplerID);
const Vec4<int> v = GetTexelCoordinatesQuadT(texlevel, t, frac_v, y, samplerID);
Nearest4 c = SampleNearest<4>(u.AsArray(), v.AsArray(), tptr[0], bufw[0], texlevel, samplerID);
Vec4<int> texcolor_tl = Vec4<int>::FromRGBA(c.v[0]);
Vec4<int> texcolor_tr = Vec4<int>::FromRGBA(c.v[1]);
@@ -555,13 +669,13 @@ static Vec4IntResult SOFTRAST_CALL SampleLinearLevel(float s, float t, int x, in
return ToVec4IntResult((top * (0x10 - frac_v) + bot * frac_v) / (16 * 16));
}
static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int texlevel, int levelFrac) {
Vec4<int> c0 = SampleLinearLevel(s, t, x, y, tptr, bufw, texlevel);
static Vec4IntResult SOFTRAST_CALL SampleLinear(float s, float t, int x, int y, Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int texlevel, int levelFrac, const SamplerID &samplerID) {
Vec4<int> c0 = SampleLinearLevel(s, t, x, y, tptr, bufw, texlevel, samplerID);
if (levelFrac) {
const Vec4<int> c1 = SampleLinearLevel(s, t, x, y, tptr + 1, bufw + 1, texlevel + 1);
const Vec4<int> c1 = SampleLinearLevel(s, t, x, y, tptr + 1, bufw + 1, texlevel + 1, samplerID);
c0 = (c1 * levelFrac + c0 * (16 - levelFrac)) / 16;
}
return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0));
return GetTextureFunctionOutput(prim_color, ToVec4IntArg(c0), samplerID);
}
};
+6 -4
View File
@@ -33,13 +33,13 @@ namespace Sampler {
#pragma GCC diagnostic ignored "-Wignored-attributes"
#endif
typedef Rasterizer::Vec4IntResult(SOFTRAST_CALL *FetchFunc)(int u, int v, const u8 *tptr, int bufw, int level);
typedef Rasterizer::Vec4IntResult(SOFTRAST_CALL *FetchFunc)(int u, int v, const u8 *tptr, int bufw, int level, const SamplerID &samplerID);
FetchFunc GetFetchFunc(SamplerID id);
typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *NearestFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac);
typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *NearestFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID);
NearestFunc GetNearestFunc(SamplerID id);
typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *LinearFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac);
typedef Rasterizer::Vec4IntResult (SOFTRAST_CALL *LinearFunc)(float s, float t, int x, int y, Rasterizer::Vec4IntArg prim_color, const u8 *const *tptr, const int *bufw, int level, int levelFrac, const SamplerID &samplerID);
LinearFunc GetLinearFunc(SamplerID id);
void Init();
@@ -67,7 +67,8 @@ private:
void Describe(const std::string &message);
Rasterizer::RegCache::Reg GetZeroVec();
Rasterizer::RegCache::Reg GetGState();
Rasterizer::RegCache::Reg GetSamplerID();
void UnlockSamplerID(Rasterizer::RegCache::Reg &r);
void WriteConstantPool(const SamplerID &id);
@@ -104,6 +105,7 @@ private:
Arm64Gen::ARM64FloatEmitter fp;
#elif PPSSPP_ARCH(AMD64) || PPSSPP_ARCH(X86)
int stackArgPos_ = 0;
int stackIDOffset_ = -1;
int stackFracUV1Offset_ = 0;
#endif
+67 -40
View File
@@ -41,6 +41,7 @@ FetchFunc SamplerJitCache::CompileFetch(const SamplerID &id) {
RegCache::GEN_ARG_TEXPTR,
RegCache::GEN_ARG_BUFW,
RegCache::GEN_ARG_LEVEL,
RegCache::GEN_ARG_ID,
});
regCache_.ChangeReg(RAX, RegCache::GEN_RESULT);
regCache_.ChangeReg(XMM0, RegCache::VEC_RESULT);
@@ -49,6 +50,15 @@ FetchFunc SamplerJitCache::CompileFetch(const SamplerID &id) {
Describe("Init");
const u8 *start = AlignCode16();
#if PPSSPP_PLATFORM(WINDOWS)
// RET and shadow space.
stackArgPos_ = 8 + 32;
stackIDOffset_ = 8;
#else
stackArgPos_ = 0;
stackIDOffset_ = -1;
#endif
// Early exit on !srcPtr.
FixupBranch zeroSrc;
if (id.hasInvalidPtr) {
@@ -123,6 +133,7 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) {
RegCache::GEN_ARG_BUFW_PTR,
RegCache::GEN_ARG_LEVEL,
RegCache::GEN_ARG_LEVELFRAC,
RegCache::GEN_ARG_ID,
});
#if PPSSPP_PLATFORM(WINDOWS)
@@ -130,11 +141,14 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) {
stackArgPos_ = 8 + 32 + 8;
// Positions: stackArgPos_+0=src, stackArgPos_+8=bufw, stackArgPos_+16=level, stackArgPos_+24=levelFrac
stackIDOffset_ = 32;
// Use the shadow space to save U1/V1. We also use this var for frac U1/V1 in linear.
stackFracUV1Offset_ = -8;
#else
stackArgPos_ = 0;
// This is the only arg that went to the stack.
stackIDOffset_ = 8;
// Use the red zone.
stackFracUV1Offset_ = -8;
#endif
@@ -283,7 +297,7 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) {
regCache_.Unlock(vReg, RegCache::GEN_ARG_V);
regCache_.ForceRetain(RegCache::GEN_ARG_V);
bool hadGState = regCache_.Has(RegCache::GEN_GSTATE);
bool hadId = regCache_.Has(RegCache::GEN_ID);
bool hadZero = regCache_.Has(RegCache::VEC_ZERO);
success = success && Jit_ReadTextureFormat(id);
@@ -299,8 +313,8 @@ NearestFunc SamplerJitCache::CompileNearest(const SamplerID &id) {
regCache_.Unlock(resultReg, RegCache::GEN_RESULT);
// Since we're inside a conditional, make sure these go away if we allocated them.
if (!hadGState && regCache_.Has(RegCache::GEN_GSTATE))
regCache_.ForceRelease(RegCache::GEN_GSTATE);
if (!hadId && regCache_.Has(RegCache::GEN_ID))
regCache_.ForceRelease(RegCache::GEN_ID);
if (!hadZero && regCache_.Has(RegCache::VEC_ZERO))
regCache_.ForceRelease(RegCache::VEC_ZERO);
@@ -493,6 +507,7 @@ LinearFunc SamplerJitCache::CompileLinear(const SamplerID &id) {
RegCache::GEN_ARG_BUFW_PTR,
RegCache::GEN_ARG_LEVEL,
RegCache::GEN_ARG_LEVELFRAC,
RegCache::GEN_ARG_ID,
});
#if PPSSPP_PLATFORM(WINDOWS)
@@ -500,8 +515,10 @@ LinearFunc SamplerJitCache::CompileLinear(const SamplerID &id) {
stackArgPos_ = 8 + 32 + 8;
// Positions: stackArgPos_+0=src, stackArgPos_+8=bufw, stackArgPos_+16=level, stackArgPos_+24=levelFrac
stackIDOffset_ = 32;
#else
stackArgPos_ = 0;
stackIDOffset_ = 8;
#endif
// Start out by saving some registers, since we'll need more.
@@ -931,13 +948,23 @@ RegCache::Reg SamplerJitCache::GetZeroVec() {
return regCache_.Find(RegCache::VEC_ZERO);
}
RegCache::Reg SamplerJitCache::GetGState() {
if (!regCache_.Has(RegCache::GEN_GSTATE)) {
X64Reg r = regCache_.Alloc(RegCache::GEN_GSTATE);
MOV(PTRBITS, R(r), ImmPtr(&gstate.nop));
RegCache::Reg SamplerJitCache::GetSamplerID() {
if (regCache_.Has(RegCache::GEN_ARG_ID))
return regCache_.Find(RegCache::GEN_ARG_ID);
if (!regCache_.Has(RegCache::GEN_ID)) {
X64Reg r = regCache_.Alloc(RegCache::GEN_ID);
_assert_(stackIDOffset_ != -1);
MOV(PTRBITS, R(r), MDisp(RSP, stackArgPos_ + stackIDOffset_));
return r;
}
return regCache_.Find(RegCache::GEN_GSTATE);
return regCache_.Find(RegCache::GEN_ID);
}
void SamplerJitCache::UnlockSamplerID(RegCache::Reg &r) {
if (regCache_.Has(RegCache::GEN_ARG_ID))
regCache_.Unlock(r, RegCache::GEN_ARG_ID);
else
regCache_.Unlock(r, RegCache::GEN_ID);
}
bool SamplerJitCache::Jit_FetchQuad(const SamplerID &id, bool level1) {
@@ -1123,14 +1150,14 @@ bool SamplerJitCache::Jit_TransformClutIndexQuad(const SamplerID &id, int bitsPe
X64Reg indexReg = regCache_.Find(RegCache::VEC_INDEX);
bool maskedIndex = false;
// Okay, first load the actual gstate clutformat bits we'll use.
// Okay, first load the actual samplerID clutformat bits we'll use.
X64Reg formatReg = regCache_.Alloc(RegCache::VEC_TEMP0);
X64Reg gstateReg = GetGState();
X64Reg idReg = GetSamplerID();
if (cpu_info.bAVX2 && !id.hasClutShift)
VPBROADCASTD(128, formatReg, MDisp(gstateReg, offsetof(GPUgstate, clutformat)));
VPBROADCASTD(128, formatReg, MDisp(idReg, offsetof(SamplerID, cached.clutFormat)));
else
MOVD_xmm(formatReg, MDisp(gstateReg, offsetof(GPUgstate, clutformat)));
regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE);
MOVD_xmm(formatReg, MDisp(idReg, offsetof(SamplerID, cached.clutFormat)));
UnlockSamplerID(idReg);
// Shift = (clutformat >> 2) & 0x1F
if (id.hasClutShift) {
@@ -1610,19 +1637,19 @@ bool SamplerJitCache::Jit_ApplyTextureFunc(const SamplerID &id) {
// We divide later.
}
X64Reg gstateReg = GetGState();
X64Reg idReg = GetSamplerID();
X64Reg texEnvReg = regCache_.Alloc(RegCache::VEC_TEMP1);
if (cpu_info.bSSE4_1) {
PMOVZXBW(texEnvReg, MDisp(gstateReg, offsetof(GPUgstate, texenvcolor)));
PMOVZXBW(texEnvReg, MDisp(idReg, offsetof(SamplerID, cached.texBlendColor)));
} else {
MOVD_xmm(texEnvReg, MDisp(gstateReg, offsetof(GPUgstate, texenvcolor)));
MOVD_xmm(texEnvReg, MDisp(idReg, offsetof(SamplerID, cached.texBlendColor)));
X64Reg zeroReg = GetZeroVec();
PUNPCKLBW(texEnvReg, R(zeroReg));
regCache_.Unlock(zeroReg, RegCache::VEC_ZERO);
}
PMULLW(resultReg, R(texEnvReg));
regCache_.Release(texEnvReg, RegCache::VEC_TEMP1);
regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE);
UnlockSamplerID(idReg);
// Add in the prim color side and divide.
PADDW(resultReg, R(tempReg));
@@ -2418,8 +2445,7 @@ bool SamplerJitCache::Jit_GetTexelCoords(const SamplerID &id) {
X64Reg tReg = regCache_.Find(RegCache::VEC_ARG_T);
if (constWidth256f_ == nullptr) {
// We have to figure out levels and the proper width, ugh.
X64Reg shiftReg = regCache_.Find(RegCache::GEN_SHIFTVAL);
X64Reg gstateReg = GetGState();
X64Reg idReg = GetSamplerID();
X64Reg tempReg = regCache_.Alloc(RegCache::GEN_TEMP0);
X64Reg levelReg = INVALID_REG;
@@ -2432,12 +2458,9 @@ bool SamplerJitCache::Jit_GetTexelCoords(const SamplerID &id) {
X64Reg tempVecReg = regCache_.Alloc(RegCache::VEC_TEMP0);
auto loadSizeAndMul = [&](X64Reg dest, bool clamp, bool isY, bool isLevel1) {
int offset = offsetof(GPUgstate, texsize) + (isY ? 1 : 0) + (isLevel1 ? 4 : 0);
// Grab the size, and shift.
MOVZX(32, 8, shiftReg, MComplex(gstateReg, levelReg, SCALE_4, offset));
AND(32, R(shiftReg), Imm8(0x0F));
MOV(32, R(tempReg), Imm32(1));
SHL(32, R(tempReg), R(shiftReg));
int offset = offsetof(SamplerID, cached.sizes[0].w) + (isY ? 2 : 0) + (isLevel1 ? 4 : 0);
// Grab the size, already pre-shifted for us.
MOVZX(32, 16, tempReg, MComplex(idReg, levelReg, SCALE_4, offset));
// Now move to a float, multiplying by 256 meanwhile with a shift.
MOVD_xmm(tempVecReg, R(tempReg));
@@ -2477,8 +2500,7 @@ bool SamplerJitCache::Jit_GetTexelCoords(const SamplerID &id) {
loadSizeAndMul(uReg, id.clampS, false, false);
loadSizeAndMul(vReg, id.clampT, true, false);
regCache_.Unlock(shiftReg, RegCache::GEN_SHIFTVAL);
regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE);
UnlockSamplerID(idReg);
regCache_.Release(tempReg, RegCache::GEN_TEMP0);
regCache_.Unlock(levelReg, RegCache::GEN_ARG_LEVEL);
@@ -2552,7 +2574,7 @@ bool SamplerJitCache::Jit_GetTexelCoordsQuad(const SamplerID &id) {
if (constWidth256f_ == nullptr) {
// We have to figure out levels and the proper width, ugh.
X64Reg gstateReg = GetGState();
X64Reg idReg = GetSamplerID();
X64Reg tempReg = regCache_.Alloc(RegCache::GEN_TEMP0);
X64Reg levelReg = INVALID_REG;
@@ -2567,18 +2589,22 @@ bool SamplerJitCache::Jit_GetTexelCoordsQuad(const SamplerID &id) {
}
// This will load the current and next level's sizes.
MOV(64, R(tempReg), MComplex(gstateReg, levelReg, SCALE_4, offsetof(GPUgstate, texsize)));
regCache_.Unlock(gstateReg, RegCache::GEN_GSTATE);
MOV(64, R(tempReg), MComplex(idReg, levelReg, SCALE_4, offsetof(SamplerID, cached.sizes[0].w)));
UnlockSamplerID(idReg);
X64Reg tempVecReg = regCache_.Alloc(RegCache::VEC_TEMP0);
auto loadSizeAndMul = [&](X64Reg dest, X64Reg size, bool isY, bool isLevel1) {
// Grab the size and mask out 4 bits using walls.
MOVQ_xmm(tempVecReg, R(tempReg));
PSLLQ(tempVecReg, 60 - (isY ? 8 : 0) - (isLevel1 ? 32 : 0));
PSRLQ(tempVecReg, 60);
MOVDQA(size, M(constOnes32_));
PSLLD(size, R(tempVecReg));
// Grab the size and mask out 16 bits using walls.
MOVQ_xmm(size, R(tempReg));
if (cpu_info.bSSE4_1) {
int lane = (isY ? 1 : 0) + (isLevel1 ? 2 : 0);
PMOVZXWD(size, R(size));
PSHUFD(size, R(size), _MM_SHUFFLE(lane, lane, lane, lane));
} else {
PSLLQ(size, 48 - (isY ? 16 : 0) - (isLevel1 ? 32 : 0));
PSRLQ(size, 48);
PSHUFD(size, R(size), _MM_SHUFFLE(0, 0, 0, 0));
}
if (cpu_info.bAVX) {
VPSLLD(128, tempVecReg, size, 8);
@@ -3354,8 +3380,9 @@ bool SamplerJitCache::Jit_TransformClutIndex(const SamplerID &id, int bitsPerInd
_assert_msg_(hasRCX, "Could not obtain RCX, locked?");
X64Reg temp1Reg = regCache_.Alloc(RegCache::GEN_TEMP1);
MOV(PTRBITS, R(temp1Reg), ImmPtr(&gstate.clutformat));
MOV(32, R(temp1Reg), MatR(temp1Reg));
X64Reg idReg = GetSamplerID();
MOV(32, R(temp1Reg), MDisp(idReg, offsetof(SamplerID, cached.clutFormat)));
UnlockSamplerID(idReg);
X64Reg resultReg = regCache_.Find(RegCache::GEN_RESULT);
@@ -3418,7 +3445,7 @@ bool SamplerJitCache::Jit_ReadClutColor(const SamplerID &id) {
} else {
#if PPSSPP_PLATFORM(WINDOWS)
// The argument was saved on the stack.
MOV(32, R(temp2Reg), MDisp(RSP, 40));
MOV(32, R(temp2Reg), MDisp(RSP, stackArgPos_));
#else
_assert_(false);
#endif
+4 -15
View File
@@ -133,22 +133,11 @@ ScreenCoords TransformUnit::ClipToScreen(const ClipCoords& coords)
return ClipToScreenInternal(coords, nullptr);
}
DrawingCoords TransformUnit::ScreenToDrawing(const ScreenCoords& coords)
{
DrawingCoords ret;
// TODO: What to do when offset > coord?
ret.x = ((s32)coords.x - gstate.getOffsetX16()) / 16;
ret.y = ((s32)coords.y - gstate.getOffsetY16()) / 16;
ret.z = coords.z;
return ret;
}
ScreenCoords TransformUnit::DrawingToScreen(const DrawingCoords& coords)
{
ScreenCoords TransformUnit::DrawingToScreen(const DrawingCoords &coords, u16 z) {
ScreenCoords ret;
ret.x = (u32)coords.x * 16 + gstate.getOffsetX16();
ret.y = (u32)coords.y * 16 + gstate.getOffsetY16();
ret.z = coords.z;
ret.z = z;
return ret;
}
@@ -708,7 +697,7 @@ bool TransformUnit::GetCurrentSimpleVertices(int count, std::vector<GPUDebugVert
float clipPos[4];
Vec3ByMatrix44(clipPos, vert.pos.AsArray(), worldviewproj);
ScreenCoords screenPos = ClipToScreen(clipPos);
DrawingCoords drawPos = ScreenToDrawing(screenPos);
DrawingCoords drawPos = ScreenToDrawing(screenPos, gstate.getOffsetX16(), gstate.getOffsetY16());
if (gstate.vertType & GE_VTYPE_TC_MASK) {
vertices[i].u = vert.uv[0] * (float)gstate.getTextureWidth(0);
@@ -719,7 +708,7 @@ bool TransformUnit::GetCurrentSimpleVertices(int count, std::vector<GPUDebugVert
}
vertices[i].x = drawPos.x;
vertices[i].y = drawPos.y;
vertices[i].z = drawPos.z;
vertices[i].z = screenPos.z;
}
if (gstate.vertType & GE_VTYPE_COL_MASK) {
+13 -18
View File
@@ -60,26 +60,12 @@ struct ScreenCoords
}
};
struct DrawingCoords
{
struct DrawingCoords {
DrawingCoords() {}
DrawingCoords(s16 x, s16 y, u16 z) : x(x), y(y), z(z) {}
DrawingCoords(s16 x, s16 y) : x(x), y(y) {}
s16 x;
s16 y;
u16 z;
Vec2<s16> xy() const { return Vec2<s16>(x, y); }
DrawingCoords operator * (const float t) const
{
return DrawingCoords((s16)(x * t), (s16)(y * t), (u16)(z * t));
}
DrawingCoords operator + (const DrawingCoords& oth) const
{
return DrawingCoords(x + oth.x, y + oth.y, z + oth.z);
}
};
struct VertexData {
@@ -116,8 +102,17 @@ public:
static ViewCoords WorldToView(const WorldCoords& coords);
static ClipCoords ViewToClip(const ViewCoords& coords);
static ScreenCoords ClipToScreen(const ClipCoords& coords);
static DrawingCoords ScreenToDrawing(const ScreenCoords& coords);
static ScreenCoords DrawingToScreen(const DrawingCoords& coords);
static inline DrawingCoords ScreenToDrawing(int x, int y, int offsetX, int offsetY) {
DrawingCoords ret;
// When offset > coord, it correctly goes negative and force-scissors.
ret.x = (x - offsetX) / 16;
ret.y = (y - offsetY) / 16;
return ret;
}
static inline DrawingCoords ScreenToDrawing(const ScreenCoords &coords, int offsetX, int offsetY) {
return ScreenToDrawing(coords.x, coords.y, offsetX, offsetY);
}
static ScreenCoords DrawingToScreen(const DrawingCoords &coords, u16 z);
void SubmitPrimitive(void* vertices, void* indices, GEPrimitiveType prim_type, int vertex_count, u32 vertex_type, int *bytesRead, SoftwareDrawEngine *drawEngine);