From d3b6d12067c4e49cffdfde6e744f74347dc8746e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sun, 1 Dec 2024 14:49:10 +0100 Subject: [PATCH] Cull through-mode 2D draws against scissor rectangle Helps texture replacement load performance in Fate Extra CCC (it does a lot of off-screen drawing), and may help in other situations too. --- GPU/Common/DrawEngineCommon.cpp | 65 +++++++++++++++++++++++++++++++++ GPU/Common/DrawEngineCommon.h | 1 + GPU/GPUCommonHW.cpp | 12 ++++++ 3 files changed, 78 insertions(+) diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 0f84d22b15..d78bdfc9f4 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -565,6 +565,71 @@ bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u return true; } +// 2D bounding box test against scissor. No indexing yet. +// Only supports non-indexed draws with float positions. +bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount, u32 vertType) { + // Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws. + if (vertexCount > 16) { + return true; + } + + float *verts = (float *)(decoded_ + 65536 * 18); + + // Although this may lead to drawing that shouldn't happen, the viewport is more complex on VR. + // Let's always say objects are within bounds. + if (gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) + return true; + + // Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder + // and a large vertex format. + u8 *temp_buffer = decoded_ + 65536 * 24; + // Simple, most common case. + VertexDecoder *dec = GetVertexDecoder(vertType); + int stride = dec->VertexSize(); + int offset = dec->posoff; + switch (vertType & GE_VTYPE_POS_MASK) { + case GE_VTYPE_POS_FLOAT: + { + for (int i = 0; i < vertexCount; i++) { + memcpy(&verts[i * 3], (const u8 *)vdata + stride * i + offset, sizeof(float) * 3); + } + break; + } + default: + _dbg_assert_(false); + } + + bool allOutsideLeft = true; + bool allOutsideTop = true; + bool allOutsideRight = true; + bool allOutsideBottom = true; + const float left = gstate.getScissorX1(); + const float top = gstate.getScissorY1(); + const float right = gstate.getScissorX2(); + const float bottom = gstate.getScissorY2(); + for (int i = 0; i < vertexCount; i++) { + const float *pos = verts + i * 3; + float x = pos[0]; + float y = pos[1]; + if (x >= left) { + allOutsideLeft = false; + } + if (x <= right) { + allOutsideRight = false; + } + if (y >= top) { + allOutsideTop = false; + } + if (y <= bottom) { + allOutsideBottom = false; + } + } + if (allOutsideLeft || allOutsideTop || allOutsideRight || allOutsideBottom) { + return false; + } + return true; +} + // TODO: This probably is not the best interface. bool DrawEngineCommon::GetCurrentSimpleVertices(int count, std::vector &vertices, std::vector &indices) { // This is always for the current vertices. diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index 1d9532155b..7ec1a92a78 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -107,6 +107,7 @@ public: // This is a less accurate version of TestBoundingBox, but faster. Can have more false positives. // Doesn't support indexing. bool TestBoundingBoxFast(const void *control_points, int vertexCount, u32 vertType); + bool TestBoundingBoxThrough(const void *vdata, int vertexCount, u32 vertType); void FlushSkin() { bool applySkin = (lastVType_ & GE_VTYPE_WEIGHT_MASK) && decOptions_.applySkinInDecode; diff --git a/GPU/GPUCommonHW.cpp b/GPU/GPUCommonHW.cpp index 678ec52428..4616b34b8a 100644 --- a/GPU/GPUCommonHW.cpp +++ b/GPU/GPUCommonHW.cpp @@ -997,6 +997,17 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) { uint32_t vertTypeID = GetVertTypeID(vertexType, gstate.getUVGenMode(), g_Config.bSoftwareSkinning); + // Through mode early-out for simple float 2D draws, like in Fate Extra CCC (very beneficial there due to avoiding texture loads) + if ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_POS_MASK | GE_VTYPE_IDX_MASK)) == (GE_VTYPE_THROUGH_MASK | GE_VTYPE_POS_FLOAT | GE_VTYPE_IDX_NONE)) { + if (!drawEngineCommon_->TestBoundingBoxThrough(verts, count, vertexType)) { + gpuStats.numCulledDraws++; + int cycles = vertexCost_ * count; + gpuStats.vertexGPUCycles += cycles; + cyclesExecuted += cycles; + return; + } + } + #define MAX_CULL_CHECK_COUNT 6 // For now, turn off culling on platforms where we don't have SIMD bounding box tests, like RISC-V. @@ -1039,6 +1050,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) { // Some games rely on this, they don't bother reloading VADDR and IADDR. // The VADDR/IADDR registers are NOT updated. AdvanceVerts(vertexType, count, bytesRead); + int totalVertCount = count; // PRIMs are often followed by more PRIMs. Save some work and submit them immediately.