Actually avoid looking up the vertex decoder more than once

This commit is contained in:
Henrik Rydgård committed 2024-12-17 22:42:07 +01:00
1 parent 9da80ac8d5
commit 0b06cd1379
4 files changed
+17 -27

No files matched your search

+5 -15
View File
@@ -235,7 +235,7 @@ void DrawEngineCommon::UpdatePlanes() {
// - Less accurate, but..
// - Only requires six plane evaluations then.
bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, u32 vertType) {
bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, VertexDecoder *dec, u32 vertType) {
// Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws.
if (vertexCount > 1024) {
return true;
@@ -285,7 +285,6 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
// TODO: Avoid normalization if just plain skinning.
// Force software skinning.
const u32 vertTypeID = GetVertTypeID(vertType, gstate.getUVGenMode(), true);
VertexDecoder *dec = GetVertexDecoder(vertTypeID);
::NormalizeVertices(corners, temp_buffer, (const u8 *)vdata, indexLowerBound, indexUpperBound, dec, vertType);
IndexConverter conv(vertType, inds);
for (int i = 0; i < vertexCount; i++) {
@@ -295,7 +294,6 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
}
} else {
// Simple, most common case.
VertexDecoder *dec = GetVertexDecoder(vertType);
int stride = dec->VertexSize();
int offset = dec->posoff;
switch (vertType & GE_VTYPE_POS_MASK) {
@@ -372,7 +370,7 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
}
// NOTE: This doesn't handle through-mode, indexing, morph, or skinning.
bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u32 vertType) {
bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, VertexDecoder *dec, u32 vertType) {
SimpleVertex *corners = (SimpleVertex *)(decoded_ + 65536 * 12);
float *verts = (float *)(decoded_ + 65536 * 18);
@@ -395,7 +393,6 @@ bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u
return true;
// Simple, most common case.
VertexDecoder *dec = GetVertexDecoder(vertType);
int stride = dec->VertexSize();
int offset = dec->posoff;
int vertStride = 3;
@@ -546,7 +543,7 @@ bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u
// 2D bounding box test against scissor. No indexing yet.
// Only supports non-indexed draws with float positions.
bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount, u32 vertType) {
bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount, VertexDecoder *dec, u32 vertType) {
// Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws.
if (vertexCount > 16) {
return true;
@@ -563,7 +560,6 @@ bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount
// and a large vertex format.
u8 *temp_buffer = decoded_ + 65536 * 24;
// Simple, most common case.
VertexDecoder *dec = GetVertexDecoder(vertType);
int stride = dec->VertexSize();
int offset = dec->posoff;
@@ -801,7 +797,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *
return cmd - start;
}
void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, u32 vertTypeID, int *bytesRead) {
void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, VertexDecoder *dec, u32 vertTypeID, int *bytesRead) {
if (!indexGen.PrimCompatible(prevPrim_, prim)) {
Flush();
}
@@ -817,13 +813,7 @@ void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, u32 vertT
prevPrim_ = prim;
}
// If vtype has changed, setup the vertex decoder.
if (vertTypeID != lastVType_ || !dec_) {
dec_ = GetVertexDecoder(vertTypeID);
lastVType_ = vertTypeID;
}
*bytesRead = vertexCount * dec_->VertexSize();
*bytesRead = vertexCount * dec->VertexSize();
}
// vertTypeID is the vertex type but with the UVGen mode smashed into the top bits.
+4 -4
View File
@@ -101,12 +101,12 @@ public:
virtual void DispatchSubmitImm(GEPrimitiveType prim, TransformedVertex *buffer, int vertexCount, int cullMode, bool continuation);
bool TestBoundingBox(const void *control_points, const void *inds, int vertexCount, u32 vertType);
bool TestBoundingBox(const void *control_points, const void *inds, int vertexCount, VertexDecoder *dec, u32 vertType);
// This is a less accurate version of TestBoundingBox, but faster. Can have more false positives.
// Doesn't support indexing.
bool TestBoundingBoxFast(const void *control_points, int vertexCount, u32 vertType);
bool TestBoundingBoxThrough(const void *vdata, int vertexCount, u32 vertType);
bool TestBoundingBoxFast(const void *control_points, int vertexCount, VertexDecoder *dec, u32 vertType);
bool TestBoundingBoxThrough(const void *vdata, int vertexCount, VertexDecoder *dec, u32 vertType);
void FlushPartialDecode() {
DecodeVerts(dec_, decoded_);
@@ -120,7 +120,7 @@ public:
int ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *stall, u32 vertTypeID, bool clockwise, int *bytesRead, bool isTriangle);
bool SubmitPrim(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, VertexDecoder *dec, u32 vertTypeID, bool clockwise, int *bytesRead);
void SkipPrim(GEPrimitiveType prim, int vertexCount, u32 vertTypeID, int *bytesRead);
void SkipPrim(GEPrimitiveType prim, int vertexCount, VertexDecoder *dec, u32 vertTypeID, int *bytesRead);
template<class Surface>
void SubmitCurve(const void *control_points, const void *indices, Surface &surface, u32 vertType, int *bytesRead, const char *scope);
+3 -3
View File
@@ -1229,12 +1229,12 @@ void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) {
if (count > 0x200) {
// The second to last set of 0x100 is checked (even for odd counts.)
size_t skipSize = (count - 0x200) * dec->VertexSize();
currentList->bboxResult = drawEngineCommon_->TestBoundingBox((const uint8_t *)control_points + skipSize, inds, 0x100, gstate.vertType);
currentList->bboxResult = drawEngineCommon_->TestBoundingBox((const uint8_t *)control_points + skipSize, inds, 0x100, dec, gstate.vertType);
} else if (count > 0x100) {
int checkSize = count - 0x100;
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, checkSize, gstate.vertType);
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, checkSize, dec, gstate.vertType);
} else {
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, count, gstate.vertType);
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, count, dec, gstate.vertType);
}
AdvanceVerts(gstate.vertType, count, bytesRead);
}
+5 -5
View File
@@ -1001,7 +1001,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
// Through mode early-out for simple float 2D draws, like in Fate Extra CCC (very beneficial there due to avoiding texture loads)
if ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_POS_MASK | GE_VTYPE_IDX_MASK)) == (GE_VTYPE_THROUGH_MASK | GE_VTYPE_POS_FLOAT | GE_VTYPE_IDX_NONE)) {
if (!drawEngineCommon_->TestBoundingBoxThrough(verts, count, vertexType)) {
if (!drawEngineCommon_->TestBoundingBoxThrough(verts, count, decoder, vertexType)) {
gpuStats.numCulledDraws++;
int cycles = vertexCost_ * count;
gpuStats.vertexGPUCycles += cycles;
@@ -1027,7 +1027,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
bool passCulling = PASSES_CULLING;
if (!passCulling) {
// Do software culling.
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, vertexType)) {
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, decoder, vertexType)) {
passCulling = true;
} else {
gpuStats.numCulledDraws++;
@@ -1044,7 +1044,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
onePassed = true;
} else {
// Still need to advance bytesRead.
drawEngineCommon_->SkipPrim(prim, count, vertTypeID, &bytesRead);
drawEngineCommon_->SkipPrim(prim, count, decoder, vertTypeID, &bytesRead);
canExtend = false;
}
@@ -1109,7 +1109,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
if (!passCulling) {
// Do software culling.
_dbg_assert_((vertexType & GE_VTYPE_IDX_MASK) == GE_VTYPE_IDX_NONE);
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, vertexType)) {
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, decoder, vertexType)) {
passCulling = true;
} else {
gpuStats.numCulledDraws++;
@@ -1123,7 +1123,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
onePassed = true;
} else {
// Still need to advance bytesRead.
drawEngineCommon_->SkipPrim(newPrim, count, vertTypeID, &bytesRead);
drawEngineCommon_->SkipPrim(newPrim, count, decoder, vertTypeID, &bytesRead);
canExtend = false;
}
AdvanceVerts(vertexType, count, bytesRead);