mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-11 13:36:24 +02:00
Actually avoid looking up the vertex decoder more than once
This commit is contained in:
1 parent
9da80ac8d5
commit
0b06cd1379
4 files changed
+17
-27
No files matched your search
@@ -235,7 +235,7 @@ void DrawEngineCommon::UpdatePlanes() {
|
||||
// - Less accurate, but..
|
||||
// - Only requires six plane evaluations then.
|
||||
|
||||
bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, u32 vertType) {
|
||||
bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, VertexDecoder *dec, u32 vertType) {
|
||||
// Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws.
|
||||
if (vertexCount > 1024) {
|
||||
return true;
|
||||
@@ -285,7 +285,6 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
|
||||
// TODO: Avoid normalization if just plain skinning.
|
||||
// Force software skinning.
|
||||
const u32 vertTypeID = GetVertTypeID(vertType, gstate.getUVGenMode(), true);
|
||||
VertexDecoder *dec = GetVertexDecoder(vertTypeID);
|
||||
::NormalizeVertices(corners, temp_buffer, (const u8 *)vdata, indexLowerBound, indexUpperBound, dec, vertType);
|
||||
IndexConverter conv(vertType, inds);
|
||||
for (int i = 0; i < vertexCount; i++) {
|
||||
@@ -295,7 +294,6 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
|
||||
}
|
||||
} else {
|
||||
// Simple, most common case.
|
||||
VertexDecoder *dec = GetVertexDecoder(vertType);
|
||||
int stride = dec->VertexSize();
|
||||
int offset = dec->posoff;
|
||||
switch (vertType & GE_VTYPE_POS_MASK) {
|
||||
@@ -372,7 +370,7 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
|
||||
}
|
||||
|
||||
// NOTE: This doesn't handle through-mode, indexing, morph, or skinning.
|
||||
bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u32 vertType) {
|
||||
bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, VertexDecoder *dec, u32 vertType) {
|
||||
SimpleVertex *corners = (SimpleVertex *)(decoded_ + 65536 * 12);
|
||||
float *verts = (float *)(decoded_ + 65536 * 18);
|
||||
|
||||
@@ -395,7 +393,6 @@ bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u
|
||||
return true;
|
||||
|
||||
// Simple, most common case.
|
||||
VertexDecoder *dec = GetVertexDecoder(vertType);
|
||||
int stride = dec->VertexSize();
|
||||
int offset = dec->posoff;
|
||||
int vertStride = 3;
|
||||
@@ -546,7 +543,7 @@ bool DrawEngineCommon::TestBoundingBoxFast(const void *vdata, int vertexCount, u
|
||||
|
||||
// 2D bounding box test against scissor. No indexing yet.
|
||||
// Only supports non-indexed draws with float positions.
|
||||
bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount, u32 vertType) {
|
||||
bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount, VertexDecoder *dec, u32 vertType) {
|
||||
// Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws.
|
||||
if (vertexCount > 16) {
|
||||
return true;
|
||||
@@ -563,7 +560,6 @@ bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount
|
||||
// and a large vertex format.
|
||||
u8 *temp_buffer = decoded_ + 65536 * 24;
|
||||
// Simple, most common case.
|
||||
VertexDecoder *dec = GetVertexDecoder(vertType);
|
||||
int stride = dec->VertexSize();
|
||||
int offset = dec->posoff;
|
||||
|
||||
@@ -801,7 +797,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *
|
||||
return cmd - start;
|
||||
}
|
||||
|
||||
void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, u32 vertTypeID, int *bytesRead) {
|
||||
void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, VertexDecoder *dec, u32 vertTypeID, int *bytesRead) {
|
||||
if (!indexGen.PrimCompatible(prevPrim_, prim)) {
|
||||
Flush();
|
||||
}
|
||||
@@ -817,13 +813,7 @@ void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, u32 vertT
|
||||
prevPrim_ = prim;
|
||||
}
|
||||
|
||||
// If vtype has changed, setup the vertex decoder.
|
||||
if (vertTypeID != lastVType_ || !dec_) {
|
||||
dec_ = GetVertexDecoder(vertTypeID);
|
||||
lastVType_ = vertTypeID;
|
||||
}
|
||||
|
||||
*bytesRead = vertexCount * dec_->VertexSize();
|
||||
*bytesRead = vertexCount * dec->VertexSize();
|
||||
}
|
||||
|
||||
// vertTypeID is the vertex type but with the UVGen mode smashed into the top bits.
|
||||
|
||||
@@ -101,12 +101,12 @@ public:
|
||||
|
||||
virtual void DispatchSubmitImm(GEPrimitiveType prim, TransformedVertex *buffer, int vertexCount, int cullMode, bool continuation);
|
||||
|
||||
bool TestBoundingBox(const void *control_points, const void *inds, int vertexCount, u32 vertType);
|
||||
bool TestBoundingBox(const void *control_points, const void *inds, int vertexCount, VertexDecoder *dec, u32 vertType);
|
||||
|
||||
// This is a less accurate version of TestBoundingBox, but faster. Can have more false positives.
|
||||
// Doesn't support indexing.
|
||||
bool TestBoundingBoxFast(const void *control_points, int vertexCount, u32 vertType);
|
||||
bool TestBoundingBoxThrough(const void *vdata, int vertexCount, u32 vertType);
|
||||
bool TestBoundingBoxFast(const void *control_points, int vertexCount, VertexDecoder *dec, u32 vertType);
|
||||
bool TestBoundingBoxThrough(const void *vdata, int vertexCount, VertexDecoder *dec, u32 vertType);
|
||||
|
||||
void FlushPartialDecode() {
|
||||
DecodeVerts(dec_, decoded_);
|
||||
@@ -120,7 +120,7 @@ public:
|
||||
|
||||
int ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *stall, u32 vertTypeID, bool clockwise, int *bytesRead, bool isTriangle);
|
||||
bool SubmitPrim(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, VertexDecoder *dec, u32 vertTypeID, bool clockwise, int *bytesRead);
|
||||
void SkipPrim(GEPrimitiveType prim, int vertexCount, u32 vertTypeID, int *bytesRead);
|
||||
void SkipPrim(GEPrimitiveType prim, int vertexCount, VertexDecoder *dec, u32 vertTypeID, int *bytesRead);
|
||||
|
||||
template<class Surface>
|
||||
void SubmitCurve(const void *control_points, const void *indices, Surface &surface, u32 vertType, int *bytesRead, const char *scope);
|
||||
|
||||
+3
-3
@@ -1229,12 +1229,12 @@ void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) {
|
||||
if (count > 0x200) {
|
||||
// The second to last set of 0x100 is checked (even for odd counts.)
|
||||
size_t skipSize = (count - 0x200) * dec->VertexSize();
|
||||
currentList->bboxResult = drawEngineCommon_->TestBoundingBox((const uint8_t *)control_points + skipSize, inds, 0x100, gstate.vertType);
|
||||
currentList->bboxResult = drawEngineCommon_->TestBoundingBox((const uint8_t *)control_points + skipSize, inds, 0x100, dec, gstate.vertType);
|
||||
} else if (count > 0x100) {
|
||||
int checkSize = count - 0x100;
|
||||
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, checkSize, gstate.vertType);
|
||||
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, checkSize, dec, gstate.vertType);
|
||||
} else {
|
||||
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, count, gstate.vertType);
|
||||
currentList->bboxResult = drawEngineCommon_->TestBoundingBox(control_points, inds, count, dec, gstate.vertType);
|
||||
}
|
||||
AdvanceVerts(gstate.vertType, count, bytesRead);
|
||||
}
|
||||
|
||||
+5
-5
@@ -1001,7 +1001,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
|
||||
// Through mode early-out for simple float 2D draws, like in Fate Extra CCC (very beneficial there due to avoiding texture loads)
|
||||
if ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_POS_MASK | GE_VTYPE_IDX_MASK)) == (GE_VTYPE_THROUGH_MASK | GE_VTYPE_POS_FLOAT | GE_VTYPE_IDX_NONE)) {
|
||||
if (!drawEngineCommon_->TestBoundingBoxThrough(verts, count, vertexType)) {
|
||||
if (!drawEngineCommon_->TestBoundingBoxThrough(verts, count, decoder, vertexType)) {
|
||||
gpuStats.numCulledDraws++;
|
||||
int cycles = vertexCost_ * count;
|
||||
gpuStats.vertexGPUCycles += cycles;
|
||||
@@ -1027,7 +1027,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
bool passCulling = PASSES_CULLING;
|
||||
if (!passCulling) {
|
||||
// Do software culling.
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, vertexType)) {
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, decoder, vertexType)) {
|
||||
passCulling = true;
|
||||
} else {
|
||||
gpuStats.numCulledDraws++;
|
||||
@@ -1044,7 +1044,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
onePassed = true;
|
||||
} else {
|
||||
// Still need to advance bytesRead.
|
||||
drawEngineCommon_->SkipPrim(prim, count, vertTypeID, &bytesRead);
|
||||
drawEngineCommon_->SkipPrim(prim, count, decoder, vertTypeID, &bytesRead);
|
||||
canExtend = false;
|
||||
}
|
||||
|
||||
@@ -1109,7 +1109,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
if (!passCulling) {
|
||||
// Do software culling.
|
||||
_dbg_assert_((vertexType & GE_VTYPE_IDX_MASK) == GE_VTYPE_IDX_NONE);
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, vertexType)) {
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(verts, count, decoder, vertexType)) {
|
||||
passCulling = true;
|
||||
} else {
|
||||
gpuStats.numCulledDraws++;
|
||||
@@ -1123,7 +1123,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
onePassed = true;
|
||||
} else {
|
||||
// Still need to advance bytesRead.
|
||||
drawEngineCommon_->SkipPrim(newPrim, count, vertTypeID, &bytesRead);
|
||||
drawEngineCommon_->SkipPrim(newPrim, count, decoder, vertTypeID, &bytesRead);
|
||||
canExtend = false;
|
||||
}
|
||||
AdvanceVerts(vertexType, count, bytesRead);
|
||||
|
||||
Reference in new issue
Block a user