mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-09-29 15:56:37 +02:00
sceMpegbase: convert with swscale, keeping the scalar path as the fallback
The planes the de-tiling produces are already the YUV420P swscale wants, and our sceMpeg HLE converts the same frames the same way, so the pixel formats and the studio-range setup come straight from MediaEngine::getSwsFormat. It is 3-4x quicker than going a pixel at a time: 0.37-0.44ms a frame becomes 0.09-0.12ms, which is the whole reason sceMpegBaseCscAvc was at the top of a profile. Chroma is upsampled with SWS_POINT rather than the HLE's SWS_BILINEAR, since replicating is what the scalar path does and, being a fixed-function block, almost certainly what the hardware does. It is not bit-identical - swscale rounds its own way. TestMpegCsc measures the gap per channel rather than per byte, so the number means something for a packed 16-bit pixel: worst 1 step of 31 for 5650 and 5551, 2 of 15 for 4444, 3 of 255 for 8888, with means around a fifth of a step. The scalar path stays as what the longhand reference is checked against, and takes anything swscale won't - an odd range origin, or a build without ffmpeg. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
02906f0510
commit
4cb01fea67
+128
-1
@@ -37,6 +37,13 @@
|
||||
#include "GPU/GPUState.h"
|
||||
#include "GPU/ge_constants.h"
|
||||
|
||||
#ifdef USE_FFMPEG
|
||||
extern "C" {
|
||||
#include "libswscale/swscale.h"
|
||||
#include "libavutil/pixfmt.h"
|
||||
}
|
||||
#endif
|
||||
|
||||
// The PES payloads gathered by sceMpegBasePESpacketCopy, keyed by the destination each was
|
||||
// copied to. It carries audio as well as video - the destination is what tells them apart - so
|
||||
// sceVideocodec has to ask for the one matching the address it was handed.
|
||||
@@ -55,6 +62,7 @@ void __MpegBaseInit() {
|
||||
g_pesPackets.clear();
|
||||
g_untileScratch.clear();
|
||||
g_untileScratch.shrink_to_fit();
|
||||
MpegCscShutdown();
|
||||
g_mpegBaseBufferWidth = 512;
|
||||
g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888;
|
||||
}
|
||||
@@ -304,7 +312,7 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) {
|
||||
// luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is
|
||||
// destStride pixels wide in the format pixelMode names, and the converted range always lands at
|
||||
// its origin.
|
||||
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
@@ -325,6 +333,125 @@ void MpegCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef USE_FFMPEG
|
||||
|
||||
// swscale is what our sceMpeg HLE converts with, and the planes the de-tiling produces are
|
||||
// already the YUV420P it wants, so the same thing works here - and it is a great deal quicker
|
||||
// than doing it a pixel at a time.
|
||||
//
|
||||
// The four output formats are the ones MediaEngine::getSwsFormat picks, for the same reasons.
|
||||
// Alpha is not among them: swscale writes RGBA opaque and leaves the spare bits of the 16-bit
|
||||
// formats clear, so the masking below is what makes the result match the hardware, exactly as
|
||||
// the HLE does after its own sws_scale.
|
||||
static AVPixelFormat SwsFormatForPixelMode(int pixelMode) {
|
||||
switch (pixelMode) {
|
||||
case GE_CMODE_16BIT_BGR5650: return AV_PIX_FMT_BGR565LE;
|
||||
case GE_CMODE_16BIT_ABGR5551: return AV_PIX_FMT_BGR555LE;
|
||||
case GE_CMODE_16BIT_ABGR4444: return AV_PIX_FMT_BGR444LE;
|
||||
default: return AV_PIX_FMT_RGBA;
|
||||
}
|
||||
}
|
||||
|
||||
static SwsContext *g_cscSws;
|
||||
static int g_cscSwsWidth, g_cscSwsHeight, g_cscSwsFormat = -1;
|
||||
|
||||
void MpegCscShutdown() {
|
||||
if (g_cscSws) {
|
||||
sws_freeContext(g_cscSws);
|
||||
g_cscSws = nullptr;
|
||||
}
|
||||
g_cscSwsWidth = 0;
|
||||
g_cscSwsHeight = 0;
|
||||
g_cscSwsFormat = -1;
|
||||
}
|
||||
|
||||
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
// Chroma is half resolution, so an odd origin would start half a sample in and there is no way
|
||||
// to say that to swscale. Nothing can actually ask for one - the ranges arrive in macroblocks -
|
||||
// but the scalar path is still there for it.
|
||||
if ((rangeX & 1) || (rangeY & 1)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const AVPixelFormat format = SwsFormatForPixelMode(pixelMode);
|
||||
if (rangeWidth != g_cscSwsWidth || rangeHeight != g_cscSwsHeight || (int)format != g_cscSwsFormat) {
|
||||
g_cscSws = sws_getCachedContext(g_cscSws, rangeWidth, rangeHeight, AV_PIX_FMT_YUV420P,
|
||||
rangeWidth, rangeHeight, format, SWS_POINT, nullptr, nullptr, nullptr);
|
||||
if (!g_cscSws) {
|
||||
return false;
|
||||
}
|
||||
// Studio swing both ways, which is the range the coefficients in the scalar path assume.
|
||||
int *invCoeff, *coeff, srcRange, dstRange, brightness, contrast, saturation;
|
||||
if (sws_getColorspaceDetails(g_cscSws, &invCoeff, &srcRange, &coeff, &dstRange, &brightness,
|
||||
&contrast, &saturation) != -1) {
|
||||
sws_setColorspaceDetails(g_cscSws, invCoeff, 0, coeff, 0, brightness, contrast, saturation);
|
||||
}
|
||||
g_cscSwsWidth = rangeWidth;
|
||||
g_cscSwsHeight = rangeHeight;
|
||||
g_cscSwsFormat = (int)format;
|
||||
}
|
||||
|
||||
const int width2 = width >> 1;
|
||||
const u8 *srcSlice[4] = {
|
||||
luma + (size_t)rangeY * width + rangeX,
|
||||
cb + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
|
||||
cr + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
|
||||
nullptr,
|
||||
};
|
||||
const int srcStride[4] = { width, width2, width2, 0 };
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
u8 *dstSlice[4] = { dest, nullptr, nullptr, nullptr };
|
||||
const int dstStride[4] = { destStride * bpp, 0, 0, 0 };
|
||||
if (sws_scale(g_cscSws, srcSlice, srcStride, 0, rangeHeight, dstSlice, dstStride) <= 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Clear the alpha swscale filled in, which the hardware leaves at zero.
|
||||
for (int y = 0; y < rangeHeight; y++) {
|
||||
u8 *row = dest + (size_t)y * destStride * bpp;
|
||||
if (bpp == 4) {
|
||||
u32_le *p32 = (u32_le *)row;
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
p32[x] = p32[x] & 0x00FFFFFF;
|
||||
}
|
||||
} else if (pixelMode != GE_CMODE_16BIT_BGR5650) {
|
||||
const u16 mask = pixelMode == GE_CMODE_16BIT_ABGR5551 ? 0x7FFF : 0x0FFF;
|
||||
u16_le *p16 = (u16_le *)row;
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
p16[x] = p16[x] & mask;
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
#else // !USE_FFMPEG
|
||||
|
||||
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void MpegCscShutdown() {}
|
||||
|
||||
#endif // USE_FFMPEG
|
||||
|
||||
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
#ifdef USE_FFMPEG
|
||||
if (MpegCscRangeSws(dest, destStride, pixelMode, luma, cb, cr, width,
|
||||
rangeX, rangeY, rangeWidth, rangeHeight)) {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
MpegCscRangeScalar(dest, destStride, pixelMode, luma, cb, cr, width,
|
||||
rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
}
|
||||
|
||||
// The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the
|
||||
// latter over the whole frame.
|
||||
static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
|
||||
|
||||
@@ -58,3 +58,17 @@ void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int siz
|
||||
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
|
||||
|
||||
// The two implementations behind it, exposed so TestMpegCsc can measure and compare them.
|
||||
// The scalar one handles anything; the swscale one refuses what it cannot express and is then
|
||||
// not used. They do not agree to the bit - swscale rounds its own way - so the scalar one is
|
||||
// what the reference in the test is checked against.
|
||||
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
|
||||
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
|
||||
|
||||
// Frees the cached swscale context.
|
||||
void MpegCscShutdown();
|
||||
|
||||
@@ -131,7 +131,7 @@ static bool CompareAgainstReference(const TestFrame &frame, int pixelMode,
|
||||
const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp;
|
||||
std::vector<u8> got(destSize, 0xCD), want(destSize, 0xCD);
|
||||
|
||||
MpegCscRange(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
MpegCscRangeScalar(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
@@ -147,7 +147,10 @@ static bool CompareAgainstReference(const TestFrame &frame, int pixelMode,
|
||||
return true;
|
||||
}
|
||||
|
||||
static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector<u8> &dest) {
|
||||
typedef void (*CscFunc)(u8 *, int, int, const u8 *, const u8 *, const u8 *, int, int, int, int, int);
|
||||
|
||||
static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector<u8> &dest,
|
||||
CscFunc fn = &MpegCscRange) {
|
||||
const int destStride = 512;
|
||||
// Long enough to swamp the clock's own resolution, short enough not to pad the test run.
|
||||
const double seconds = 0.2;
|
||||
@@ -155,7 +158,7 @@ static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode,
|
||||
const double start = time_now_d();
|
||||
do {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
MpegCscRange(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, 0, 0, frame.width, frame.height);
|
||||
frames++;
|
||||
}
|
||||
@@ -309,12 +312,65 @@ bool TestMpegCsc() {
|
||||
}
|
||||
|
||||
std::vector<u8> dest((size_t)512 * 272 * 4, 0);
|
||||
// How far swscale lands from the conversion written out longhand. It rounds its own way, so
|
||||
// this is not expected to be zero - the question is whether it is close enough to use.
|
||||
printf("swscale against the reference, per channel:\n");
|
||||
for (int pixelMode : pixelModes) {
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
std::vector<u8> sws((size_t)512 * 272 * 4, 0), ref((size_t)512 * 272 * 4, 0);
|
||||
if (!MpegCscRangeSws(sws.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), 480, 0, 0, 480, 272)) {
|
||||
printf(" %s: declined\n", PixelModeName(pixelMode));
|
||||
continue;
|
||||
}
|
||||
ReferenceCscRange(ref.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), 480, 0, 0, 480, 272);
|
||||
// Per channel, because that is what "how different does it look" means - a byte-wise diff
|
||||
// on a packed 16-bit pixel could be one step in one channel or a disaster in three.
|
||||
int shifts[3], masks[3];
|
||||
if (bpp == 4) {
|
||||
shifts[0] = 0; shifts[1] = 8; shifts[2] = 16;
|
||||
masks[0] = masks[1] = masks[2] = 0xFF;
|
||||
} else if (pixelMode == GE_CMODE_16BIT_BGR5650) {
|
||||
shifts[0] = 0; shifts[1] = 5; shifts[2] = 11;
|
||||
masks[0] = 0x1F; masks[1] = 0x3F; masks[2] = 0x1F;
|
||||
} else if (pixelMode == GE_CMODE_16BIT_ABGR5551) {
|
||||
shifts[0] = 0; shifts[1] = 5; shifts[2] = 10;
|
||||
masks[0] = masks[1] = masks[2] = 0x1F;
|
||||
} else {
|
||||
shifts[0] = 0; shifts[1] = 4; shifts[2] = 8;
|
||||
masks[0] = masks[1] = masks[2] = 0x0F;
|
||||
}
|
||||
int worst = 0;
|
||||
double total = 0.0;
|
||||
int count = 0;
|
||||
for (int y = 0; y < 272; y++) {
|
||||
for (int x = 0; x < 480; x++) {
|
||||
const size_t off = ((size_t)y * 512 + x) * bpp;
|
||||
u32 a = 0, b = 0;
|
||||
memcpy(&a, &sws[off], bpp);
|
||||
memcpy(&b, &ref[off], bpp);
|
||||
for (int ch = 0; ch < 3; ch++) {
|
||||
const int d = abs((int)((a >> shifts[ch]) & masks[ch]) -
|
||||
(int)((b >> shifts[ch]) & masks[ch]));
|
||||
worst = worst > d ? worst : d;
|
||||
total += d;
|
||||
count++;
|
||||
}
|
||||
}
|
||||
}
|
||||
printf(" %s: worst channel step %d, mean %.3f\n", PixelModeName(pixelMode), worst,
|
||||
total / count);
|
||||
}
|
||||
|
||||
printf("MpegCscRange, 480x272:\n");
|
||||
for (int pixelMode : pixelModes) {
|
||||
const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest);
|
||||
const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRangeScalar);
|
||||
const double swsMps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRange);
|
||||
// A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it.
|
||||
printf(" %s: %6.1f MPix/s (%5.2f ms/frame)\n", PixelModeName(pixelMode), mps,
|
||||
480.0 * 272.0 / mps / 1000.0);
|
||||
printf(" %s: scalar %6.1f MPix/s (%5.2f ms), swscale %6.1f MPix/s (%5.2f ms)\n",
|
||||
PixelModeName(pixelMode), mps, 480.0 * 272.0 / mps / 1000.0,
|
||||
swsMps, 480.0 * 272.0 / swsMps / 1000.0);
|
||||
}
|
||||
|
||||
return true;
|
||||
|
||||
Reference in New Issue
Block a user