sceMpegbase: convert with swscale, keeping the scalar path as the fallback

The planes the de-tiling produces are already the YUV420P swscale wants, and our
sceMpeg HLE converts the same frames the same way, so the pixel formats and the
studio-range setup come straight from MediaEngine::getSwsFormat. It is 3-4x
quicker than going a pixel at a time: 0.37-0.44ms a frame becomes 0.09-0.12ms,
which is the whole reason sceMpegBaseCscAvc was at the top of a profile.

Chroma is upsampled with SWS_POINT rather than the HLE's SWS_BILINEAR, since
replicating is what the scalar path does and, being a fixed-function block,
almost certainly what the hardware does.

It is not bit-identical - swscale rounds its own way. TestMpegCsc measures the
gap per channel rather than per byte, so the number means something for a packed
16-bit pixel: worst 1 step of 31 for 5650 and 5551, 2 of 15 for 4444, 3 of 255
for 8888, with means around a fifth of a step. The scalar path stays as what the
longhand reference is checked against, and takes anything swscale won't - an odd
range origin, or a build without ffmpeg.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Henrik Rydgård
2026-09-18 11:38:49 -06:00
co-authored by Claude Opus 5
parent 02906f0510
commit 4cb01fea67
3 changed files with 204 additions and 7 deletions
+128 -1
View File
@@ -37,6 +37,13 @@
#include "GPU/GPUState.h"
#include "GPU/ge_constants.h"
#ifdef USE_FFMPEG
extern "C" {
#include "libswscale/swscale.h"
#include "libavutil/pixfmt.h"
}
#endif
// The PES payloads gathered by sceMpegBasePESpacketCopy, keyed by the destination each was
// copied to. It carries audio as well as video - the destination is what tells them apart - so
// sceVideocodec has to ask for the one matching the address it was handed.
@@ -55,6 +62,7 @@ void __MpegBaseInit() {
g_pesPackets.clear();
g_untileScratch.clear();
g_untileScratch.shrink_to_fit();
MpegCscShutdown();
g_mpegBaseBufferWidth = 512;
g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888;
}
@@ -304,7 +312,7 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) {
// luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is
// destStride pixels wide in the format pixelMode names, and the converted range always lands at
// its origin.
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
@@ -325,6 +333,125 @@ void MpegCscRange(u8 *dest, int destStride, int pixelMode,
}
}
#ifdef USE_FFMPEG
// swscale is what our sceMpeg HLE converts with, and the planes the de-tiling produces are
// already the YUV420P it wants, so the same thing works here - and it is a great deal quicker
// than doing it a pixel at a time.
//
// The four output formats are the ones MediaEngine::getSwsFormat picks, for the same reasons.
// Alpha is not among them: swscale writes RGBA opaque and leaves the spare bits of the 16-bit
// formats clear, so the masking below is what makes the result match the hardware, exactly as
// the HLE does after its own sws_scale.
static AVPixelFormat SwsFormatForPixelMode(int pixelMode) {
switch (pixelMode) {
case GE_CMODE_16BIT_BGR5650: return AV_PIX_FMT_BGR565LE;
case GE_CMODE_16BIT_ABGR5551: return AV_PIX_FMT_BGR555LE;
case GE_CMODE_16BIT_ABGR4444: return AV_PIX_FMT_BGR444LE;
default: return AV_PIX_FMT_RGBA;
}
}
static SwsContext *g_cscSws;
static int g_cscSwsWidth, g_cscSwsHeight, g_cscSwsFormat = -1;
void MpegCscShutdown() {
if (g_cscSws) {
sws_freeContext(g_cscSws);
g_cscSws = nullptr;
}
g_cscSwsWidth = 0;
g_cscSwsHeight = 0;
g_cscSwsFormat = -1;
}
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
// Chroma is half resolution, so an odd origin would start half a sample in and there is no way
// to say that to swscale. Nothing can actually ask for one - the ranges arrive in macroblocks -
// but the scalar path is still there for it.
if ((rangeX & 1) || (rangeY & 1)) {
return false;
}
const AVPixelFormat format = SwsFormatForPixelMode(pixelMode);
if (rangeWidth != g_cscSwsWidth || rangeHeight != g_cscSwsHeight || (int)format != g_cscSwsFormat) {
g_cscSws = sws_getCachedContext(g_cscSws, rangeWidth, rangeHeight, AV_PIX_FMT_YUV420P,
rangeWidth, rangeHeight, format, SWS_POINT, nullptr, nullptr, nullptr);
if (!g_cscSws) {
return false;
}
// Studio swing both ways, which is the range the coefficients in the scalar path assume.
int *invCoeff, *coeff, srcRange, dstRange, brightness, contrast, saturation;
if (sws_getColorspaceDetails(g_cscSws, &invCoeff, &srcRange, &coeff, &dstRange, &brightness,
&contrast, &saturation) != -1) {
sws_setColorspaceDetails(g_cscSws, invCoeff, 0, coeff, 0, brightness, contrast, saturation);
}
g_cscSwsWidth = rangeWidth;
g_cscSwsHeight = rangeHeight;
g_cscSwsFormat = (int)format;
}
const int width2 = width >> 1;
const u8 *srcSlice[4] = {
luma + (size_t)rangeY * width + rangeX,
cb + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
cr + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
nullptr,
};
const int srcStride[4] = { width, width2, width2, 0 };
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
u8 *dstSlice[4] = { dest, nullptr, nullptr, nullptr };
const int dstStride[4] = { destStride * bpp, 0, 0, 0 };
if (sws_scale(g_cscSws, srcSlice, srcStride, 0, rangeHeight, dstSlice, dstStride) <= 0) {
return false;
}
// Clear the alpha swscale filled in, which the hardware leaves at zero.
for (int y = 0; y < rangeHeight; y++) {
u8 *row = dest + (size_t)y * destStride * bpp;
if (bpp == 4) {
u32_le *p32 = (u32_le *)row;
for (int x = 0; x < rangeWidth; x++) {
p32[x] = p32[x] & 0x00FFFFFF;
}
} else if (pixelMode != GE_CMODE_16BIT_BGR5650) {
const u16 mask = pixelMode == GE_CMODE_16BIT_ABGR5551 ? 0x7FFF : 0x0FFF;
u16_le *p16 = (u16_le *)row;
for (int x = 0; x < rangeWidth; x++) {
p16[x] = p16[x] & mask;
}
}
}
return true;
}
#else // !USE_FFMPEG
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
return false;
}
void MpegCscShutdown() {}
#endif // USE_FFMPEG
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
#ifdef USE_FFMPEG
if (MpegCscRangeSws(dest, destStride, pixelMode, luma, cb, cr, width,
rangeX, rangeY, rangeWidth, rangeHeight)) {
return;
}
#endif
MpegCscRangeScalar(dest, destStride, pixelMode, luma, cb, cr, width,
rangeX, rangeY, rangeWidth, rangeHeight);
}
// The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the
// latter over the whole frame.
static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
+14
View File
@@ -58,3 +58,17 @@ void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int siz
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
// The two implementations behind it, exposed so TestMpegCsc can measure and compare them.
// The scalar one handles anything; the swscale one refuses what it cannot express and is then
// not used. They do not agree to the bit - swscale rounds its own way - so the scalar one is
// what the reference in the test is checked against.
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
// Frees the cached swscale context.
void MpegCscShutdown();
+62 -6
View File
@@ -131,7 +131,7 @@ static bool CompareAgainstReference(const TestFrame &frame, int pixelMode,
const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp;
std::vector<u8> got(destSize, 0xCD), want(destSize, 0xCD);
MpegCscRange(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
MpegCscRangeScalar(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
@@ -147,7 +147,10 @@ static bool CompareAgainstReference(const TestFrame &frame, int pixelMode,
return true;
}
static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector<u8> &dest) {
typedef void (*CscFunc)(u8 *, int, int, const u8 *, const u8 *, const u8 *, int, int, int, int, int);
static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector<u8> &dest,
CscFunc fn = &MpegCscRange) {
const int destStride = 512;
// Long enough to swamp the clock's own resolution, short enough not to pad the test run.
const double seconds = 0.2;
@@ -155,7 +158,7 @@ static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode,
const double start = time_now_d();
do {
for (int i = 0; i < 4; i++) {
MpegCscRange(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), frame.width, 0, 0, frame.width, frame.height);
frames++;
}
@@ -309,12 +312,65 @@ bool TestMpegCsc() {
}
std::vector<u8> dest((size_t)512 * 272 * 4, 0);
// How far swscale lands from the conversion written out longhand. It rounds its own way, so
// this is not expected to be zero - the question is whether it is close enough to use.
printf("swscale against the reference, per channel:\n");
for (int pixelMode : pixelModes) {
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
std::vector<u8> sws((size_t)512 * 272 * 4, 0), ref((size_t)512 * 272 * 4, 0);
if (!MpegCscRangeSws(sws.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), 480, 0, 0, 480, 272)) {
printf(" %s: declined\n", PixelModeName(pixelMode));
continue;
}
ReferenceCscRange(ref.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), 480, 0, 0, 480, 272);
// Per channel, because that is what "how different does it look" means - a byte-wise diff
// on a packed 16-bit pixel could be one step in one channel or a disaster in three.
int shifts[3], masks[3];
if (bpp == 4) {
shifts[0] = 0; shifts[1] = 8; shifts[2] = 16;
masks[0] = masks[1] = masks[2] = 0xFF;
} else if (pixelMode == GE_CMODE_16BIT_BGR5650) {
shifts[0] = 0; shifts[1] = 5; shifts[2] = 11;
masks[0] = 0x1F; masks[1] = 0x3F; masks[2] = 0x1F;
} else if (pixelMode == GE_CMODE_16BIT_ABGR5551) {
shifts[0] = 0; shifts[1] = 5; shifts[2] = 10;
masks[0] = masks[1] = masks[2] = 0x1F;
} else {
shifts[0] = 0; shifts[1] = 4; shifts[2] = 8;
masks[0] = masks[1] = masks[2] = 0x0F;
}
int worst = 0;
double total = 0.0;
int count = 0;
for (int y = 0; y < 272; y++) {
for (int x = 0; x < 480; x++) {
const size_t off = ((size_t)y * 512 + x) * bpp;
u32 a = 0, b = 0;
memcpy(&a, &sws[off], bpp);
memcpy(&b, &ref[off], bpp);
for (int ch = 0; ch < 3; ch++) {
const int d = abs((int)((a >> shifts[ch]) & masks[ch]) -
(int)((b >> shifts[ch]) & masks[ch]));
worst = worst > d ? worst : d;
total += d;
count++;
}
}
}
printf(" %s: worst channel step %d, mean %.3f\n", PixelModeName(pixelMode), worst,
total / count);
}
printf("MpegCscRange, 480x272:\n");
for (int pixelMode : pixelModes) {
const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest);
const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRangeScalar);
const double swsMps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRange);
// A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it.
printf(" %s: %6.1f MPix/s (%5.2f ms/frame)\n", PixelModeName(pixelMode), mps,
480.0 * 272.0 / mps / 1000.0);
printf(" %s: scalar %6.1f MPix/s (%5.2f ms), swscale %6.1f MPix/s (%5.2f ms)\n",
PixelModeName(pixelMode), mps, 480.0 * 272.0 / mps / 1000.0,
swsMps, 480.0 * 272.0 / swsMps / 1000.0);
}
return true;