diff --git a/Core/HLE/sceMpegbase.cpp b/Core/HLE/sceMpegbase.cpp index 11082bc40f..ac56f808d8 100644 --- a/Core/HLE/sceMpegbase.cpp +++ b/Core/HLE/sceMpegbase.cpp @@ -37,6 +37,13 @@ #include "GPU/GPUState.h" #include "GPU/ge_constants.h" +#ifdef USE_FFMPEG +extern "C" { +#include "libswscale/swscale.h" +#include "libavutil/pixfmt.h" +} +#endif + // The PES payloads gathered by sceMpegBasePESpacketCopy, keyed by the destination each was // copied to. It carries audio as well as video - the destination is what tells them apart - so // sceVideocodec has to ask for the one matching the address it was handed. @@ -55,6 +62,7 @@ void __MpegBaseInit() { g_pesPackets.clear(); g_untileScratch.clear(); g_untileScratch.shrink_to_fit(); + MpegCscShutdown(); g_mpegBaseBufferWidth = 512; g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888; } @@ -304,7 +312,7 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) { // luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is // destStride pixels wide in the format pixelMode names, and the converted range always lands at // its origin. -void MpegCscRange(u8 *dest, int destStride, int pixelMode, +void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode, const u8 *luma, const u8 *cb, const u8 *cr, int width, int rangeX, int rangeY, int rangeWidth, int rangeHeight) { const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; @@ -325,6 +333,125 @@ void MpegCscRange(u8 *dest, int destStride, int pixelMode, } } +#ifdef USE_FFMPEG + +// swscale is what our sceMpeg HLE converts with, and the planes the de-tiling produces are +// already the YUV420P it wants, so the same thing works here - and it is a great deal quicker +// than doing it a pixel at a time. +// +// The four output formats are the ones MediaEngine::getSwsFormat picks, for the same reasons. +// Alpha is not among them: swscale writes RGBA opaque and leaves the spare bits of the 16-bit +// formats clear, so the masking below is what makes the result match the hardware, exactly as +// the HLE does after its own sws_scale. +static AVPixelFormat SwsFormatForPixelMode(int pixelMode) { + switch (pixelMode) { + case GE_CMODE_16BIT_BGR5650: return AV_PIX_FMT_BGR565LE; + case GE_CMODE_16BIT_ABGR5551: return AV_PIX_FMT_BGR555LE; + case GE_CMODE_16BIT_ABGR4444: return AV_PIX_FMT_BGR444LE; + default: return AV_PIX_FMT_RGBA; + } +} + +static SwsContext *g_cscSws; +static int g_cscSwsWidth, g_cscSwsHeight, g_cscSwsFormat = -1; + +void MpegCscShutdown() { + if (g_cscSws) { + sws_freeContext(g_cscSws); + g_cscSws = nullptr; + } + g_cscSwsWidth = 0; + g_cscSwsHeight = 0; + g_cscSwsFormat = -1; +} + +bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + // Chroma is half resolution, so an odd origin would start half a sample in and there is no way + // to say that to swscale. Nothing can actually ask for one - the ranges arrive in macroblocks - + // but the scalar path is still there for it. + if ((rangeX & 1) || (rangeY & 1)) { + return false; + } + + const AVPixelFormat format = SwsFormatForPixelMode(pixelMode); + if (rangeWidth != g_cscSwsWidth || rangeHeight != g_cscSwsHeight || (int)format != g_cscSwsFormat) { + g_cscSws = sws_getCachedContext(g_cscSws, rangeWidth, rangeHeight, AV_PIX_FMT_YUV420P, + rangeWidth, rangeHeight, format, SWS_POINT, nullptr, nullptr, nullptr); + if (!g_cscSws) { + return false; + } + // Studio swing both ways, which is the range the coefficients in the scalar path assume. + int *invCoeff, *coeff, srcRange, dstRange, brightness, contrast, saturation; + if (sws_getColorspaceDetails(g_cscSws, &invCoeff, &srcRange, &coeff, &dstRange, &brightness, + &contrast, &saturation) != -1) { + sws_setColorspaceDetails(g_cscSws, invCoeff, 0, coeff, 0, brightness, contrast, saturation); + } + g_cscSwsWidth = rangeWidth; + g_cscSwsHeight = rangeHeight; + g_cscSwsFormat = (int)format; + } + + const int width2 = width >> 1; + const u8 *srcSlice[4] = { + luma + (size_t)rangeY * width + rangeX, + cb + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1), + cr + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1), + nullptr, + }; + const int srcStride[4] = { width, width2, width2, 0 }; + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + u8 *dstSlice[4] = { dest, nullptr, nullptr, nullptr }; + const int dstStride[4] = { destStride * bpp, 0, 0, 0 }; + if (sws_scale(g_cscSws, srcSlice, srcStride, 0, rangeHeight, dstSlice, dstStride) <= 0) { + return false; + } + + // Clear the alpha swscale filled in, which the hardware leaves at zero. + for (int y = 0; y < rangeHeight; y++) { + u8 *row = dest + (size_t)y * destStride * bpp; + if (bpp == 4) { + u32_le *p32 = (u32_le *)row; + for (int x = 0; x < rangeWidth; x++) { + p32[x] = p32[x] & 0x00FFFFFF; + } + } else if (pixelMode != GE_CMODE_16BIT_BGR5650) { + const u16 mask = pixelMode == GE_CMODE_16BIT_ABGR5551 ? 0x7FFF : 0x0FFF; + u16_le *p16 = (u16_le *)row; + for (int x = 0; x < rangeWidth; x++) { + p16[x] = p16[x] & mask; + } + } + } + return true; +} + +#else // !USE_FFMPEG + +bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + return false; +} + +void MpegCscShutdown() {} + +#endif // USE_FFMPEG + +void MpegCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { +#ifdef USE_FFMPEG + if (MpegCscRangeSws(dest, destStride, pixelMode, luma, cb, cr, width, + rangeX, rangeY, rangeWidth, rangeHeight)) { + return; + } +#endif + MpegCscRangeScalar(dest, destStride, pixelMode, luma, cb, cr, width, + rangeX, rangeY, rangeWidth, rangeHeight); +} + // The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the // latter over the whole frame. static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth, diff --git a/Core/HLE/sceMpegbase.h b/Core/HLE/sceMpegbase.h index f112daf327..09eb626c62 100644 --- a/Core/HLE/sceMpegbase.h +++ b/Core/HLE/sceMpegbase.h @@ -58,3 +58,17 @@ void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int siz void MpegCscRange(u8 *dest, int destStride, int pixelMode, const u8 *luma, const u8 *cb, const u8 *cr, int width, int rangeX, int rangeY, int rangeWidth, int rangeHeight); + +// The two implementations behind it, exposed so TestMpegCsc can measure and compare them. +// The scalar one handles anything; the swscale one refuses what it cannot express and is then +// not used. They do not agree to the bit - swscale rounds its own way - so the scalar one is +// what the reference in the test is checked against. +void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight); +bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight); + +// Frees the cached swscale context. +void MpegCscShutdown(); diff --git a/unittest/TestMpegCsc.cpp b/unittest/TestMpegCsc.cpp index 585d8d4eb7..c9a845297e 100644 --- a/unittest/TestMpegCsc.cpp +++ b/unittest/TestMpegCsc.cpp @@ -131,7 +131,7 @@ static bool CompareAgainstReference(const TestFrame &frame, int pixelMode, const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp; std::vector got(destSize, 0xCD), want(destSize, 0xCD); - MpegCscRange(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + MpegCscRangeScalar(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight); ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight); @@ -147,7 +147,10 @@ static bool CompareAgainstReference(const TestFrame &frame, int pixelMode, return true; } -static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector &dest) { +typedef void (*CscFunc)(u8 *, int, int, const u8 *, const u8 *, const u8 *, int, int, int, int, int); + +static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector &dest, + CscFunc fn = &MpegCscRange) { const int destStride = 512; // Long enough to swamp the clock's own resolution, short enough not to pad the test run. const double seconds = 0.2; @@ -155,7 +158,7 @@ static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, const double start = time_now_d(); do { for (int i = 0; i < 4; i++) { - MpegCscRange(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), frame.cr.data(), frame.width, 0, 0, frame.width, frame.height); frames++; } @@ -309,12 +312,65 @@ bool TestMpegCsc() { } std::vector dest((size_t)512 * 272 * 4, 0); + // How far swscale lands from the conversion written out longhand. It rounds its own way, so + // this is not expected to be zero - the question is whether it is close enough to use. + printf("swscale against the reference, per channel:\n"); + for (int pixelMode : pixelModes) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + std::vector sws((size_t)512 * 272 * 4, 0), ref((size_t)512 * 272 * 4, 0); + if (!MpegCscRangeSws(sws.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), 480, 0, 0, 480, 272)) { + printf(" %s: declined\n", PixelModeName(pixelMode)); + continue; + } + ReferenceCscRange(ref.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), 480, 0, 0, 480, 272); + // Per channel, because that is what "how different does it look" means - a byte-wise diff + // on a packed 16-bit pixel could be one step in one channel or a disaster in three. + int shifts[3], masks[3]; + if (bpp == 4) { + shifts[0] = 0; shifts[1] = 8; shifts[2] = 16; + masks[0] = masks[1] = masks[2] = 0xFF; + } else if (pixelMode == GE_CMODE_16BIT_BGR5650) { + shifts[0] = 0; shifts[1] = 5; shifts[2] = 11; + masks[0] = 0x1F; masks[1] = 0x3F; masks[2] = 0x1F; + } else if (pixelMode == GE_CMODE_16BIT_ABGR5551) { + shifts[0] = 0; shifts[1] = 5; shifts[2] = 10; + masks[0] = masks[1] = masks[2] = 0x1F; + } else { + shifts[0] = 0; shifts[1] = 4; shifts[2] = 8; + masks[0] = masks[1] = masks[2] = 0x0F; + } + int worst = 0; + double total = 0.0; + int count = 0; + for (int y = 0; y < 272; y++) { + for (int x = 0; x < 480; x++) { + const size_t off = ((size_t)y * 512 + x) * bpp; + u32 a = 0, b = 0; + memcpy(&a, &sws[off], bpp); + memcpy(&b, &ref[off], bpp); + for (int ch = 0; ch < 3; ch++) { + const int d = abs((int)((a >> shifts[ch]) & masks[ch]) - + (int)((b >> shifts[ch]) & masks[ch])); + worst = worst > d ? worst : d; + total += d; + count++; + } + } + } + printf(" %s: worst channel step %d, mean %.3f\n", PixelModeName(pixelMode), worst, + total / count); + } + printf("MpegCscRange, 480x272:\n"); for (int pixelMode : pixelModes) { - const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest); + const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRangeScalar); + const double swsMps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRange); // A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it. - printf(" %s: %6.1f MPix/s (%5.2f ms/frame)\n", PixelModeName(pixelMode), mps, - 480.0 * 272.0 / mps / 1000.0); + printf(" %s: scalar %6.1f MPix/s (%5.2f ms), swscale %6.1f MPix/s (%5.2f ms)\n", + PixelModeName(pixelMode), mps, 480.0 * 272.0 / mps / 1000.0, + swsMps, 480.0 * 272.0 / swsMps / 1000.0); } return true;