diff --git a/CMakeLists.txt b/CMakeLists.txt index a936e3afcd..c81d415524 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1368,6 +1368,7 @@ if(UNITTEST) unittest/TestVFS.cpp unittest/TestZipSlip.cpp unittest/TestLzrc.cpp + unittest/TestMpegCsc.cpp unittest/TestDemangle.cpp unittest/TestTextureReplacer.cpp unittest/TestRiscVEmitter.cpp diff --git a/Core/HLE/sceMpegbase.cpp b/Core/HLE/sceMpegbase.cpp index 7167ec9f86..aa7f85a73c 100644 --- a/Core/HLE/sceMpegbase.cpp +++ b/Core/HLE/sceMpegbase.cpp @@ -269,6 +269,34 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) { } } +// The conversion itself, with nothing around it. Pure, so that TestMpegCsc can measure it and +// check it - it is the hottest thing in video playback, and the point of having it out here is +// that it can be worked on without a game in the loop. +// +// luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is +// destStride pixels wide in the format pixelMode names, and the converted range always lands at +// its origin. +void MpegCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + const int width2 = width >> 1; + for (int y = 0; y < rangeHeight; y++) { + const int sy = rangeY + y; + for (int x = 0; x < rangeWidth; x++) { + const int sx = rangeX + x; + const int ci = (sy >> 1) * width2 + (sx >> 1); + const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode); + if (bpp == 4) { + memcpy(dest + (y * destStride + x) * 4, &pixel, 4); + } else { + const u16 p16 = (u16)pixel; + memcpy(dest + (y * destStride + x) * 2, &p16, 2); + } + } + } +} + // The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the // latter over the whole frame. static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth, @@ -318,21 +346,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth, return hleLogError(Log::Mpeg, -1, "output buffer not writable"); } - const int width2 = width >> 1; - for (int y = 0; y < rangeHeight; y++) { - const int sy = rangeY + y; - for (int x = 0; x < rangeWidth; x++) { - const int sx = rangeX + x; - const int ci = (sy >> 1) * width2 + (sx >> 1); - const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], g_mpegBasePixelMode); - if (bpp == 4) { - memcpy(dest + (y * bufferWidth + x) * 4, &pixel, 4); - } else { - const u16 p16 = (u16)pixel; - memcpy(dest + (y * bufferWidth + x) * 2, &p16, 2); - } - } - } + MpegCscRange(dest, bufferWidth, g_mpegBasePixelMode, luma.data(), cb.data(), cr.data(), width, + rangeX, rangeY, rangeWidth, rangeHeight); NotifyMemInfo(MemBlockFlags::WRITE, bufferRGB, destSize, "MpegBaseCsc"); // The CPU just wrote a video frame into what is usually a display buffer. The hardware backends // don't see that on their own, so without telling them the screen keeps showing the last frame diff --git a/Core/HLE/sceMpegbase.h b/Core/HLE/sceMpegbase.h index cf85d1a4ca..d31457dd74 100644 --- a/Core/HLE/sceMpegbase.h +++ b/Core/HLE/sceMpegbase.h @@ -41,3 +41,12 @@ std::vector MpegBaseTakePESPacket(u32 dest); // out at half width and half height, as YUV420 does. bool ReadTiledYCbCr(const u32 *buffers, int width, int height, std::vector &luma, std::vector &cb, std::vector &cr); + +// Converts a rectangle of a planar YCbCr420 frame to RGB, the way the DMACPLUS does on the way to +// the screen. Pure, so it can be measured and checked on its own - see TestMpegCsc. +// +// luma is width by height; cb and cr are half that in both directions. dest is destStride pixels +// wide in the format pixelMode names (a GEBufferFormat), and the range lands at its origin. +void MpegCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight); diff --git a/unittest/TestMpegCsc.cpp b/unittest/TestMpegCsc.cpp new file mode 100644 index 0000000000..24dbb99f43 --- /dev/null +++ b/unittest/TestMpegCsc.cpp @@ -0,0 +1,196 @@ +// Copyright (c) 2012- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +// Correctness and speed of the Media Engine's colour conversion, which is the hottest thing in +// video playback once sceMpeg runs the real mpeg.prx - sceMpegBaseCscAvc sits at the top of a +// profile of a movie. +// +// The correctness half is a reference implementation written out longhand, so an optimized +// MpegCscRange has something to be wrong against that isn't itself. The speed half reports +// megapixels per second for a 480x272 frame, the size a PSP movie actually is. + +#include +#include +#include + +#include "Common/CommonTypes.h" +#include "Common/TimeUtil.h" +#include "Core/HLE/sceMpegbase.h" +#include "GPU/ge_constants.h" + +#include "unittest/UnitTest.h" + +// The conversion, spelled out. Deliberately the slowest, most obvious thing that could work: it is +// here to disagree with MpegCscRange when MpegCscRange is wrong, so it must not share any of its +// cleverness. +static u32 ReferencePixel(int y, int cbv, int crv, int pixelMode) { + const int c = y - 16, d = cbv - 128, e = crv - 128; + int r = (298 * c + 409 * e + 128) >> 8; + int g = (298 * c - 100 * d - 208 * e + 128) >> 8; + int b = (298 * c + 516 * d + 128) >> 8; + r = r < 0 ? 0 : (r > 255 ? 255 : r); + g = g < 0 ? 0 : (g > 255 ? 255 : g); + b = b < 0 ? 0 : (b > 255 ? 255 : b); + switch (pixelMode) { + case GE_CMODE_16BIT_BGR5650: + return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3); + case GE_CMODE_16BIT_ABGR5551: + return (1 << 15) | ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3); + case GE_CMODE_16BIT_ABGR4444: + return (0xF << 12) | ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4); + default: + return 0xFF000000 | (b << 16) | (g << 8) | r; + } +} + +static void ReferenceCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + const int width2 = width >> 1; + for (int y = 0; y < rangeHeight; y++) { + for (int x = 0; x < rangeWidth; x++) { + const int sx = rangeX + x, sy = rangeY + y; + const int ci = (sy >> 1) * width2 + (sx >> 1); + const u32 pixel = ReferencePixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode); + u8 *out = dest + (y * destStride + x) * bpp; + if (bpp == 4) { + memcpy(out, &pixel, 4); + } else { + const u16 p16 = (u16)pixel; + memcpy(out, &p16, 2); + } + } + } +} + +// A frame with something in every direction: a gradient so neighbouring pixels differ, plus values +// that drive the conversion past both ends of the 0..255 clamp, since that is where an optimized +// version is most likely to disagree. +struct TestFrame { + int width = 0; + int height = 0; + std::vector luma, cb, cr; + + TestFrame(int w, int h) : width(w), height(h) { + luma.resize((size_t)w * h); + cb.resize((size_t)(w / 2) * (h / 2)); + cr.resize((size_t)(w / 2) * (h / 2)); + for (int y = 0; y < h; y++) { + for (int x = 0; x < w; x++) { + luma[(size_t)y * w + x] = (u8)((x * 3 + y * 5) & 0xFF); + } + } + for (int y = 0; y < h / 2; y++) { + for (int x = 0; x < w / 2; x++) { + const size_t i = (size_t)y * (w / 2) + x; + cb[i] = (u8)((x * 7 + y * 2) & 0xFF); + cr[i] = (u8)((x * 2 + y * 11) & 0xFF); + } + } + } +}; + +static const int pixelModes[4] = { + GE_CMODE_16BIT_BGR5650, + GE_CMODE_16BIT_ABGR5551, + GE_CMODE_16BIT_ABGR4444, + GE_CMODE_32BIT_ABGR8888, +}; + +static const char *PixelModeName(int mode) { + switch (mode) { + case GE_CMODE_16BIT_BGR5650: return "5650"; + case GE_CMODE_16BIT_ABGR5551: return "5551"; + case GE_CMODE_16BIT_ABGR4444: return "4444"; + default: return "8888"; + } +} + +static bool CompareAgainstReference(const TestFrame &frame, int pixelMode, + int rangeX, int rangeY, int rangeWidth, int rangeHeight, int destStride) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + // Padded, and prefilled with a value neither implementation would write, so that writing + // outside the range - or short of it - is a failure rather than a coincidence. + const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp; + std::vector got(destSize, 0xCD), want(destSize, 0xCD); + + MpegCscRange(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight); + ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight); + + for (size_t i = 0; i < destSize; i++) { + if (got[i] != want[i]) { + printf(" %s %dx%d at %d,%d stride %d: byte %d is %02x, should be %02x\n", + PixelModeName(pixelMode), rangeWidth, rangeHeight, rangeX, rangeY, destStride, + (int)i, got[i], want[i]); + return false; + } + } + return true; +} + +static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector &dest) { + const int destStride = 512; + // Long enough to swamp the clock's own resolution, short enough not to pad the test run. + const double seconds = 0.2; + int frames = 0; + const double start = time_now_d(); + do { + for (int i = 0; i < 4; i++) { + MpegCscRange(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), frame.width, 0, 0, frame.width, frame.height); + frames++; + } + } while (time_now_d() - start < seconds); + const double elapsed = time_now_d() - start; + return (double)frames * frame.width * frame.height / elapsed / 1000000.0; +} + +bool TestMpegCsc() { + // The size a PSP movie is, so the speed below is the speed that matters. + TestFrame frame(480, 272); + + for (int pixelMode : pixelModes) { + // A whole frame, which is what sceMpegBaseCscAvc asks for. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 0, 480, 272, 512)); + // Partial ranges, as sceMpegBaseCscAvcRange asks for. Odd offsets and sizes on purpose: + // chroma is half resolution, so an odd left edge starts mid-chroma-sample, and an odd + // width leaves a pixel that a two-at-a-time inner loop would have to handle separately. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 16, 16, 64, 32, 512)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 1, 1, 63, 31, 512)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 33, 7, 17, 5, 128)); + // A range reaching the far edge, where reading one sample too far would go off the frame. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 464, 256, 16, 16, 64)); + // One pixel, one row, one column. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 5, 9, 1, 1, 16)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 100, 480, 1, 512)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 100, 0, 1, 272, 16)); + } + + std::vector dest((size_t)512 * 272 * 4, 0); + printf("MpegCscRange, 480x272:\n"); + for (int pixelMode : pixelModes) { + const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest); + // A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it. + printf(" %s: %6.1f MPix/s (%5.2f ms/frame)\n", PixelModeName(pixelMode), mps, + 480.0 * 272.0 / mps / 1000.0); + } + + return true; +} diff --git a/unittest/UnitTest.cpp b/unittest/UnitTest.cpp index df11db0120..4ac480f7c2 100644 --- a/unittest/UnitTest.cpp +++ b/unittest/UnitTest.cpp @@ -2981,6 +2981,7 @@ bool TestThreadManager(); bool TestVFS(); bool TestZipSlip(); bool TestLzrc(); +bool TestMpegCsc(); bool TestDemangle(); // The 8.3 short names games read out of d_private. These aren't verified against hardware yet (no @@ -3200,6 +3201,7 @@ TestItem availableTests[] = { TEST_ITEM(CmdLine), TEST_ITEM(ZipSlip), TEST_ITEM(Lzrc), + TEST_ITEM(MpegCsc), TEST_ITEM(Demangle), TEST_ITEM(TextureReplacer), TEST_ITEM(UITabOrder), diff --git a/unittest/UnitTests.vcxproj b/unittest/UnitTests.vcxproj index edb256fe89..5793f24e1e 100644 --- a/unittest/UnitTests.vcxproj +++ b/unittest/UnitTests.vcxproj @@ -294,6 +294,7 @@ + diff --git a/unittest/UnitTests.vcxproj.filters b/unittest/UnitTests.vcxproj.filters index 8c79b4d58b..37af8831d9 100644 --- a/unittest/UnitTests.vcxproj.filters +++ b/unittest/UnitTests.vcxproj.filters @@ -17,6 +17,7 @@ +