diff --git a/AGENTS.md b/AGENTS.md index 51a5f65db1..812e758400 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -75,9 +75,17 @@ for it: ```powershell $installPath = & "C:\Program Files (x86)\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath $msbuild = "$installPath\MSBuild\Current\Bin\MSBuild.exe" -& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=x64 /m +& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform= /m ``` +- **`` is whatever the machine is - look it up, don't assume.** It is `ARM64` or `x64`, and + the build lands in `Windows\\\` to match, so building one and running the + other is easy to do without noticing. On Windows-on-ARM an x64 build runs anyway, under emulation, + which is what makes it easy to miss: it works, but it is slower than the native build, it is not + the code ARM users get, and any benchmark from it measures the emulator. Get the host from + `python -c "import platform; print(platform.machine())"`, not `$PROCESSOR_ARCHITECTURE`, which + describes the *shell* and says `AMD64` from an emulated one. The binaries say which they are too - + `UnitTest.exe` prints an `ABI:` line at startup. - Kill leftover `PPSSPPHeadless.exe`/`PPSSPP*.exe` instances before building - one holding the exe makes the link fail with `LNK1168`, which looks like a build problem and isn't. - **A stale binary lies consistently.** After a `git stash` cycle that touched a header, do a @@ -90,7 +98,7 @@ UWP, the legacy Android NDK build and the libretro core have their own build sys After a chunk of work (not after every edit), run both suites: -- C++ unit tests: build the `UnitTest` project and run `Windows/x64/Debug/UnitTest.exe all` +- C++ unit tests: build the `UnitTest` project and run `Windows//Debug/UnitTest.exe all` (Linux/Mac: configure with `-DUNITTEST=ON`, run `build/PPSSPPUnitTest all`). Tests are listed in `availableTests` in `unittest/UnitTest.cpp`; pass names instead of `all` to run a subset. - pspautotests (HLE coverage) - run them **exactly the way CI does**: @@ -103,8 +111,10 @@ python test.py -g --graphics=software around a hundred failures that mean nothing is wrong. The only meaningful result is `0 tests failed`. (The debug-CRT "Detected memory leaks!" dump after the summary line is normal, not a failure.) -New unit tests are added to `availableTests`; large ones go in their own file in `unittest/`, listed in -both CMakeLists.txt and the Visual Studio project. +New unit tests are added to `availableTests`; large ones go in their own file in `unittest/`, which has +to be listed in **three** build files, not two: `CMakeLists.txt`, `unittest/UnitTests.vcxproj` (and its +`.filters`), and `android/jni/Android.mk`, which builds a unit test executable of its own. Miss the last +one and it builds everywhere you can easily try it, and fails on Android CI. ## Multiplatform considerations diff --git a/CMakeLists.txt b/CMakeLists.txt index a936e3afcd..c81d415524 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1368,6 +1368,7 @@ if(UNITTEST) unittest/TestVFS.cpp unittest/TestZipSlip.cpp unittest/TestLzrc.cpp + unittest/TestMpegCsc.cpp unittest/TestDemangle.cpp unittest/TestTextureReplacer.cpp unittest/TestRiscVEmitter.cpp diff --git a/Core/HLE/sceKernel.cpp b/Core/HLE/sceKernel.cpp index 86241a7239..efc63c1b85 100644 --- a/Core/HLE/sceKernel.cpp +++ b/Core/HLE/sceKernel.cpp @@ -144,6 +144,7 @@ void __KernelInit() __PowerInit(); __UtilityInit(); __UmdInit(); + __MpegBaseInit(); __MpegInit(); __PsmfInit(); __CtrlInit(); @@ -210,6 +211,7 @@ void __KernelShutdown() __Mp3Shutdown(); __MpegShutdown(); + __MpegBaseShutdown(); __PsmfShutdown(); __PPGeShutdown(); diff --git a/Core/HLE/sceMpeg.cpp b/Core/HLE/sceMpeg.cpp index 70e4c065a0..5871db61e9 100644 --- a/Core/HLE/sceMpeg.cpp +++ b/Core/HLE/sceMpeg.cpp @@ -356,7 +356,6 @@ static void ClearMpegContexts() { } void __MpegInit() { - __MpegBaseInit(); // getMpegCtx keys on a handle read out of game memory, so don't leave contexts from a previous // game around for the next one to find. ClearMpegContexts(); diff --git a/Core/HLE/sceMpegbase.cpp b/Core/HLE/sceMpegbase.cpp index 7167ec9f86..12c02560f4 100644 --- a/Core/HLE/sceMpegbase.cpp +++ b/Core/HLE/sceMpegbase.cpp @@ -20,7 +20,10 @@ // mpeg.prx drives these directly, so they have to be real for the firmware module to run in place // of our sceMpeg HLE. +#include +#include #include +#include #include #include "Common/Serialize/Serializer.h" @@ -37,6 +40,13 @@ #include "GPU/GPUState.h" #include "GPU/ge_constants.h" +#ifdef USE_FFMPEG +extern "C" { +#include "libswscale/swscale.h" +#include "libavutil/pixfmt.h" +} +#endif + // The PES payloads gathered by sceMpegBasePESpacketCopy, keyed by the destination each was // copied to. It carries audio as well as video - the destination is what tells them apart - so // sceVideocodec has to ask for the one matching the address it was handed. @@ -45,13 +55,30 @@ static std::map> g_pesPackets; static int g_mpegBaseBufferWidth = 512; static int g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888; +// Scratch for the planes the de-tiling produces. Reused between calls: a movie converts one of +// these every frame, and they are a couple of hundred kilobytes, so allocating them per call was +// pure overhead. Nothing here needs saving - it is rebuilt from the ME's buffers on every call. +static std::vector g_untileScratch; + void __MpegBaseInit() { // None of this survives a boot on hardware. g_pesPackets.clear(); + g_untileScratch.clear(); + g_untileScratch.shrink_to_fit(); + MpegCscShutdown(); g_mpegBaseBufferWidth = 512; g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888; } +void __MpegBaseShutdown() { + // The scratch and the swscale context are worth a few hundred kilobytes between them, and a + // game that played one video early on has no use for either afterwards. + g_pesPackets.clear(); + g_untileScratch.clear(); + g_untileScratch.shrink_to_fit(); + MpegCscShutdown(); +} + void __MpegBaseDoState(PointerWrap &p) { auto s = p.Section("sceMpegbase", 0, 1); if (!s) { @@ -174,40 +201,22 @@ static const u8 *MpegBaseFramePointer(u32 addr, int size) { // // Untangling it into plain planes costs one pass per frame, which keeps the conversion below // readable and is not where the time goes. -bool ReadTiledYCbCr(const u32 *buffers, int width, int height, - std::vector &luma, std::vector &cb, std::vector &cr) { +// The de-tiling itself, with the address resolution left outside so it can be measured and +// checked on its own - see TestMpegCsc. src is the eight buffers in sceVideocodec order (four +// luma, then four chroma) and sizes says how big each one is. +// +// Every byte of the output is written for any frame the hardware can produce, so the caller does +// not have to clear it first. +void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8], + int width, int height) { const int width2 = width >> 1; const int height2 = height >> 1; - - int sizes[8]; - VideocodecFrameBufferLayout(width, height, sizes, nullptr); const int *ySize = sizes; const int *cSize = sizes + 4; - const u8 *y[4] = {}; - const u8 *c[4] = {}; - for (int i = 0; i < 4; i++) { - if (ySize[i] > 0) { - y[i] = MpegBaseFramePointer(buffers[i], ySize[i]); - if (!y[i]) { - return false; - } - } - if (cSize[i] > 0) { - c[i] = MpegBaseFramePointer(buffers[4 + i], cSize[i]); - if (!c[i]) { - return false; - } - } - } - - luma.assign((size_t)width * height, 0); - cb.assign((size_t)width2 * height2, 128); - cr.assign((size_t)width2 * height2, 128); - // Luma: four buffers, keyed by (left/right half of the band, even/odd row). for (int b = 0; b < 4; b++) { - if (!y[b]) { + if (!src[b]) { continue; } const int xOffset = (b & 1) ? 16 : 0; @@ -219,33 +228,69 @@ bool ReadTiledYCbCr(const u32 *buffers, int width, int height, if (run <= 0 || j + run > ySize[b]) { continue; } - memcpy(&luma[(size_t)row * width + bandX], y[b] + j, run); + memcpy(luma + (size_t)row * width + bandX, src[b] + j, run); } } } - // Chroma: same shape in half-resolution coordinates, with interleaved Cb/Cr pairs. + // Chroma: same shape in half-resolution coordinates, but the two planes arrive interleaved as + // (Cb,Cr) pairs, so each group of 8 pixels is a 16-byte run to pull apart. The bounds the old + // version checked per pixel only depend on the group, so they are hoisted out here - that inner + // loop was the expensive half of this function. for (int b = 0; b < 4; b++) { - if (!c[b]) { + if (!src[b + 4]) { continue; } const int xOffset = (b & 1) ? 8 : 0; const int yStart = (b >> 1) ? 1 : 0; int j = 0; for (int bandX = xOffset; bandX < width2; bandX += 16) { - for (int row = yStart; row < height2; row += 2) { - for (int k = 0; k < 8; k++, j += 2) { - const int x = bandX + k; - if (x >= width2 || j + 1 >= cSize[b]) { - continue; - } - const size_t i = (size_t)row * width2 + x; - cb[i] = c[b][j]; - cr[i] = c[b][j + 1]; + for (int row = yStart; row < height2; row += 2, j += 16) { + // How many of the 8 fit both in the row and in what the buffer actually holds. + const int fits = std::min(8, (cSize[b] - j) >> 1); + const int run = std::min(width2 - bandX, fits); + if (run <= 0) { + continue; + } + const u8 *from = src[b + 4] + j; + u8 *toCb = cb + (size_t)row * width2 + bandX; + u8 *toCr = cr + (size_t)row * width2 + bandX; + for (int k = 0; k < run; k++) { + toCb[k] = from[k * 2]; + toCr[k] = from[k * 2 + 1]; } } } } +} + +bool ReadTiledYCbCr(const u32 *buffers, int width, int height, + const u8 **luma, const u8 **cb, const u8 **cr) { + int sizes[8]; + VideocodecFrameBufferLayout(width, height, sizes, nullptr); + + const u8 *src[8]{}; + for (int i = 0; i < 8; i++) { + if (sizes[i] > 0) { + src[i] = MpegBaseFramePointer(buffers[i], sizes[i]); + if (!src[i]) { + return false; + } + } + } + + const size_t lumaBytes = (size_t)width * height; + const size_t chromaBytes = (size_t)(width >> 1) * (height >> 1); + if (g_untileScratch.size() < lumaBytes + chromaBytes * 2) { + g_untileScratch.resize(lumaBytes + chromaBytes * 2); + } + u8 *l = g_untileScratch.data(); + u8 *b = l + lumaBytes; + u8 *r = b + chromaBytes; + UntileYCbCr(l, b, r, src, sizes, width, height); + *luma = l; + *cb = b; + *cr = r; return true; } @@ -257,18 +302,174 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) { r = std::min(255, std::max(0, r)); g = std::min(255, std::max(0, g)); b = std::min(255, std::max(0, b)); + // Alpha comes out zero, not opaque. That is what the hardware does - our sceMpeg HLE masks it + // off for the same reason, and names Sword Art Online as a game that depends on it, because it + // doesn't clear the alpha in the buffer it hands over and expects the video to leave it clear. switch (pixelMode) { case GE_CMODE_16BIT_BGR5650: return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3); case GE_CMODE_16BIT_ABGR5551: - return (1 << 15) | ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3); + return ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3); case GE_CMODE_16BIT_ABGR4444: - return (0xF << 12) | ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4); + return ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4); default: - return 0xFF000000 | (b << 16) | (g << 8) | r; + return (b << 16) | (g << 8) | r; } } +// The conversion itself, with nothing around it. Pure, so that TestMpegCsc can measure it and +// check it - it is the hottest thing in video playback, and the point of having it out here is +// that it can be worked on without a game in the loop. +// +// luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is +// destStride pixels wide in the format pixelMode names, and the converted range always lands at +// its origin. +void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + const int width2 = width >> 1; + for (int y = 0; y < rangeHeight; y++) { + const int sy = rangeY + y; + for (int x = 0; x < rangeWidth; x++) { + const int sx = rangeX + x; + const int ci = (sy >> 1) * width2 + (sx >> 1); + const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode); + if (bpp == 4) { + memcpy(dest + (y * destStride + x) * 4, &pixel, 4); + } else { + const u16 p16 = (u16)pixel; + memcpy(dest + (y * destStride + x) * 2, &p16, 2); + } + } + } +} + +#ifdef USE_FFMPEG + +// swscale is what our sceMpeg HLE converts with, and the planes the de-tiling produces are +// already the YUV420P it wants, so the same thing works here - and it is a great deal quicker +// than doing it a pixel at a time. +// +// The four output formats are the ones MediaEngine::getSwsFormat picks, for the same reasons. +// Alpha is not among them: swscale writes RGBA opaque and leaves the spare bits of the 16-bit +// formats clear, so the masking below is what makes the result match the hardware, exactly as +// the HLE does after its own sws_scale. +static AVPixelFormat SwsFormatForPixelMode(int pixelMode) { + switch (pixelMode) { + case GE_CMODE_16BIT_BGR5650: return AV_PIX_FMT_BGR565LE; + case GE_CMODE_16BIT_ABGR5551: return AV_PIX_FMT_BGR555LE; + case GE_CMODE_16BIT_ABGR4444: return AV_PIX_FMT_BGR444LE; + default: return AV_PIX_FMT_RGBA; + } +} + +// Nothing is being scaled here, so this only picks how chroma reaches full resolution: SWS_POINT +// repeats each 2x2 block's sample, as the scalar path and presumably the hardware do, while +// SWS_BILINEAR smooths between samples, as our sceMpeg HLE does. Swap the line to taste - it +// deserves a real option eventually. +static const int MPEG_CSC_SWS_FLAGS = SWS_POINT; + +static SwsContext *g_cscSws; +static int g_cscSwsWidth, g_cscSwsHeight, g_cscSwsFormat = -1; + +void MpegCscShutdown() { + if (g_cscSws) { + sws_freeContext(g_cscSws); + g_cscSws = nullptr; + } + g_cscSwsWidth = 0; + g_cscSwsHeight = 0; + g_cscSwsFormat = -1; +} + +bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + // Chroma is half resolution, so an odd origin would start half a sample in and there is no way + // to say that to swscale. Nothing can actually ask for one - the ranges arrive in macroblocks - + // but the scalar path is still there for it. + if ((rangeX & 1) || (rangeY & 1)) { + return false; + } + + const AVPixelFormat format = SwsFormatForPixelMode(pixelMode); + if (rangeWidth != g_cscSwsWidth || rangeHeight != g_cscSwsHeight || (int)format != g_cscSwsFormat) { + g_cscSws = sws_getCachedContext(g_cscSws, rangeWidth, rangeHeight, AV_PIX_FMT_YUV420P, + rangeWidth, rangeHeight, format, MPEG_CSC_SWS_FLAGS, nullptr, nullptr, nullptr); + if (!g_cscSws) { + return false; + } + // Studio swing both ways, which is the range the coefficients in the scalar path assume. + int *invCoeff, *coeff, srcRange, dstRange, brightness, contrast, saturation; + if (sws_getColorspaceDetails(g_cscSws, &invCoeff, &srcRange, &coeff, &dstRange, &brightness, + &contrast, &saturation) != -1) { + sws_setColorspaceDetails(g_cscSws, invCoeff, 0, coeff, 0, brightness, contrast, saturation); + } + g_cscSwsWidth = rangeWidth; + g_cscSwsHeight = rangeHeight; + g_cscSwsFormat = (int)format; + } + + const int width2 = width >> 1; + const u8 *srcSlice[4] = { + luma + (size_t)rangeY * width + rangeX, + cb + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1), + cr + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1), + nullptr, + }; + const int srcStride[4] = { width, width2, width2, 0 }; + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + u8 *dstSlice[4] = { dest, nullptr, nullptr, nullptr }; + const int dstStride[4] = { destStride * bpp, 0, 0, 0 }; + if (sws_scale(g_cscSws, srcSlice, srcStride, 0, rangeHeight, dstSlice, dstStride) <= 0) { + return false; + } + + // Clear the alpha swscale filled in, which the hardware leaves at zero. + for (int y = 0; y < rangeHeight; y++) { + u8 *row = dest + (size_t)y * destStride * bpp; + if (bpp == 4) { + u32_le *p32 = (u32_le *)row; + for (int x = 0; x < rangeWidth; x++) { + p32[x] = p32[x] & 0x00FFFFFF; + } + } else if (pixelMode != GE_CMODE_16BIT_BGR5650) { + const u16 mask = pixelMode == GE_CMODE_16BIT_ABGR5551 ? 0x7FFF : 0x0FFF; + u16_le *p16 = (u16_le *)row; + for (int x = 0; x < rangeWidth; x++) { + p16[x] = p16[x] & mask; + } + } + } + return true; +} + +#else // !USE_FFMPEG + +bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + return false; +} + +void MpegCscShutdown() {} + +#endif // USE_FFMPEG + +void MpegCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { +#ifdef USE_FFMPEG + if (MpegCscRangeSws(dest, destStride, pixelMode, luma, cb, cr, width, + rangeX, rangeY, rangeWidth, rangeHeight)) { + return; + } +#endif + MpegCscRangeScalar(dest, destStride, pixelMode, luma, cb, cr, width, + rangeX, rangeY, rangeWidth, rangeHeight); +} + // The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the // latter over the whole frame. static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth, @@ -303,8 +504,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth, return hleLogError(Log::Mpeg, -1, "range outside the frame"); } - std::vector luma, cb, cr; - if (!ReadTiledYCbCr(buffers, width, height, luma, cb, cr)) { + const u8 *luma, *cb, *cr; + if (!ReadTiledYCbCr(buffers, width, height, &luma, &cb, &cr)) { return hleLogError(Log::Mpeg, -1, "YCbCr buffers not readable"); } @@ -318,21 +519,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth, return hleLogError(Log::Mpeg, -1, "output buffer not writable"); } - const int width2 = width >> 1; - for (int y = 0; y < rangeHeight; y++) { - const int sy = rangeY + y; - for (int x = 0; x < rangeWidth; x++) { - const int sx = rangeX + x; - const int ci = (sy >> 1) * width2 + (sx >> 1); - const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], g_mpegBasePixelMode); - if (bpp == 4) { - memcpy(dest + (y * bufferWidth + x) * 4, &pixel, 4); - } else { - const u16 p16 = (u16)pixel; - memcpy(dest + (y * bufferWidth + x) * 2, &p16, 2); - } - } - } + MpegCscRange(dest, bufferWidth, g_mpegBasePixelMode, luma, cb, cr, width, + rangeX, rangeY, rangeWidth, rangeHeight); NotifyMemInfo(MemBlockFlags::WRITE, bufferRGB, destSize, "MpegBaseCsc"); // The CPU just wrote a video frame into what is usually a display buffer. The hardware backends // don't see that on their own, so without telling them the screen keeps showing the last frame diff --git a/Core/HLE/sceMpegbase.h b/Core/HLE/sceMpegbase.h index cf85d1a4ca..2ce7fcc395 100644 --- a/Core/HLE/sceMpegbase.h +++ b/Core/HLE/sceMpegbase.h @@ -25,8 +25,9 @@ class PointerWrap; void Register_sceMpegbase(); -// Called per boot, from __MpegInit. +// Called per boot, from __KernelInit and __KernelShutdown, around sceMpeg's own pair. void __MpegBaseInit(); +void __MpegBaseShutdown(); void __MpegBaseDoState(PointerWrap &p); @@ -39,5 +40,36 @@ std::vector MpegBaseTakePESPacket(u32 dest); // Un-tiles a decoded frame from the eight buffers the Media Engine lays it out in into three // planes. The buffers are in sceVideocodec's order: four luma, then four chroma. cb and cr come // out at half width and half height, as YUV420 does. +// +// The planes point into scratch that is reused by the next call, so read them before calling again. bool ReadTiledYCbCr(const u32 *buffers, int width, int height, - std::vector &luma, std::vector &cb, std::vector &cr); + const u8 **luma, const u8 **cb, const u8 **cr); + +// The de-tiling on its own, taking the eight buffers already resolved to host pointers, so it can +// be measured and checked without a game - see TestMpegCsc. A null entry in src leaves that +// buffer's share of the output alone. +void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8], + int width, int height); + +// Converts a rectangle of a planar YCbCr420 frame to RGB, the way the DMACPLUS does on the way to +// the screen. Pure, so it can be measured and checked on its own - see TestMpegCsc. +// +// luma is width by height; cb and cr are half that in both directions. dest is destStride pixels +// wide in the format pixelMode names (a GEBufferFormat), and the range lands at its origin. +void MpegCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight); + +// The two implementations behind it, exposed so TestMpegCsc can measure and compare them. +// The scalar one handles anything; the swscale one refuses what it cannot express and is then +// not used. They do not agree to the bit - swscale rounds its own way - so the scalar one is +// what the reference in the test is checked against. +void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight); +bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight); + +// Frees the cached swscale context. +void MpegCscShutdown(); diff --git a/Core/HLE/sceVideocodec.cpp b/Core/HLE/sceVideocodec.cpp index f99dffa34f..0c109e25f1 100644 --- a/Core/HLE/sceVideocodec.cpp +++ b/Core/HLE/sceVideocodec.cpp @@ -26,7 +26,9 @@ // by looking at sceMpegBaseYCrCbCopy output on a real PSP. #include +#include #include +#include #include #include "Common/Serialize/Serializer.h" @@ -682,8 +684,8 @@ static int sceVideocodecCopyYCbCr(u32 ctxAddr, int type) { buffers[fromDescriptor[i]] = Memory::ReadUnchecked_U32(ctxAddr + 0x0c + i * 4); } - std::vector luma, cb, cr; - if (!ReadTiledYCbCr(buffers, width, height, luma, cb, cr)) { + const u8 *luma, *cb, *cr; + if (!ReadTiledYCbCr(buffers, width, height, &luma, &cb, &cr)) { return hleLogError(Log::ME, -1, "YCbCr buffers not readable"); } @@ -692,13 +694,17 @@ static int sceVideocodecCopyYCbCr(u32 ctxAddr, int type) { Memory::ReadUnchecked_U32(ctxAddr + 0x30), Memory::ReadUnchecked_U32(ctxAddr + 0x34), }; - const std::vector *planes[3] = { &luma, &cb, &cr }; + const u8 *planes[3] = { luma, cb, cr }; + const u32 planeSizes[3] = { + (u32)(width * height), + (u32)((width >> 1) * (height >> 1)), + (u32)((width >> 1) * (height >> 1)), + }; for (int i = 0; i < 3; i++) { - const u32 size = (u32)planes[i]->size(); - if (!Memory::IsValidRange(dst[i], size)) { - return hleLogError(Log::ME, -1, "plane %d (%08x, %d bytes) not writable", i, dst[i], size); + if (!Memory::IsValidRange(dst[i], planeSizes[i])) { + return hleLogError(Log::ME, -1, "plane %d (%08x, %d bytes) not writable", i, dst[i], planeSizes[i]); } - Memory::MemcpyUnchecked(dst[i], planes[i]->data(), size); + Memory::MemcpyUnchecked(dst[i], planes[i], planeSizes[i]); } return hleLogDebug(Log::ME, 0, "%dx%d -> %08x %08x %08x", width, height, dst[0], dst[1], dst[2]); } diff --git a/android/jni/Android.mk b/android/jni/Android.mk index 63cff285ec..2e0a7ab5a9 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -1060,6 +1060,7 @@ ifeq ($(UNITTEST),1) $(SRC)/unittest/TestVFS.cpp \ $(SRC)/unittest/TestDemangle.cpp \ $(SRC)/unittest/TestLzrc.cpp \ + $(SRC)/unittest/TestMpegCsc.cpp \ $(SRC)/unittest/TestZipSlip.cpp \ $(SRC)/unittest/UnitTest.cpp diff --git a/docs/building.md b/docs/building.md index af90594c91..f763be1f5a 100644 --- a/docs/building.md +++ b/docs/building.md @@ -16,16 +16,24 @@ An agent can drive the VS solution non-interactively with `MSBuild.exe` instead ```powershell $installPath = & "C:\Program Files (x86)\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath $msbuild = "$installPath\MSBuild\Current\Bin\MSBuild.exe" -& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=x64 /m +& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform= /m ``` (swap `/t:UnitTest` for `/t:PPSSPPWindows` or another project name as needed; drop it entirely to build the whole solution). +`` is `ARM64` or `x64` - whichever the machine actually is, so look it up rather than +picking a default. The output directory follows it (`Windows\\\`), which +makes building one and running the other an easy mistake. It is easiest to make on Windows-on-ARM, +where an x64 build runs anyway under emulation: everything appears to work, but it is slower than +the native build and any performance measurement from it describes the emulator rather than the +code. `platform.machine()` in Python reports the host; `$PROCESSOR_ARCHITECTURE` reports the shell, +which is `AMD64` in an emulated shell even on an ARM64 machine. + In addition to the pspautotests runner (test.py), there is a separate binary with C++ unit tests in the /unittest subdirectory. After substantial changes (at the end of a chunk of work, not necessarily after every edit), run these too: -- Windows: build the `UnitTest` project (unittest/UnitTests.vcxproj), then run `Windows/x64/Debug/UnitTest.exe all` +- Windows: build the `UnitTest` project (unittest/UnitTests.vcxproj), then run `Windows//Debug/UnitTest.exe all` (`` being `ARM64` or `x64`, whichever you built) - Linux/Mac: configure with `-DUNITTEST=ON`, then run `build/PPSSPPUnitTest all` This runs all tests in `availableTests` in unittest/UnitTest.cpp. You can run one or more @@ -127,7 +135,10 @@ main functions (and also stub out most of the System_ functions as needed). Take when making cross platform changes. New unit tests are added by listing them in availableTests in unittest.cpp. If they are large, put them in -separate files in the unittest subdirectory. Remember to update both CMakeLists.txt and the visual studio project. +separate files in the unittest subdirectory. A new file has to be listed in three build files, not two: +`CMakeLists.txt`, `unittest/UnitTests.vcxproj` (and its `.filters`), and `android/jni/Android.mk`, which +builds a unit test executable of its own. The Android one is the easiest to forget, since missing it builds +fine everywhere you are likely to try it and only fails on Android CI. A unit test is often the first thing to call a given function from outside its own .cpp, which makes the `ppsspp_unittest` target in the legacy Android build (`android/jni/Android.mk`, see above) the strictest check diff --git a/test.py b/test.py index 1246f5ed54..23025d1638 100755 --- a/test.py +++ b/test.py @@ -7,12 +7,18 @@ import os import subprocess import threading import glob +import platform PPSSPP_EXECUTABLES = [ - # Windows + # Windows. The machine's own architecture comes first: an x64 build runs on Windows-on-ARM too, + # under emulation, so looking for it first would quietly test the emulated build instead. "Windows\\Debug\\PPSSPPHeadless.exe", "Windows\\Release\\PPSSPPHeadless.exe", +] + ([ + "Windows\\ARM64\\Debug\\PPSSPPHeadless.exe", + "Windows\\ARM64\\Release\\PPSSPPHeadless.exe", +] if platform.machine().lower() in ("arm64", "aarch64") else []) + [ "Windows\\x64\\Debug\\PPSSPPHeadless.exe", "Windows\\x64\\Release\\PPSSPPHeadless.exe", "build*/PPSSPPHeadless.exe", diff --git a/unittest/TestMpegCsc.cpp b/unittest/TestMpegCsc.cpp new file mode 100644 index 0000000000..5e20b7d40b --- /dev/null +++ b/unittest/TestMpegCsc.cpp @@ -0,0 +1,379 @@ +// Copyright (c) 2012- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +// Correctness and speed of the Media Engine's colour conversion, which is the hottest thing in +// video playback once sceMpeg runs the real mpeg.prx - sceMpegBaseCscAvc sits at the top of a +// profile of a movie. +// +// The correctness half is a reference implementation written out longhand, so an optimized +// MpegCscRange has something to be wrong against that isn't itself. The speed half reports +// megapixels per second for a 480x272 frame, the size a PSP movie actually is. + +#include +#include +#include +#include + +#include "Common/CommonTypes.h" +#include "Common/TimeUtil.h" +#include "Core/HLE/sceMpegbase.h" +#include "Core/HLE/sceVideocodec.h" +#include "GPU/ge_constants.h" + +#include "unittest/UnitTest.h" + +// The conversion, spelled out. Deliberately the slowest, most obvious thing that could work: it is +// here to disagree with MpegCscRange when MpegCscRange is wrong, so it must not share any of its +// cleverness. +static u32 ReferencePixel(int y, int cbv, int crv, int pixelMode) { + const int c = y - 16, d = cbv - 128, e = crv - 128; + int r = (298 * c + 409 * e + 128) >> 8; + int g = (298 * c - 100 * d - 208 * e + 128) >> 8; + int b = (298 * c + 516 * d + 128) >> 8; + r = r < 0 ? 0 : (r > 255 ? 255 : r); + g = g < 0 ? 0 : (g > 255 ? 255 : g); + b = b < 0 ? 0 : (b > 255 ? 255 : b); + // Alpha zero, matching the hardware - see the note in sceMpegbase.cpp. + switch (pixelMode) { + case GE_CMODE_16BIT_BGR5650: + return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3); + case GE_CMODE_16BIT_ABGR5551: + return ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3); + case GE_CMODE_16BIT_ABGR4444: + return ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4); + default: + return (b << 16) | (g << 8) | r; + } +} + +static void ReferenceCscRange(u8 *dest, int destStride, int pixelMode, + const u8 *luma, const u8 *cb, const u8 *cr, int width, + int rangeX, int rangeY, int rangeWidth, int rangeHeight) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + const int width2 = width >> 1; + for (int y = 0; y < rangeHeight; y++) { + for (int x = 0; x < rangeWidth; x++) { + const int sx = rangeX + x, sy = rangeY + y; + const int ci = (sy >> 1) * width2 + (sx >> 1); + const u32 pixel = ReferencePixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode); + u8 *out = dest + (y * destStride + x) * bpp; + if (bpp == 4) { + memcpy(out, &pixel, 4); + } else { + const u16 p16 = (u16)pixel; + memcpy(out, &p16, 2); + } + } + } +} + +// A frame with something in every direction: a gradient so neighbouring pixels differ, plus values +// that drive the conversion past both ends of the 0..255 clamp, since that is where an optimized +// version is most likely to disagree. +struct TestFrame { + int width = 0; + int height = 0; + std::vector luma, cb, cr; + + TestFrame(int w, int h) : width(w), height(h) { + luma.resize((size_t)w * h); + cb.resize((size_t)(w / 2) * (h / 2)); + cr.resize((size_t)(w / 2) * (h / 2)); + for (int y = 0; y < h; y++) { + for (int x = 0; x < w; x++) { + luma[(size_t)y * w + x] = (u8)((x * 3 + y * 5) & 0xFF); + } + } + for (int y = 0; y < h / 2; y++) { + for (int x = 0; x < w / 2; x++) { + const size_t i = (size_t)y * (w / 2) + x; + cb[i] = (u8)((x * 7 + y * 2) & 0xFF); + cr[i] = (u8)((x * 2 + y * 11) & 0xFF); + } + } + } +}; + +static const int pixelModes[4] = { + GE_CMODE_16BIT_BGR5650, + GE_CMODE_16BIT_ABGR5551, + GE_CMODE_16BIT_ABGR4444, + GE_CMODE_32BIT_ABGR8888, +}; + +static const char *PixelModeName(int mode) { + switch (mode) { + case GE_CMODE_16BIT_BGR5650: return "5650"; + case GE_CMODE_16BIT_ABGR5551: return "5551"; + case GE_CMODE_16BIT_ABGR4444: return "4444"; + default: return "8888"; + } +} + +static bool CompareAgainstReference(const TestFrame &frame, int pixelMode, + int rangeX, int rangeY, int rangeWidth, int rangeHeight, int destStride) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + // Padded, and prefilled with a value neither implementation would write, so that writing + // outside the range - or short of it - is a failure rather than a coincidence. + const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp; + std::vector got(destSize, 0xCD), want(destSize, 0xCD); + + MpegCscRangeScalar(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight); + ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight); + + for (size_t i = 0; i < destSize; i++) { + if (got[i] != want[i]) { + printf(" %s %dx%d at %d,%d stride %d: byte %d is %02x, should be %02x\n", + PixelModeName(pixelMode), rangeWidth, rangeHeight, rangeX, rangeY, destStride, + (int)i, got[i], want[i]); + return false; + } + } + return true; +} + +typedef void (*CscFunc)(u8 *, int, int, const u8 *, const u8 *, const u8 *, int, int, int, int, int); + +static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector &dest, + CscFunc fn = &MpegCscRange) { + const int destStride = 512; + // Long enough to swamp the clock's own resolution, short enough not to pad the test run. + const double seconds = 0.2; + int frames = 0; + const double start = time_now_d(); + do { + for (int i = 0; i < 4; i++) { + fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), frame.width, 0, 0, frame.width, frame.height); + frames++; + } + } while (time_now_d() - start < seconds); + const double elapsed = time_now_d() - start; + return (double)frames * frame.width * frame.height / elapsed / 1000000.0; +} + +// The de-tiling as it was originally written, straight from the description of the layout: bounds +// checked per pixel, chroma pulled apart one byte at a time. Kept as the thing UntileYCbCr has to +// agree with, and as something to measure it against. +static void ReferenceUntile(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8], + int width, int height) { + const int width2 = width >> 1, height2 = height >> 1; + const int *ySize = sizes; + const int *cSize = sizes + 4; + for (int b = 0; b < 4; b++) { + if (!src[b]) { + continue; + } + const int xOffset = (b & 1) ? 16 : 0; + const int yStart = (b >> 1) ? 1 : 0; + int j = 0; + for (int bandX = xOffset; bandX < width; bandX += 32) { + const int run = width - bandX < 16 ? width - bandX : 16; + for (int row = yStart; row < height; row += 2, j += 16) { + if (run <= 0 || j + run > ySize[b]) { + continue; + } + memcpy(luma + (size_t)row * width + bandX, src[b] + j, run); + } + } + } + for (int b = 0; b < 4; b++) { + if (!src[b + 4]) { + continue; + } + const int xOffset = (b & 1) ? 8 : 0; + const int yStart = (b >> 1) ? 1 : 0; + int j = 0; + for (int bandX = xOffset; bandX < width2; bandX += 16) { + for (int row = yStart; row < height2; row += 2) { + for (int k = 0; k < 8; k++, j += 2) { + const int x = bandX + k; + if (x >= width2 || j + 1 >= cSize[b]) { + continue; + } + const size_t i = (size_t)row * width2 + x; + cb[i] = src[b + 4][j]; + cr[i] = src[b + 4][j + 1]; + } + } + } + } +} + +// The eight buffers the Media Engine would have produced, laid out as UntileYCbCr expects. The +// contents do not matter for speed and the de-tiling is a pure shuffle, so any pattern will do - +// but make it vary so a broken copy is visible. +struct TiledFrame { + std::vector storage[8]; + const u8 *src[8]{}; + int sizes[8]{}; + + TiledFrame(int width, int height) { + VideocodecFrameBufferLayout(width, height, sizes, nullptr); + for (int i = 0; i < 8; i++) { + storage[i].resize(sizes[i] > 0 ? sizes[i] : 1); + for (int j = 0; j < sizes[i]; j++) { + storage[i][j] = (u8)((j * 7 + i * 31) & 0xFF); + } + src[i] = sizes[i] > 0 ? storage[i].data() : nullptr; + } + } +}; + +typedef void (*UntileFunc)(u8 *, u8 *, u8 *, const u8 *const[8], const int[8], int, int); + +static double MeasureUntileMegapixelsPerSecond(const TiledFrame &tiled, int width, int height, + std::vector &planes, UntileFunc fn = &UntileYCbCr) { + u8 *luma = planes.data(); + u8 *cb = luma + (size_t)width * height; + u8 *cr = cb + (size_t)(width / 2) * (height / 2); + const double seconds = 0.2; + int frames = 0; + const double start = time_now_d(); + do { + for (int i = 0; i < 4; i++) { + fn(luma, cb, cr, tiled.src, tiled.sizes, width, height); + frames++; + } + } while (time_now_d() - start < seconds); + const double elapsed = time_now_d() - start; + return (double)frames * width * height / elapsed / 1000000.0; +} + +bool TestMpegCsc() { + // The size a PSP movie is, so the speed below is the speed that matters. + TestFrame frame(480, 272); + + for (int pixelMode : pixelModes) { + // A whole frame, which is what sceMpegBaseCscAvc asks for. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 0, 480, 272, 512)); + // Partial ranges, as sceMpegBaseCscAvcRange asks for. Odd offsets and sizes on purpose: + // chroma is half resolution, so an odd left edge starts mid-chroma-sample, and an odd + // width leaves a pixel that a two-at-a-time inner loop would have to handle separately. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 16, 16, 64, 32, 512)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 1, 1, 63, 31, 512)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 33, 7, 17, 5, 128)); + // A range reaching the far edge, where reading one sample too far would go off the frame. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 464, 256, 16, 16, 64)); + // One pixel, one row, one column. + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 5, 9, 1, 1, 16)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 100, 480, 1, 512)); + EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 100, 0, 1, 272, 16)); + } + + // De-tiling, which runs once per frame ahead of the conversion. + { + const size_t planeBytes = (size_t)480 * 272 + (size_t)240 * 136 * 2; + // Odd frame sizes as well as the real one: the guards in here are about buffers that don't + // divide evenly into bands, which is the only thing that makes them fire. + for (auto dims : { std::make_pair(480, 272), std::make_pair(64, 32), std::make_pair(48, 16) }) { + const int w = dims.first, h = dims.second; + TiledFrame tiled(w, h); + std::vector got((size_t)w * h + (size_t)(w / 2) * (h / 2) * 2, 0xCD); + std::vector want(got.size(), 0xCD); + u8 *gl = got.data(), *gb = gl + (size_t)w * h, *gr = gb + (size_t)(w / 2) * (h / 2); + u8 *wl = want.data(), *wb = wl + (size_t)w * h, *wr = wb + (size_t)(w / 2) * (h / 2); + UntileYCbCr(gl, gb, gr, tiled.src, tiled.sizes, w, h); + ReferenceUntile(wl, wb, wr, tiled.src, tiled.sizes, w, h); + EXPECT_TRUE(got == want); + // Nothing left at the fill value: the de-tiling covers every byte of a frame, which is + // what lets the caller skip clearing the planes first. A real frame could contain the + // fill byte by chance, so this is looking for whole rows left behind, not exact cover. + size_t untouched = 0; + for (u8 v : got) { + if (v == 0xCD) { + untouched++; + } + } + EXPECT_TRUE(untouched < got.size() / 100); + } + + TiledFrame tiled(480, 272); + std::vector planes(planeBytes, 0); + const double mps = MeasureUntileMegapixelsPerSecond(tiled, 480, 272, planes); + const double refMps = MeasureUntileMegapixelsPerSecond(tiled, 480, 272, planes, ReferenceUntile); + printf("UntileYCbCr, 480x272: %6.1f MPix/s (%5.2f ms/frame), was %6.1f (%5.2f ms)\n", + mps, 480.0 * 272.0 / mps / 1000.0, refMps, 480.0 * 272.0 / refMps / 1000.0); + } + + std::vector dest((size_t)512 * 272 * 4, 0); + // How far swscale lands from the conversion written out longhand. It rounds its own way, so + // this is not expected to be zero - the question is whether it is close enough to use. + printf("swscale against the reference, per channel:\n"); + for (int pixelMode : pixelModes) { + const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2; + std::vector sws((size_t)512 * 272 * 4, 0), ref((size_t)512 * 272 * 4, 0); + if (!MpegCscRangeSws(sws.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), 480, 0, 0, 480, 272)) { + printf(" %s: declined\n", PixelModeName(pixelMode)); + continue; + } + ReferenceCscRange(ref.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(), + frame.cr.data(), 480, 0, 0, 480, 272); + // Per channel, because that is what "how different does it look" means - a byte-wise diff + // on a packed 16-bit pixel could be one step in one channel or a disaster in three. + int shifts[3], masks[3]; + if (bpp == 4) { + shifts[0] = 0; shifts[1] = 8; shifts[2] = 16; + masks[0] = masks[1] = masks[2] = 0xFF; + } else if (pixelMode == GE_CMODE_16BIT_BGR5650) { + shifts[0] = 0; shifts[1] = 5; shifts[2] = 11; + masks[0] = 0x1F; masks[1] = 0x3F; masks[2] = 0x1F; + } else if (pixelMode == GE_CMODE_16BIT_ABGR5551) { + shifts[0] = 0; shifts[1] = 5; shifts[2] = 10; + masks[0] = masks[1] = masks[2] = 0x1F; + } else { + shifts[0] = 0; shifts[1] = 4; shifts[2] = 8; + masks[0] = masks[1] = masks[2] = 0x0F; + } + int worst = 0; + double total = 0.0; + int count = 0; + for (int y = 0; y < 272; y++) { + for (int x = 0; x < 480; x++) { + const size_t off = ((size_t)y * 512 + x) * bpp; + u32 a = 0, b = 0; + memcpy(&a, &sws[off], bpp); + memcpy(&b, &ref[off], bpp); + for (int ch = 0; ch < 3; ch++) { + const int va = (int)((a >> shifts[ch]) & masks[ch]); + const int vb = (int)((b >> shifts[ch]) & masks[ch]); + const int d = va > vb ? va - vb : vb - va; + worst = worst > d ? worst : d; + total += d; + count++; + } + } + } + printf(" %s: worst channel step %d, mean %.3f\n", PixelModeName(pixelMode), worst, + total / count); + } + + printf("MpegCscRange, 480x272:\n"); + for (int pixelMode : pixelModes) { + const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRangeScalar); + const double swsMps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRange); + // A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it. + printf(" %s: scalar %6.1f MPix/s (%5.2f ms), swscale %6.1f MPix/s (%5.2f ms)\n", + PixelModeName(pixelMode), mps, 480.0 * 272.0 / mps / 1000.0, + swsMps, 480.0 * 272.0 / swsMps / 1000.0); + } + + return true; +} diff --git a/unittest/UnitTest.cpp b/unittest/UnitTest.cpp index df11db0120..4ac480f7c2 100644 --- a/unittest/UnitTest.cpp +++ b/unittest/UnitTest.cpp @@ -2981,6 +2981,7 @@ bool TestThreadManager(); bool TestVFS(); bool TestZipSlip(); bool TestLzrc(); +bool TestMpegCsc(); bool TestDemangle(); // The 8.3 short names games read out of d_private. These aren't verified against hardware yet (no @@ -3200,6 +3201,7 @@ TestItem availableTests[] = { TEST_ITEM(CmdLine), TEST_ITEM(ZipSlip), TEST_ITEM(Lzrc), + TEST_ITEM(MpegCsc), TEST_ITEM(Demangle), TEST_ITEM(TextureReplacer), TEST_ITEM(UITabOrder), diff --git a/unittest/UnitTest.h b/unittest/UnitTest.h index 6588edd40d..35f42f53af 100644 --- a/unittest/UnitTest.h +++ b/unittest/UnitTest.h @@ -1,6 +1,8 @@ #pragma once #include +#include +#include #include inline bool rel_equal(float a, float b, float precision) { diff --git a/unittest/UnitTests.vcxproj b/unittest/UnitTests.vcxproj index edb256fe89..5793f24e1e 100644 --- a/unittest/UnitTests.vcxproj +++ b/unittest/UnitTests.vcxproj @@ -294,6 +294,7 @@ + diff --git a/unittest/UnitTests.vcxproj.filters b/unittest/UnitTests.vcxproj.filters index 8c79b4d58b..37af8831d9 100644 --- a/unittest/UnitTests.vcxproj.filters +++ b/unittest/UnitTests.vcxproj.filters @@ -17,6 +17,7 @@ +