Merge pull request #22307 from hrydgard/csc-perf

sceMpeg LLE: Improve color space conversion perf by using sws_scale
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-09-18 12:54:20 -06:00
commit 95e12a05cf
15 files changed
+718 -77

No files matched your search

+14 -4
View File
@@ -75,9 +75,17 @@ for it:
```powershell
$installPath = & "C:\Program Files (x86)\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath
$msbuild = "$installPath\MSBuild\Current\Bin\MSBuild.exe"
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=x64 /m
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=<platform> /m
```
- **`<platform>` is whatever the machine is - look it up, don't assume.** It is `ARM64` or `x64`, and
the build lands in `Windows\<platform>\<configuration>\` to match, so building one and running the
other is easy to do without noticing. On Windows-on-ARM an x64 build runs anyway, under emulation,
which is what makes it easy to miss: it works, but it is slower than the native build, it is not
the code ARM users get, and any benchmark from it measures the emulator. Get the host from
`python -c "import platform; print(platform.machine())"`, not `$PROCESSOR_ARCHITECTURE`, which
describes the *shell* and says `AMD64` from an emulated one. The binaries say which they are too -
`UnitTest.exe` prints an `ABI:` line at startup.
- Kill leftover `PPSSPPHeadless.exe`/`PPSSPP*.exe` instances before building - one holding the exe makes
the link fail with `LNK1168`, which looks like a build problem and isn't.
- **A stale binary lies consistently.** After a `git stash` cycle that touched a header, do a
@@ -90,7 +98,7 @@ UWP, the legacy Android NDK build and the libretro core have their own build sys
After a chunk of work (not after every edit), run both suites:
- C++ unit tests: build the `UnitTest` project and run `Windows/x64/Debug/UnitTest.exe all`
- C++ unit tests: build the `UnitTest` project and run `Windows/<platform>/Debug/UnitTest.exe all`
(Linux/Mac: configure with `-DUNITTEST=ON`, run `build/PPSSPPUnitTest all`). Tests are listed in
`availableTests` in `unittest/UnitTest.cpp`; pass names instead of `all` to run a subset.
- pspautotests (HLE coverage) - run them **exactly the way CI does**:
@@ -103,8 +111,10 @@ python test.py -g --graphics=software
around a hundred failures that mean nothing is wrong. The only meaningful result is `0 tests failed`.
(The debug-CRT "Detected memory leaks!" dump after the summary line is normal, not a failure.)
New unit tests are added to `availableTests`; large ones go in their own file in `unittest/`, listed in
both CMakeLists.txt and the Visual Studio project.
New unit tests are added to `availableTests`; large ones go in their own file in `unittest/`, which has
to be listed in **three** build files, not two: `CMakeLists.txt`, `unittest/UnitTests.vcxproj` (and its
`.filters`), and `android/jni/Android.mk`, which builds a unit test executable of its own. Miss the last
one and it builds everywhere you can easily try it, and fails on Android CI.
## Multiplatform considerations
+1
View File
@@ -1368,6 +1368,7 @@ if(UNITTEST)
unittest/TestVFS.cpp
unittest/TestZipSlip.cpp
unittest/TestLzrc.cpp
unittest/TestMpegCsc.cpp
unittest/TestDemangle.cpp
unittest/TestTextureReplacer.cpp
unittest/TestRiscVEmitter.cpp
+2
View File
@@ -144,6 +144,7 @@ void __KernelInit()
__PowerInit();
__UtilityInit();
__UmdInit();
__MpegBaseInit();
__MpegInit();
__PsmfInit();
__CtrlInit();
@@ -210,6 +211,7 @@ void __KernelShutdown()
__Mp3Shutdown();
__MpegShutdown();
__MpegBaseShutdown();
__PsmfShutdown();
__PPGeShutdown();
-1
View File
@@ -356,7 +356,6 @@ static void ClearMpegContexts() {
}
void __MpegInit() {
__MpegBaseInit();
// getMpegCtx keys on a handle read out of game memory, so don't leave contexts from a previous
// game around for the next one to find.
ClearMpegContexts();
+247 -59
View File
@@ -20,7 +20,10 @@
// mpeg.prx drives these directly, so they have to be real for the firmware module to run in place
// of our sceMpeg HLE.
#include <algorithm>
#include <cstring>
#include <map>
#include <utility>
#include <vector>
#include "Common/Serialize/Serializer.h"
@@ -37,6 +40,13 @@
#include "GPU/GPUState.h"
#include "GPU/ge_constants.h"
#ifdef USE_FFMPEG
extern "C" {
#include "libswscale/swscale.h"
#include "libavutil/pixfmt.h"
}
#endif
// The PES payloads gathered by sceMpegBasePESpacketCopy, keyed by the destination each was
// copied to. It carries audio as well as video - the destination is what tells them apart - so
// sceVideocodec has to ask for the one matching the address it was handed.
@@ -45,13 +55,30 @@ static std::map<u32, std::vector<u8>> g_pesPackets;
static int g_mpegBaseBufferWidth = 512;
static int g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888;
// Scratch for the planes the de-tiling produces. Reused between calls: a movie converts one of
// these every frame, and they are a couple of hundred kilobytes, so allocating them per call was
// pure overhead. Nothing here needs saving - it is rebuilt from the ME's buffers on every call.
static std::vector<u8> g_untileScratch;
void __MpegBaseInit() {
// None of this survives a boot on hardware.
g_pesPackets.clear();
g_untileScratch.clear();
g_untileScratch.shrink_to_fit();
MpegCscShutdown();
g_mpegBaseBufferWidth = 512;
g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888;
}
void __MpegBaseShutdown() {
// The scratch and the swscale context are worth a few hundred kilobytes between them, and a
// game that played one video early on has no use for either afterwards.
g_pesPackets.clear();
g_untileScratch.clear();
g_untileScratch.shrink_to_fit();
MpegCscShutdown();
}
void __MpegBaseDoState(PointerWrap &p) {
auto s = p.Section("sceMpegbase", 0, 1);
if (!s) {
@@ -174,40 +201,22 @@ static const u8 *MpegBaseFramePointer(u32 addr, int size) {
//
// Untangling it into plain planes costs one pass per frame, which keeps the conversion below
// readable and is not where the time goes.
bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
std::vector<u8> &luma, std::vector<u8> &cb, std::vector<u8> &cr) {
// The de-tiling itself, with the address resolution left outside so it can be measured and
// checked on its own - see TestMpegCsc. src is the eight buffers in sceVideocodec order (four
// luma, then four chroma) and sizes says how big each one is.
//
// Every byte of the output is written for any frame the hardware can produce, so the caller does
// not have to clear it first.
void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8],
int width, int height) {
const int width2 = width >> 1;
const int height2 = height >> 1;
int sizes[8];
VideocodecFrameBufferLayout(width, height, sizes, nullptr);
const int *ySize = sizes;
const int *cSize = sizes + 4;
const u8 *y[4] = {};
const u8 *c[4] = {};
for (int i = 0; i < 4; i++) {
if (ySize[i] > 0) {
y[i] = MpegBaseFramePointer(buffers[i], ySize[i]);
if (!y[i]) {
return false;
}
}
if (cSize[i] > 0) {
c[i] = MpegBaseFramePointer(buffers[4 + i], cSize[i]);
if (!c[i]) {
return false;
}
}
}
luma.assign((size_t)width * height, 0);
cb.assign((size_t)width2 * height2, 128);
cr.assign((size_t)width2 * height2, 128);
// Luma: four buffers, keyed by (left/right half of the band, even/odd row).
for (int b = 0; b < 4; b++) {
if (!y[b]) {
if (!src[b]) {
continue;
}
const int xOffset = (b & 1) ? 16 : 0;
@@ -219,33 +228,69 @@ bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
if (run <= 0 || j + run > ySize[b]) {
continue;
}
memcpy(&luma[(size_t)row * width + bandX], y[b] + j, run);
memcpy(luma + (size_t)row * width + bandX, src[b] + j, run);
}
}
}
// Chroma: same shape in half-resolution coordinates, with interleaved Cb/Cr pairs.
// Chroma: same shape in half-resolution coordinates, but the two planes arrive interleaved as
// (Cb,Cr) pairs, so each group of 8 pixels is a 16-byte run to pull apart. The bounds the old
// version checked per pixel only depend on the group, so they are hoisted out here - that inner
// loop was the expensive half of this function.
for (int b = 0; b < 4; b++) {
if (!c[b]) {
if (!src[b + 4]) {
continue;
}
const int xOffset = (b & 1) ? 8 : 0;
const int yStart = (b >> 1) ? 1 : 0;
int j = 0;
for (int bandX = xOffset; bandX < width2; bandX += 16) {
for (int row = yStart; row < height2; row += 2) {
for (int k = 0; k < 8; k++, j += 2) {
const int x = bandX + k;
if (x >= width2 || j + 1 >= cSize[b]) {
continue;
}
const size_t i = (size_t)row * width2 + x;
cb[i] = c[b][j];
cr[i] = c[b][j + 1];
for (int row = yStart; row < height2; row += 2, j += 16) {
// How many of the 8 fit both in the row and in what the buffer actually holds.
const int fits = std::min(8, (cSize[b] - j) >> 1);
const int run = std::min(width2 - bandX, fits);
if (run <= 0) {
continue;
}
const u8 *from = src[b + 4] + j;
u8 *toCb = cb + (size_t)row * width2 + bandX;
u8 *toCr = cr + (size_t)row * width2 + bandX;
for (int k = 0; k < run; k++) {
toCb[k] = from[k * 2];
toCr[k] = from[k * 2 + 1];
}
}
}
}
}
bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
const u8 **luma, const u8 **cb, const u8 **cr) {
int sizes[8];
VideocodecFrameBufferLayout(width, height, sizes, nullptr);
const u8 *src[8]{};
for (int i = 0; i < 8; i++) {
if (sizes[i] > 0) {
src[i] = MpegBaseFramePointer(buffers[i], sizes[i]);
if (!src[i]) {
return false;
}
}
}
const size_t lumaBytes = (size_t)width * height;
const size_t chromaBytes = (size_t)(width >> 1) * (height >> 1);
if (g_untileScratch.size() < lumaBytes + chromaBytes * 2) {
g_untileScratch.resize(lumaBytes + chromaBytes * 2);
}
u8 *l = g_untileScratch.data();
u8 *b = l + lumaBytes;
u8 *r = b + chromaBytes;
UntileYCbCr(l, b, r, src, sizes, width, height);
*luma = l;
*cb = b;
*cr = r;
return true;
}
@@ -257,18 +302,174 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) {
r = std::min(255, std::max(0, r));
g = std::min(255, std::max(0, g));
b = std::min(255, std::max(0, b));
// Alpha comes out zero, not opaque. That is what the hardware does - our sceMpeg HLE masks it
// off for the same reason, and names Sword Art Online as a game that depends on it, because it
// doesn't clear the alpha in the buffer it hands over and expects the video to leave it clear.
switch (pixelMode) {
case GE_CMODE_16BIT_BGR5650:
return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3);
case GE_CMODE_16BIT_ABGR5551:
return (1 << 15) | ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3);
return ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3);
case GE_CMODE_16BIT_ABGR4444:
return (0xF << 12) | ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4);
return ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4);
default:
return 0xFF000000 | (b << 16) | (g << 8) | r;
return (b << 16) | (g << 8) | r;
}
}
// The conversion itself, with nothing around it. Pure, so that TestMpegCsc can measure it and
// check it - it is the hottest thing in video playback, and the point of having it out here is
// that it can be worked on without a game in the loop.
//
// luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is
// destStride pixels wide in the format pixelMode names, and the converted range always lands at
// its origin.
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
const int width2 = width >> 1;
for (int y = 0; y < rangeHeight; y++) {
const int sy = rangeY + y;
for (int x = 0; x < rangeWidth; x++) {
const int sx = rangeX + x;
const int ci = (sy >> 1) * width2 + (sx >> 1);
const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode);
if (bpp == 4) {
memcpy(dest + (y * destStride + x) * 4, &pixel, 4);
} else {
const u16 p16 = (u16)pixel;
memcpy(dest + (y * destStride + x) * 2, &p16, 2);
}
}
}
}
#ifdef USE_FFMPEG
// swscale is what our sceMpeg HLE converts with, and the planes the de-tiling produces are
// already the YUV420P it wants, so the same thing works here - and it is a great deal quicker
// than doing it a pixel at a time.
//
// The four output formats are the ones MediaEngine::getSwsFormat picks, for the same reasons.
// Alpha is not among them: swscale writes RGBA opaque and leaves the spare bits of the 16-bit
// formats clear, so the masking below is what makes the result match the hardware, exactly as
// the HLE does after its own sws_scale.
static AVPixelFormat SwsFormatForPixelMode(int pixelMode) {
switch (pixelMode) {
case GE_CMODE_16BIT_BGR5650: return AV_PIX_FMT_BGR565LE;
case GE_CMODE_16BIT_ABGR5551: return AV_PIX_FMT_BGR555LE;
case GE_CMODE_16BIT_ABGR4444: return AV_PIX_FMT_BGR444LE;
default: return AV_PIX_FMT_RGBA;
}
}
// Nothing is being scaled here, so this only picks how chroma reaches full resolution: SWS_POINT
// repeats each 2x2 block's sample, as the scalar path and presumably the hardware do, while
// SWS_BILINEAR smooths between samples, as our sceMpeg HLE does. Swap the line to taste - it
// deserves a real option eventually.
static const int MPEG_CSC_SWS_FLAGS = SWS_POINT;
static SwsContext *g_cscSws;
static int g_cscSwsWidth, g_cscSwsHeight, g_cscSwsFormat = -1;
void MpegCscShutdown() {
if (g_cscSws) {
sws_freeContext(g_cscSws);
g_cscSws = nullptr;
}
g_cscSwsWidth = 0;
g_cscSwsHeight = 0;
g_cscSwsFormat = -1;
}
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
// Chroma is half resolution, so an odd origin would start half a sample in and there is no way
// to say that to swscale. Nothing can actually ask for one - the ranges arrive in macroblocks -
// but the scalar path is still there for it.
if ((rangeX & 1) || (rangeY & 1)) {
return false;
}
const AVPixelFormat format = SwsFormatForPixelMode(pixelMode);
if (rangeWidth != g_cscSwsWidth || rangeHeight != g_cscSwsHeight || (int)format != g_cscSwsFormat) {
g_cscSws = sws_getCachedContext(g_cscSws, rangeWidth, rangeHeight, AV_PIX_FMT_YUV420P,
rangeWidth, rangeHeight, format, MPEG_CSC_SWS_FLAGS, nullptr, nullptr, nullptr);
if (!g_cscSws) {
return false;
}
// Studio swing both ways, which is the range the coefficients in the scalar path assume.
int *invCoeff, *coeff, srcRange, dstRange, brightness, contrast, saturation;
if (sws_getColorspaceDetails(g_cscSws, &invCoeff, &srcRange, &coeff, &dstRange, &brightness,
&contrast, &saturation) != -1) {
sws_setColorspaceDetails(g_cscSws, invCoeff, 0, coeff, 0, brightness, contrast, saturation);
}
g_cscSwsWidth = rangeWidth;
g_cscSwsHeight = rangeHeight;
g_cscSwsFormat = (int)format;
}
const int width2 = width >> 1;
const u8 *srcSlice[4] = {
luma + (size_t)rangeY * width + rangeX,
cb + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
cr + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
nullptr,
};
const int srcStride[4] = { width, width2, width2, 0 };
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
u8 *dstSlice[4] = { dest, nullptr, nullptr, nullptr };
const int dstStride[4] = { destStride * bpp, 0, 0, 0 };
if (sws_scale(g_cscSws, srcSlice, srcStride, 0, rangeHeight, dstSlice, dstStride) <= 0) {
return false;
}
// Clear the alpha swscale filled in, which the hardware leaves at zero.
for (int y = 0; y < rangeHeight; y++) {
u8 *row = dest + (size_t)y * destStride * bpp;
if (bpp == 4) {
u32_le *p32 = (u32_le *)row;
for (int x = 0; x < rangeWidth; x++) {
p32[x] = p32[x] & 0x00FFFFFF;
}
} else if (pixelMode != GE_CMODE_16BIT_BGR5650) {
const u16 mask = pixelMode == GE_CMODE_16BIT_ABGR5551 ? 0x7FFF : 0x0FFF;
u16_le *p16 = (u16_le *)row;
for (int x = 0; x < rangeWidth; x++) {
p16[x] = p16[x] & mask;
}
}
}
return true;
}
#else // !USE_FFMPEG
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
return false;
}
void MpegCscShutdown() {}
#endif // USE_FFMPEG
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
#ifdef USE_FFMPEG
if (MpegCscRangeSws(dest, destStride, pixelMode, luma, cb, cr, width,
rangeX, rangeY, rangeWidth, rangeHeight)) {
return;
}
#endif
MpegCscRangeScalar(dest, destStride, pixelMode, luma, cb, cr, width,
rangeX, rangeY, rangeWidth, rangeHeight);
}
// The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the
// latter over the whole frame.
static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
@@ -303,8 +504,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
return hleLogError(Log::Mpeg, -1, "range outside the frame");
}
std::vector<u8> luma, cb, cr;
if (!ReadTiledYCbCr(buffers, width, height, luma, cb, cr)) {
const u8 *luma, *cb, *cr;
if (!ReadTiledYCbCr(buffers, width, height, &luma, &cb, &cr)) {
return hleLogError(Log::Mpeg, -1, "YCbCr buffers not readable");
}
@@ -318,21 +519,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
return hleLogError(Log::Mpeg, -1, "output buffer not writable");
}
const int width2 = width >> 1;
for (int y = 0; y < rangeHeight; y++) {
const int sy = rangeY + y;
for (int x = 0; x < rangeWidth; x++) {
const int sx = rangeX + x;
const int ci = (sy >> 1) * width2 + (sx >> 1);
const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], g_mpegBasePixelMode);
if (bpp == 4) {
memcpy(dest + (y * bufferWidth + x) * 4, &pixel, 4);
} else {
const u16 p16 = (u16)pixel;
memcpy(dest + (y * bufferWidth + x) * 2, &p16, 2);
}
}
}
MpegCscRange(dest, bufferWidth, g_mpegBasePixelMode, luma, cb, cr, width,
rangeX, rangeY, rangeWidth, rangeHeight);
NotifyMemInfo(MemBlockFlags::WRITE, bufferRGB, destSize, "MpegBaseCsc");
// The CPU just wrote a video frame into what is usually a display buffer. The hardware backends
// don't see that on their own, so without telling them the screen keeps showing the last frame
+34 -2
View File
@@ -25,8 +25,9 @@ class PointerWrap;
void Register_sceMpegbase();
// Called per boot, from __MpegInit.
// Called per boot, from __KernelInit and __KernelShutdown, around sceMpeg's own pair.
void __MpegBaseInit();
void __MpegBaseShutdown();
void __MpegBaseDoState(PointerWrap &p);
@@ -39,5 +40,36 @@ std::vector<u8> MpegBaseTakePESPacket(u32 dest);
// Un-tiles a decoded frame from the eight buffers the Media Engine lays it out in into three
// planes. The buffers are in sceVideocodec's order: four luma, then four chroma. cb and cr come
// out at half width and half height, as YUV420 does.
//
// The planes point into scratch that is reused by the next call, so read them before calling again.
bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
std::vector<u8> &luma, std::vector<u8> &cb, std::vector<u8> &cr);
const u8 **luma, const u8 **cb, const u8 **cr);
// The de-tiling on its own, taking the eight buffers already resolved to host pointers, so it can
// be measured and checked without a game - see TestMpegCsc. A null entry in src leaves that
// buffer's share of the output alone.
void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8],
int width, int height);
// Converts a rectangle of a planar YCbCr420 frame to RGB, the way the DMACPLUS does on the way to
// the screen. Pure, so it can be measured and checked on its own - see TestMpegCsc.
//
// luma is width by height; cb and cr are half that in both directions. dest is destStride pixels
// wide in the format pixelMode names (a GEBufferFormat), and the range lands at its origin.
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
// The two implementations behind it, exposed so TestMpegCsc can measure and compare them.
// The scalar one handles anything; the swscale one refuses what it cannot express and is then
// not used. They do not agree to the bit - swscale rounds its own way - so the scalar one is
// what the reference in the test is checked against.
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
// Frees the cached swscale context.
void MpegCscShutdown();
+13 -7
View File
@@ -26,7 +26,9 @@
// by looking at sceMpegBaseYCrCbCopy output on a real PSP.
#include <algorithm>
#include <cstring>
#include <map>
#include <utility>
#include <vector>
#include "Common/Serialize/Serializer.h"
@@ -682,8 +684,8 @@ static int sceVideocodecCopyYCbCr(u32 ctxAddr, int type) {
buffers[fromDescriptor[i]] = Memory::ReadUnchecked_U32(ctxAddr + 0x0c + i * 4);
}
std::vector<u8> luma, cb, cr;
if (!ReadTiledYCbCr(buffers, width, height, luma, cb, cr)) {
const u8 *luma, *cb, *cr;
if (!ReadTiledYCbCr(buffers, width, height, &luma, &cb, &cr)) {
return hleLogError(Log::ME, -1, "YCbCr buffers not readable");
}
@@ -692,13 +694,17 @@ static int sceVideocodecCopyYCbCr(u32 ctxAddr, int type) {
Memory::ReadUnchecked_U32(ctxAddr + 0x30),
Memory::ReadUnchecked_U32(ctxAddr + 0x34),
};
const std::vector<u8> *planes[3] = { &luma, &cb, &cr };
const u8 *planes[3] = { luma, cb, cr };
const u32 planeSizes[3] = {
(u32)(width * height),
(u32)((width >> 1) * (height >> 1)),
(u32)((width >> 1) * (height >> 1)),
};
for (int i = 0; i < 3; i++) {
const u32 size = (u32)planes[i]->size();
if (!Memory::IsValidRange(dst[i], size)) {
return hleLogError(Log::ME, -1, "plane %d (%08x, %d bytes) not writable", i, dst[i], size);
if (!Memory::IsValidRange(dst[i], planeSizes[i])) {
return hleLogError(Log::ME, -1, "plane %d (%08x, %d bytes) not writable", i, dst[i], planeSizes[i]);
}
Memory::MemcpyUnchecked(dst[i], planes[i]->data(), size);
Memory::MemcpyUnchecked(dst[i], planes[i], planeSizes[i]);
}
return hleLogDebug(Log::ME, 0, "%dx%d -> %08x %08x %08x", width, height, dst[0], dst[1], dst[2]);
}
+1
View File
@@ -1060,6 +1060,7 @@ ifeq ($(UNITTEST),1)
$(SRC)/unittest/TestVFS.cpp \
$(SRC)/unittest/TestDemangle.cpp \
$(SRC)/unittest/TestLzrc.cpp \
$(SRC)/unittest/TestMpegCsc.cpp \
$(SRC)/unittest/TestZipSlip.cpp \
$(SRC)/unittest/UnitTest.cpp
+14 -3
View File
@@ -16,16 +16,24 @@ An agent can drive the VS solution non-interactively with `MSBuild.exe` instead
```powershell
$installPath = & "C:\Program Files (x86)\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath
$msbuild = "$installPath\MSBuild\Current\Bin\MSBuild.exe"
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=x64 /m
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=<platform> /m
```
(swap `/t:UnitTest` for `/t:PPSSPPWindows` or another project name as needed; drop it entirely to build the whole solution).
`<platform>` is `ARM64` or `x64` - whichever the machine actually is, so look it up rather than
picking a default. The output directory follows it (`Windows\<platform>\<configuration>\`), which
makes building one and running the other an easy mistake. It is easiest to make on Windows-on-ARM,
where an x64 build runs anyway under emulation: everything appears to work, but it is slower than
the native build and any performance measurement from it describes the emulator rather than the
code. `platform.machine()` in Python reports the host; `$PROCESSOR_ARCHITECTURE` reports the shell,
which is `AMD64` in an emulated shell even on an ARM64 machine.
In addition to the pspautotests runner (test.py), there is a separate binary with C++ unit tests
in the /unittest subdirectory. After substantial changes (at the end of a chunk of work, not
necessarily after every edit), run these too:
- Windows: build the `UnitTest` project (unittest/UnitTests.vcxproj), then run `Windows/x64/Debug/UnitTest.exe all`
- Windows: build the `UnitTest` project (unittest/UnitTests.vcxproj), then run `Windows/<platform>/Debug/UnitTest.exe all` (`<platform>` being `ARM64` or `x64`, whichever you built)
- Linux/Mac: configure with `-DUNITTEST=ON`, then run `build/PPSSPPUnitTest all`
This runs all tests in `availableTests` in unittest/UnitTest.cpp. You can run one or more
@@ -127,7 +135,10 @@ main functions (and also stub out most of the System_ functions as needed). Take
when making cross platform changes.
New unit tests are added by listing them in availableTests in unittest.cpp. If they are large, put them in
separate files in the unittest subdirectory. Remember to update both CMakeLists.txt and the visual studio project.
separate files in the unittest subdirectory. A new file has to be listed in three build files, not two:
`CMakeLists.txt`, `unittest/UnitTests.vcxproj` (and its `.filters`), and `android/jni/Android.mk`, which
builds a unit test executable of its own. The Android one is the easiest to forget, since missing it builds
fine everywhere you are likely to try it and only fails on Android CI.
A unit test is often the first thing to call a given function from outside its own .cpp, which makes the
`ppsspp_unittest` target in the legacy Android build (`android/jni/Android.mk`, see above) the strictest check
+7 -1
View File
@@ -7,12 +7,18 @@ import os
import subprocess
import threading
import glob
import platform
PPSSPP_EXECUTABLES = [
# Windows
# Windows. The machine's own architecture comes first: an x64 build runs on Windows-on-ARM too,
# under emulation, so looking for it first would quietly test the emulated build instead.
"Windows\\Debug\\PPSSPPHeadless.exe",
"Windows\\Release\\PPSSPPHeadless.exe",
] + ([
"Windows\\ARM64\\Debug\\PPSSPPHeadless.exe",
"Windows\\ARM64\\Release\\PPSSPPHeadless.exe",
] if platform.machine().lower() in ("arm64", "aarch64") else []) + [
"Windows\\x64\\Debug\\PPSSPPHeadless.exe",
"Windows\\x64\\Release\\PPSSPPHeadless.exe",
"build*/PPSSPPHeadless.exe",
+379
View File
@@ -0,0 +1,379 @@
// Copyright (c) 2012- PPSSPP Project.
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU General Public License as published by
// the Free Software Foundation, version 2.0 or later versions.
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU General Public License 2.0 for more details.
// A copy of the GPL 2.0 should have been included with the program.
// If not, see http://www.gnu.org/licenses/
// Official git repository and contact information can be found at
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
// Correctness and speed of the Media Engine's colour conversion, which is the hottest thing in
// video playback once sceMpeg runs the real mpeg.prx - sceMpegBaseCscAvc sits at the top of a
// profile of a movie.
//
// The correctness half is a reference implementation written out longhand, so an optimized
// MpegCscRange has something to be wrong against that isn't itself. The speed half reports
// megapixels per second for a 480x272 frame, the size a PSP movie actually is.
#include <cstdio>
#include <cstring>
#include <utility>
#include <vector>
#include "Common/CommonTypes.h"
#include "Common/TimeUtil.h"
#include "Core/HLE/sceMpegbase.h"
#include "Core/HLE/sceVideocodec.h"
#include "GPU/ge_constants.h"
#include "unittest/UnitTest.h"
// The conversion, spelled out. Deliberately the slowest, most obvious thing that could work: it is
// here to disagree with MpegCscRange when MpegCscRange is wrong, so it must not share any of its
// cleverness.
static u32 ReferencePixel(int y, int cbv, int crv, int pixelMode) {
const int c = y - 16, d = cbv - 128, e = crv - 128;
int r = (298 * c + 409 * e + 128) >> 8;
int g = (298 * c - 100 * d - 208 * e + 128) >> 8;
int b = (298 * c + 516 * d + 128) >> 8;
r = r < 0 ? 0 : (r > 255 ? 255 : r);
g = g < 0 ? 0 : (g > 255 ? 255 : g);
b = b < 0 ? 0 : (b > 255 ? 255 : b);
// Alpha zero, matching the hardware - see the note in sceMpegbase.cpp.
switch (pixelMode) {
case GE_CMODE_16BIT_BGR5650:
return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3);
case GE_CMODE_16BIT_ABGR5551:
return ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3);
case GE_CMODE_16BIT_ABGR4444:
return ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4);
default:
return (b << 16) | (g << 8) | r;
}
}
static void ReferenceCscRange(u8 *dest, int destStride, int pixelMode,
const u8 *luma, const u8 *cb, const u8 *cr, int width,
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
const int width2 = width >> 1;
for (int y = 0; y < rangeHeight; y++) {
for (int x = 0; x < rangeWidth; x++) {
const int sx = rangeX + x, sy = rangeY + y;
const int ci = (sy >> 1) * width2 + (sx >> 1);
const u32 pixel = ReferencePixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode);
u8 *out = dest + (y * destStride + x) * bpp;
if (bpp == 4) {
memcpy(out, &pixel, 4);
} else {
const u16 p16 = (u16)pixel;
memcpy(out, &p16, 2);
}
}
}
}
// A frame with something in every direction: a gradient so neighbouring pixels differ, plus values
// that drive the conversion past both ends of the 0..255 clamp, since that is where an optimized
// version is most likely to disagree.
struct TestFrame {
int width = 0;
int height = 0;
std::vector<u8> luma, cb, cr;
TestFrame(int w, int h) : width(w), height(h) {
luma.resize((size_t)w * h);
cb.resize((size_t)(w / 2) * (h / 2));
cr.resize((size_t)(w / 2) * (h / 2));
for (int y = 0; y < h; y++) {
for (int x = 0; x < w; x++) {
luma[(size_t)y * w + x] = (u8)((x * 3 + y * 5) & 0xFF);
}
}
for (int y = 0; y < h / 2; y++) {
for (int x = 0; x < w / 2; x++) {
const size_t i = (size_t)y * (w / 2) + x;
cb[i] = (u8)((x * 7 + y * 2) & 0xFF);
cr[i] = (u8)((x * 2 + y * 11) & 0xFF);
}
}
}
};
static const int pixelModes[4] = {
GE_CMODE_16BIT_BGR5650,
GE_CMODE_16BIT_ABGR5551,
GE_CMODE_16BIT_ABGR4444,
GE_CMODE_32BIT_ABGR8888,
};
static const char *PixelModeName(int mode) {
switch (mode) {
case GE_CMODE_16BIT_BGR5650: return "5650";
case GE_CMODE_16BIT_ABGR5551: return "5551";
case GE_CMODE_16BIT_ABGR4444: return "4444";
default: return "8888";
}
}
static bool CompareAgainstReference(const TestFrame &frame, int pixelMode,
int rangeX, int rangeY, int rangeWidth, int rangeHeight, int destStride) {
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
// Padded, and prefilled with a value neither implementation would write, so that writing
// outside the range - or short of it - is a failure rather than a coincidence.
const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp;
std::vector<u8> got(destSize, 0xCD), want(destSize, 0xCD);
MpegCscRangeScalar(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
for (size_t i = 0; i < destSize; i++) {
if (got[i] != want[i]) {
printf(" %s %dx%d at %d,%d stride %d: byte %d is %02x, should be %02x\n",
PixelModeName(pixelMode), rangeWidth, rangeHeight, rangeX, rangeY, destStride,
(int)i, got[i], want[i]);
return false;
}
}
return true;
}
typedef void (*CscFunc)(u8 *, int, int, const u8 *, const u8 *, const u8 *, int, int, int, int, int);
static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector<u8> &dest,
CscFunc fn = &MpegCscRange) {
const int destStride = 512;
// Long enough to swamp the clock's own resolution, short enough not to pad the test run.
const double seconds = 0.2;
int frames = 0;
const double start = time_now_d();
do {
for (int i = 0; i < 4; i++) {
fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), frame.width, 0, 0, frame.width, frame.height);
frames++;
}
} while (time_now_d() - start < seconds);
const double elapsed = time_now_d() - start;
return (double)frames * frame.width * frame.height / elapsed / 1000000.0;
}
// The de-tiling as it was originally written, straight from the description of the layout: bounds
// checked per pixel, chroma pulled apart one byte at a time. Kept as the thing UntileYCbCr has to
// agree with, and as something to measure it against.
static void ReferenceUntile(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8],
int width, int height) {
const int width2 = width >> 1, height2 = height >> 1;
const int *ySize = sizes;
const int *cSize = sizes + 4;
for (int b = 0; b < 4; b++) {
if (!src[b]) {
continue;
}
const int xOffset = (b & 1) ? 16 : 0;
const int yStart = (b >> 1) ? 1 : 0;
int j = 0;
for (int bandX = xOffset; bandX < width; bandX += 32) {
const int run = width - bandX < 16 ? width - bandX : 16;
for (int row = yStart; row < height; row += 2, j += 16) {
if (run <= 0 || j + run > ySize[b]) {
continue;
}
memcpy(luma + (size_t)row * width + bandX, src[b] + j, run);
}
}
}
for (int b = 0; b < 4; b++) {
if (!src[b + 4]) {
continue;
}
const int xOffset = (b & 1) ? 8 : 0;
const int yStart = (b >> 1) ? 1 : 0;
int j = 0;
for (int bandX = xOffset; bandX < width2; bandX += 16) {
for (int row = yStart; row < height2; row += 2) {
for (int k = 0; k < 8; k++, j += 2) {
const int x = bandX + k;
if (x >= width2 || j + 1 >= cSize[b]) {
continue;
}
const size_t i = (size_t)row * width2 + x;
cb[i] = src[b + 4][j];
cr[i] = src[b + 4][j + 1];
}
}
}
}
}
// The eight buffers the Media Engine would have produced, laid out as UntileYCbCr expects. The
// contents do not matter for speed and the de-tiling is a pure shuffle, so any pattern will do -
// but make it vary so a broken copy is visible.
struct TiledFrame {
std::vector<u8> storage[8];
const u8 *src[8]{};
int sizes[8]{};
TiledFrame(int width, int height) {
VideocodecFrameBufferLayout(width, height, sizes, nullptr);
for (int i = 0; i < 8; i++) {
storage[i].resize(sizes[i] > 0 ? sizes[i] : 1);
for (int j = 0; j < sizes[i]; j++) {
storage[i][j] = (u8)((j * 7 + i * 31) & 0xFF);
}
src[i] = sizes[i] > 0 ? storage[i].data() : nullptr;
}
}
};
typedef void (*UntileFunc)(u8 *, u8 *, u8 *, const u8 *const[8], const int[8], int, int);
static double MeasureUntileMegapixelsPerSecond(const TiledFrame &tiled, int width, int height,
std::vector<u8> &planes, UntileFunc fn = &UntileYCbCr) {
u8 *luma = planes.data();
u8 *cb = luma + (size_t)width * height;
u8 *cr = cb + (size_t)(width / 2) * (height / 2);
const double seconds = 0.2;
int frames = 0;
const double start = time_now_d();
do {
for (int i = 0; i < 4; i++) {
fn(luma, cb, cr, tiled.src, tiled.sizes, width, height);
frames++;
}
} while (time_now_d() - start < seconds);
const double elapsed = time_now_d() - start;
return (double)frames * width * height / elapsed / 1000000.0;
}
bool TestMpegCsc() {
// The size a PSP movie is, so the speed below is the speed that matters.
TestFrame frame(480, 272);
for (int pixelMode : pixelModes) {
// A whole frame, which is what sceMpegBaseCscAvc asks for.
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 0, 480, 272, 512));
// Partial ranges, as sceMpegBaseCscAvcRange asks for. Odd offsets and sizes on purpose:
// chroma is half resolution, so an odd left edge starts mid-chroma-sample, and an odd
// width leaves a pixel that a two-at-a-time inner loop would have to handle separately.
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 16, 16, 64, 32, 512));
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 1, 1, 63, 31, 512));
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 33, 7, 17, 5, 128));
// A range reaching the far edge, where reading one sample too far would go off the frame.
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 464, 256, 16, 16, 64));
// One pixel, one row, one column.
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 5, 9, 1, 1, 16));
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 100, 480, 1, 512));
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 100, 0, 1, 272, 16));
}
// De-tiling, which runs once per frame ahead of the conversion.
{
const size_t planeBytes = (size_t)480 * 272 + (size_t)240 * 136 * 2;
// Odd frame sizes as well as the real one: the guards in here are about buffers that don't
// divide evenly into bands, which is the only thing that makes them fire.
for (auto dims : { std::make_pair(480, 272), std::make_pair(64, 32), std::make_pair(48, 16) }) {
const int w = dims.first, h = dims.second;
TiledFrame tiled(w, h);
std::vector<u8> got((size_t)w * h + (size_t)(w / 2) * (h / 2) * 2, 0xCD);
std::vector<u8> want(got.size(), 0xCD);
u8 *gl = got.data(), *gb = gl + (size_t)w * h, *gr = gb + (size_t)(w / 2) * (h / 2);
u8 *wl = want.data(), *wb = wl + (size_t)w * h, *wr = wb + (size_t)(w / 2) * (h / 2);
UntileYCbCr(gl, gb, gr, tiled.src, tiled.sizes, w, h);
ReferenceUntile(wl, wb, wr, tiled.src, tiled.sizes, w, h);
EXPECT_TRUE(got == want);
// Nothing left at the fill value: the de-tiling covers every byte of a frame, which is
// what lets the caller skip clearing the planes first. A real frame could contain the
// fill byte by chance, so this is looking for whole rows left behind, not exact cover.
size_t untouched = 0;
for (u8 v : got) {
if (v == 0xCD) {
untouched++;
}
}
EXPECT_TRUE(untouched < got.size() / 100);
}
TiledFrame tiled(480, 272);
std::vector<u8> planes(planeBytes, 0);
const double mps = MeasureUntileMegapixelsPerSecond(tiled, 480, 272, planes);
const double refMps = MeasureUntileMegapixelsPerSecond(tiled, 480, 272, planes, ReferenceUntile);
printf("UntileYCbCr, 480x272: %6.1f MPix/s (%5.2f ms/frame), was %6.1f (%5.2f ms)\n",
mps, 480.0 * 272.0 / mps / 1000.0, refMps, 480.0 * 272.0 / refMps / 1000.0);
}
std::vector<u8> dest((size_t)512 * 272 * 4, 0);
// How far swscale lands from the conversion written out longhand. It rounds its own way, so
// this is not expected to be zero - the question is whether it is close enough to use.
printf("swscale against the reference, per channel:\n");
for (int pixelMode : pixelModes) {
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
std::vector<u8> sws((size_t)512 * 272 * 4, 0), ref((size_t)512 * 272 * 4, 0);
if (!MpegCscRangeSws(sws.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), 480, 0, 0, 480, 272)) {
printf(" %s: declined\n", PixelModeName(pixelMode));
continue;
}
ReferenceCscRange(ref.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
frame.cr.data(), 480, 0, 0, 480, 272);
// Per channel, because that is what "how different does it look" means - a byte-wise diff
// on a packed 16-bit pixel could be one step in one channel or a disaster in three.
int shifts[3], masks[3];
if (bpp == 4) {
shifts[0] = 0; shifts[1] = 8; shifts[2] = 16;
masks[0] = masks[1] = masks[2] = 0xFF;
} else if (pixelMode == GE_CMODE_16BIT_BGR5650) {
shifts[0] = 0; shifts[1] = 5; shifts[2] = 11;
masks[0] = 0x1F; masks[1] = 0x3F; masks[2] = 0x1F;
} else if (pixelMode == GE_CMODE_16BIT_ABGR5551) {
shifts[0] = 0; shifts[1] = 5; shifts[2] = 10;
masks[0] = masks[1] = masks[2] = 0x1F;
} else {
shifts[0] = 0; shifts[1] = 4; shifts[2] = 8;
masks[0] = masks[1] = masks[2] = 0x0F;
}
int worst = 0;
double total = 0.0;
int count = 0;
for (int y = 0; y < 272; y++) {
for (int x = 0; x < 480; x++) {
const size_t off = ((size_t)y * 512 + x) * bpp;
u32 a = 0, b = 0;
memcpy(&a, &sws[off], bpp);
memcpy(&b, &ref[off], bpp);
for (int ch = 0; ch < 3; ch++) {
const int va = (int)((a >> shifts[ch]) & masks[ch]);
const int vb = (int)((b >> shifts[ch]) & masks[ch]);
const int d = va > vb ? va - vb : vb - va;
worst = worst > d ? worst : d;
total += d;
count++;
}
}
}
printf(" %s: worst channel step %d, mean %.3f\n", PixelModeName(pixelMode), worst,
total / count);
}
printf("MpegCscRange, 480x272:\n");
for (int pixelMode : pixelModes) {
const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRangeScalar);
const double swsMps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRange);
// A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it.
printf(" %s: scalar %6.1f MPix/s (%5.2f ms), swscale %6.1f MPix/s (%5.2f ms)\n",
PixelModeName(pixelMode), mps, 480.0 * 272.0 / mps / 1000.0,
swsMps, 480.0 * 272.0 / swsMps / 1000.0);
}
return true;
}
+2
View File
@@ -2981,6 +2981,7 @@ bool TestThreadManager();
bool TestVFS();
bool TestZipSlip();
bool TestLzrc();
bool TestMpegCsc();
bool TestDemangle();
// The 8.3 short names games read out of d_private. These aren't verified against hardware yet (no
@@ -3200,6 +3201,7 @@ TestItem availableTests[] = {
TEST_ITEM(CmdLine),
TEST_ITEM(ZipSlip),
TEST_ITEM(Lzrc),
TEST_ITEM(MpegCsc),
TEST_ITEM(Demangle),
TEST_ITEM(TextureReplacer),
TEST_ITEM(UITabOrder),
+2
View File
@@ -1,6 +1,8 @@
#pragma once
#include <cmath>
#include <cstdio>
#include <cstring>
#include <algorithm>
inline bool rel_equal(float a, float b, float precision) {
+1
View File
@@ -294,6 +294,7 @@
<ClCompile Include="TestSoftwareGPUJit.cpp" />
<ClCompile Include="TestDemangle.cpp" />
<ClCompile Include="TestLzrc.cpp" />
<ClCompile Include="TestMpegCsc.cpp" />
<ClCompile Include="TestThreadManager.cpp" />
<ClCompile Include="TestTextureReplacer.cpp" />
<ClCompile Include="TestVertexJit.cpp" />
+1
View File
@@ -17,6 +17,7 @@
<ClCompile Include="TestIRPassSimplify.cpp" />
<ClCompile Include="TestDemangle.cpp" />
<ClCompile Include="TestLzrc.cpp" />
<ClCompile Include="TestMpegCsc.cpp" />
<ClCompile Include="TestTextureReplacer.cpp" />
<ClCompile Include="TestRiscVEmitter.cpp" />
<ClCompile Include="TestVFS.cpp" />