mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #22307 from hrydgard/csc-perf
sceMpeg LLE: Improve color space conversion perf by using sws_scale
This commit is contained in:
15 files changed
+718
-77
No files matched your search
@@ -75,9 +75,17 @@ for it:
|
||||
```powershell
|
||||
$installPath = & "C:\Program Files (x86)\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath
|
||||
$msbuild = "$installPath\MSBuild\Current\Bin\MSBuild.exe"
|
||||
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=x64 /m
|
||||
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=<platform> /m
|
||||
```
|
||||
|
||||
- **`<platform>` is whatever the machine is - look it up, don't assume.** It is `ARM64` or `x64`, and
|
||||
the build lands in `Windows\<platform>\<configuration>\` to match, so building one and running the
|
||||
other is easy to do without noticing. On Windows-on-ARM an x64 build runs anyway, under emulation,
|
||||
which is what makes it easy to miss: it works, but it is slower than the native build, it is not
|
||||
the code ARM users get, and any benchmark from it measures the emulator. Get the host from
|
||||
`python -c "import platform; print(platform.machine())"`, not `$PROCESSOR_ARCHITECTURE`, which
|
||||
describes the *shell* and says `AMD64` from an emulated one. The binaries say which they are too -
|
||||
`UnitTest.exe` prints an `ABI:` line at startup.
|
||||
- Kill leftover `PPSSPPHeadless.exe`/`PPSSPP*.exe` instances before building - one holding the exe makes
|
||||
the link fail with `LNK1168`, which looks like a build problem and isn't.
|
||||
- **A stale binary lies consistently.** After a `git stash` cycle that touched a header, do a
|
||||
@@ -90,7 +98,7 @@ UWP, the legacy Android NDK build and the libretro core have their own build sys
|
||||
|
||||
After a chunk of work (not after every edit), run both suites:
|
||||
|
||||
- C++ unit tests: build the `UnitTest` project and run `Windows/x64/Debug/UnitTest.exe all`
|
||||
- C++ unit tests: build the `UnitTest` project and run `Windows/<platform>/Debug/UnitTest.exe all`
|
||||
(Linux/Mac: configure with `-DUNITTEST=ON`, run `build/PPSSPPUnitTest all`). Tests are listed in
|
||||
`availableTests` in `unittest/UnitTest.cpp`; pass names instead of `all` to run a subset.
|
||||
- pspautotests (HLE coverage) - run them **exactly the way CI does**:
|
||||
@@ -103,8 +111,10 @@ python test.py -g --graphics=software
|
||||
around a hundred failures that mean nothing is wrong. The only meaningful result is `0 tests failed`.
|
||||
(The debug-CRT "Detected memory leaks!" dump after the summary line is normal, not a failure.)
|
||||
|
||||
New unit tests are added to `availableTests`; large ones go in their own file in `unittest/`, listed in
|
||||
both CMakeLists.txt and the Visual Studio project.
|
||||
New unit tests are added to `availableTests`; large ones go in their own file in `unittest/`, which has
|
||||
to be listed in **three** build files, not two: `CMakeLists.txt`, `unittest/UnitTests.vcxproj` (and its
|
||||
`.filters`), and `android/jni/Android.mk`, which builds a unit test executable of its own. Miss the last
|
||||
one and it builds everywhere you can easily try it, and fails on Android CI.
|
||||
|
||||
## Multiplatform considerations
|
||||
|
||||
|
||||
@@ -1368,6 +1368,7 @@ if(UNITTEST)
|
||||
unittest/TestVFS.cpp
|
||||
unittest/TestZipSlip.cpp
|
||||
unittest/TestLzrc.cpp
|
||||
unittest/TestMpegCsc.cpp
|
||||
unittest/TestDemangle.cpp
|
||||
unittest/TestTextureReplacer.cpp
|
||||
unittest/TestRiscVEmitter.cpp
|
||||
|
||||
@@ -144,6 +144,7 @@ void __KernelInit()
|
||||
__PowerInit();
|
||||
__UtilityInit();
|
||||
__UmdInit();
|
||||
__MpegBaseInit();
|
||||
__MpegInit();
|
||||
__PsmfInit();
|
||||
__CtrlInit();
|
||||
@@ -210,6 +211,7 @@ void __KernelShutdown()
|
||||
|
||||
__Mp3Shutdown();
|
||||
__MpegShutdown();
|
||||
__MpegBaseShutdown();
|
||||
__PsmfShutdown();
|
||||
__PPGeShutdown();
|
||||
|
||||
|
||||
@@ -356,7 +356,6 @@ static void ClearMpegContexts() {
|
||||
}
|
||||
|
||||
void __MpegInit() {
|
||||
__MpegBaseInit();
|
||||
// getMpegCtx keys on a handle read out of game memory, so don't leave contexts from a previous
|
||||
// game around for the next one to find.
|
||||
ClearMpegContexts();
|
||||
|
||||
+247
-59
@@ -20,7 +20,10 @@
|
||||
// mpeg.prx drives these directly, so they have to be real for the firmware module to run in place
|
||||
// of our sceMpeg HLE.
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "Common/Serialize/Serializer.h"
|
||||
@@ -37,6 +40,13 @@
|
||||
#include "GPU/GPUState.h"
|
||||
#include "GPU/ge_constants.h"
|
||||
|
||||
#ifdef USE_FFMPEG
|
||||
extern "C" {
|
||||
#include "libswscale/swscale.h"
|
||||
#include "libavutil/pixfmt.h"
|
||||
}
|
||||
#endif
|
||||
|
||||
// The PES payloads gathered by sceMpegBasePESpacketCopy, keyed by the destination each was
|
||||
// copied to. It carries audio as well as video - the destination is what tells them apart - so
|
||||
// sceVideocodec has to ask for the one matching the address it was handed.
|
||||
@@ -45,13 +55,30 @@ static std::map<u32, std::vector<u8>> g_pesPackets;
|
||||
static int g_mpegBaseBufferWidth = 512;
|
||||
static int g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888;
|
||||
|
||||
// Scratch for the planes the de-tiling produces. Reused between calls: a movie converts one of
|
||||
// these every frame, and they are a couple of hundred kilobytes, so allocating them per call was
|
||||
// pure overhead. Nothing here needs saving - it is rebuilt from the ME's buffers on every call.
|
||||
static std::vector<u8> g_untileScratch;
|
||||
|
||||
void __MpegBaseInit() {
|
||||
// None of this survives a boot on hardware.
|
||||
g_pesPackets.clear();
|
||||
g_untileScratch.clear();
|
||||
g_untileScratch.shrink_to_fit();
|
||||
MpegCscShutdown();
|
||||
g_mpegBaseBufferWidth = 512;
|
||||
g_mpegBasePixelMode = GE_CMODE_32BIT_ABGR8888;
|
||||
}
|
||||
|
||||
void __MpegBaseShutdown() {
|
||||
// The scratch and the swscale context are worth a few hundred kilobytes between them, and a
|
||||
// game that played one video early on has no use for either afterwards.
|
||||
g_pesPackets.clear();
|
||||
g_untileScratch.clear();
|
||||
g_untileScratch.shrink_to_fit();
|
||||
MpegCscShutdown();
|
||||
}
|
||||
|
||||
void __MpegBaseDoState(PointerWrap &p) {
|
||||
auto s = p.Section("sceMpegbase", 0, 1);
|
||||
if (!s) {
|
||||
@@ -174,40 +201,22 @@ static const u8 *MpegBaseFramePointer(u32 addr, int size) {
|
||||
//
|
||||
// Untangling it into plain planes costs one pass per frame, which keeps the conversion below
|
||||
// readable and is not where the time goes.
|
||||
bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
|
||||
std::vector<u8> &luma, std::vector<u8> &cb, std::vector<u8> &cr) {
|
||||
// The de-tiling itself, with the address resolution left outside so it can be measured and
|
||||
// checked on its own - see TestMpegCsc. src is the eight buffers in sceVideocodec order (four
|
||||
// luma, then four chroma) and sizes says how big each one is.
|
||||
//
|
||||
// Every byte of the output is written for any frame the hardware can produce, so the caller does
|
||||
// not have to clear it first.
|
||||
void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8],
|
||||
int width, int height) {
|
||||
const int width2 = width >> 1;
|
||||
const int height2 = height >> 1;
|
||||
|
||||
int sizes[8];
|
||||
VideocodecFrameBufferLayout(width, height, sizes, nullptr);
|
||||
const int *ySize = sizes;
|
||||
const int *cSize = sizes + 4;
|
||||
|
||||
const u8 *y[4] = {};
|
||||
const u8 *c[4] = {};
|
||||
for (int i = 0; i < 4; i++) {
|
||||
if (ySize[i] > 0) {
|
||||
y[i] = MpegBaseFramePointer(buffers[i], ySize[i]);
|
||||
if (!y[i]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (cSize[i] > 0) {
|
||||
c[i] = MpegBaseFramePointer(buffers[4 + i], cSize[i]);
|
||||
if (!c[i]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
luma.assign((size_t)width * height, 0);
|
||||
cb.assign((size_t)width2 * height2, 128);
|
||||
cr.assign((size_t)width2 * height2, 128);
|
||||
|
||||
// Luma: four buffers, keyed by (left/right half of the band, even/odd row).
|
||||
for (int b = 0; b < 4; b++) {
|
||||
if (!y[b]) {
|
||||
if (!src[b]) {
|
||||
continue;
|
||||
}
|
||||
const int xOffset = (b & 1) ? 16 : 0;
|
||||
@@ -219,33 +228,69 @@ bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
|
||||
if (run <= 0 || j + run > ySize[b]) {
|
||||
continue;
|
||||
}
|
||||
memcpy(&luma[(size_t)row * width + bandX], y[b] + j, run);
|
||||
memcpy(luma + (size_t)row * width + bandX, src[b] + j, run);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Chroma: same shape in half-resolution coordinates, with interleaved Cb/Cr pairs.
|
||||
// Chroma: same shape in half-resolution coordinates, but the two planes arrive interleaved as
|
||||
// (Cb,Cr) pairs, so each group of 8 pixels is a 16-byte run to pull apart. The bounds the old
|
||||
// version checked per pixel only depend on the group, so they are hoisted out here - that inner
|
||||
// loop was the expensive half of this function.
|
||||
for (int b = 0; b < 4; b++) {
|
||||
if (!c[b]) {
|
||||
if (!src[b + 4]) {
|
||||
continue;
|
||||
}
|
||||
const int xOffset = (b & 1) ? 8 : 0;
|
||||
const int yStart = (b >> 1) ? 1 : 0;
|
||||
int j = 0;
|
||||
for (int bandX = xOffset; bandX < width2; bandX += 16) {
|
||||
for (int row = yStart; row < height2; row += 2) {
|
||||
for (int k = 0; k < 8; k++, j += 2) {
|
||||
const int x = bandX + k;
|
||||
if (x >= width2 || j + 1 >= cSize[b]) {
|
||||
continue;
|
||||
}
|
||||
const size_t i = (size_t)row * width2 + x;
|
||||
cb[i] = c[b][j];
|
||||
cr[i] = c[b][j + 1];
|
||||
for (int row = yStart; row < height2; row += 2, j += 16) {
|
||||
// How many of the 8 fit both in the row and in what the buffer actually holds.
|
||||
const int fits = std::min(8, (cSize[b] - j) >> 1);
|
||||
const int run = std::min(width2 - bandX, fits);
|
||||
if (run <= 0) {
|
||||
continue;
|
||||
}
|
||||
const u8 *from = src[b + 4] + j;
|
||||
u8 *toCb = cb + (size_t)row * width2 + bandX;
|
||||
u8 *toCr = cr + (size_t)row * width2 + bandX;
|
||||
for (int k = 0; k < run; k++) {
|
||||
toCb[k] = from[k * 2];
|
||||
toCr[k] = from[k * 2 + 1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
|
||||
const u8 **luma, const u8 **cb, const u8 **cr) {
|
||||
int sizes[8];
|
||||
VideocodecFrameBufferLayout(width, height, sizes, nullptr);
|
||||
|
||||
const u8 *src[8]{};
|
||||
for (int i = 0; i < 8; i++) {
|
||||
if (sizes[i] > 0) {
|
||||
src[i] = MpegBaseFramePointer(buffers[i], sizes[i]);
|
||||
if (!src[i]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const size_t lumaBytes = (size_t)width * height;
|
||||
const size_t chromaBytes = (size_t)(width >> 1) * (height >> 1);
|
||||
if (g_untileScratch.size() < lumaBytes + chromaBytes * 2) {
|
||||
g_untileScratch.resize(lumaBytes + chromaBytes * 2);
|
||||
}
|
||||
u8 *l = g_untileScratch.data();
|
||||
u8 *b = l + lumaBytes;
|
||||
u8 *r = b + chromaBytes;
|
||||
UntileYCbCr(l, b, r, src, sizes, width, height);
|
||||
*luma = l;
|
||||
*cb = b;
|
||||
*cr = r;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -257,18 +302,174 @@ static u32 YCbCrToPixel(int y, int cbv, int crv, int pixelMode) {
|
||||
r = std::min(255, std::max(0, r));
|
||||
g = std::min(255, std::max(0, g));
|
||||
b = std::min(255, std::max(0, b));
|
||||
// Alpha comes out zero, not opaque. That is what the hardware does - our sceMpeg HLE masks it
|
||||
// off for the same reason, and names Sword Art Online as a game that depends on it, because it
|
||||
// doesn't clear the alpha in the buffer it hands over and expects the video to leave it clear.
|
||||
switch (pixelMode) {
|
||||
case GE_CMODE_16BIT_BGR5650:
|
||||
return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3);
|
||||
case GE_CMODE_16BIT_ABGR5551:
|
||||
return (1 << 15) | ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3);
|
||||
return ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3);
|
||||
case GE_CMODE_16BIT_ABGR4444:
|
||||
return (0xF << 12) | ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4);
|
||||
return ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4);
|
||||
default:
|
||||
return 0xFF000000 | (b << 16) | (g << 8) | r;
|
||||
return (b << 16) | (g << 8) | r;
|
||||
}
|
||||
}
|
||||
|
||||
// The conversion itself, with nothing around it. Pure, so that TestMpegCsc can measure it and
|
||||
// check it - it is the hottest thing in video playback, and the point of having it out here is
|
||||
// that it can be worked on without a game in the loop.
|
||||
//
|
||||
// luma is width by height; cb and cr are half that in both directions, as YUV420 is. dest is
|
||||
// destStride pixels wide in the format pixelMode names, and the converted range always lands at
|
||||
// its origin.
|
||||
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
const int width2 = width >> 1;
|
||||
for (int y = 0; y < rangeHeight; y++) {
|
||||
const int sy = rangeY + y;
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
const int sx = rangeX + x;
|
||||
const int ci = (sy >> 1) * width2 + (sx >> 1);
|
||||
const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode);
|
||||
if (bpp == 4) {
|
||||
memcpy(dest + (y * destStride + x) * 4, &pixel, 4);
|
||||
} else {
|
||||
const u16 p16 = (u16)pixel;
|
||||
memcpy(dest + (y * destStride + x) * 2, &p16, 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef USE_FFMPEG
|
||||
|
||||
// swscale is what our sceMpeg HLE converts with, and the planes the de-tiling produces are
|
||||
// already the YUV420P it wants, so the same thing works here - and it is a great deal quicker
|
||||
// than doing it a pixel at a time.
|
||||
//
|
||||
// The four output formats are the ones MediaEngine::getSwsFormat picks, for the same reasons.
|
||||
// Alpha is not among them: swscale writes RGBA opaque and leaves the spare bits of the 16-bit
|
||||
// formats clear, so the masking below is what makes the result match the hardware, exactly as
|
||||
// the HLE does after its own sws_scale.
|
||||
static AVPixelFormat SwsFormatForPixelMode(int pixelMode) {
|
||||
switch (pixelMode) {
|
||||
case GE_CMODE_16BIT_BGR5650: return AV_PIX_FMT_BGR565LE;
|
||||
case GE_CMODE_16BIT_ABGR5551: return AV_PIX_FMT_BGR555LE;
|
||||
case GE_CMODE_16BIT_ABGR4444: return AV_PIX_FMT_BGR444LE;
|
||||
default: return AV_PIX_FMT_RGBA;
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing is being scaled here, so this only picks how chroma reaches full resolution: SWS_POINT
|
||||
// repeats each 2x2 block's sample, as the scalar path and presumably the hardware do, while
|
||||
// SWS_BILINEAR smooths between samples, as our sceMpeg HLE does. Swap the line to taste - it
|
||||
// deserves a real option eventually.
|
||||
static const int MPEG_CSC_SWS_FLAGS = SWS_POINT;
|
||||
|
||||
static SwsContext *g_cscSws;
|
||||
static int g_cscSwsWidth, g_cscSwsHeight, g_cscSwsFormat = -1;
|
||||
|
||||
void MpegCscShutdown() {
|
||||
if (g_cscSws) {
|
||||
sws_freeContext(g_cscSws);
|
||||
g_cscSws = nullptr;
|
||||
}
|
||||
g_cscSwsWidth = 0;
|
||||
g_cscSwsHeight = 0;
|
||||
g_cscSwsFormat = -1;
|
||||
}
|
||||
|
||||
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
// Chroma is half resolution, so an odd origin would start half a sample in and there is no way
|
||||
// to say that to swscale. Nothing can actually ask for one - the ranges arrive in macroblocks -
|
||||
// but the scalar path is still there for it.
|
||||
if ((rangeX & 1) || (rangeY & 1)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const AVPixelFormat format = SwsFormatForPixelMode(pixelMode);
|
||||
if (rangeWidth != g_cscSwsWidth || rangeHeight != g_cscSwsHeight || (int)format != g_cscSwsFormat) {
|
||||
g_cscSws = sws_getCachedContext(g_cscSws, rangeWidth, rangeHeight, AV_PIX_FMT_YUV420P,
|
||||
rangeWidth, rangeHeight, format, MPEG_CSC_SWS_FLAGS, nullptr, nullptr, nullptr);
|
||||
if (!g_cscSws) {
|
||||
return false;
|
||||
}
|
||||
// Studio swing both ways, which is the range the coefficients in the scalar path assume.
|
||||
int *invCoeff, *coeff, srcRange, dstRange, brightness, contrast, saturation;
|
||||
if (sws_getColorspaceDetails(g_cscSws, &invCoeff, &srcRange, &coeff, &dstRange, &brightness,
|
||||
&contrast, &saturation) != -1) {
|
||||
sws_setColorspaceDetails(g_cscSws, invCoeff, 0, coeff, 0, brightness, contrast, saturation);
|
||||
}
|
||||
g_cscSwsWidth = rangeWidth;
|
||||
g_cscSwsHeight = rangeHeight;
|
||||
g_cscSwsFormat = (int)format;
|
||||
}
|
||||
|
||||
const int width2 = width >> 1;
|
||||
const u8 *srcSlice[4] = {
|
||||
luma + (size_t)rangeY * width + rangeX,
|
||||
cb + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
|
||||
cr + (size_t)(rangeY >> 1) * width2 + (rangeX >> 1),
|
||||
nullptr,
|
||||
};
|
||||
const int srcStride[4] = { width, width2, width2, 0 };
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
u8 *dstSlice[4] = { dest, nullptr, nullptr, nullptr };
|
||||
const int dstStride[4] = { destStride * bpp, 0, 0, 0 };
|
||||
if (sws_scale(g_cscSws, srcSlice, srcStride, 0, rangeHeight, dstSlice, dstStride) <= 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Clear the alpha swscale filled in, which the hardware leaves at zero.
|
||||
for (int y = 0; y < rangeHeight; y++) {
|
||||
u8 *row = dest + (size_t)y * destStride * bpp;
|
||||
if (bpp == 4) {
|
||||
u32_le *p32 = (u32_le *)row;
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
p32[x] = p32[x] & 0x00FFFFFF;
|
||||
}
|
||||
} else if (pixelMode != GE_CMODE_16BIT_BGR5650) {
|
||||
const u16 mask = pixelMode == GE_CMODE_16BIT_ABGR5551 ? 0x7FFF : 0x0FFF;
|
||||
u16_le *p16 = (u16_le *)row;
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
p16[x] = p16[x] & mask;
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
#else // !USE_FFMPEG
|
||||
|
||||
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void MpegCscShutdown() {}
|
||||
|
||||
#endif // USE_FFMPEG
|
||||
|
||||
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
#ifdef USE_FFMPEG
|
||||
if (MpegCscRangeSws(dest, destStride, pixelMode, luma, cb, cr, width,
|
||||
rangeX, rangeY, rangeWidth, rangeHeight)) {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
MpegCscRangeScalar(dest, destStride, pixelMode, luma, cb, cr, width,
|
||||
rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
}
|
||||
|
||||
// The shared body of sceMpegBaseCscAvc and sceMpegBaseCscAvcRange - the former is just the
|
||||
// latter over the whole frame.
|
||||
static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
|
||||
@@ -303,8 +504,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
|
||||
return hleLogError(Log::Mpeg, -1, "range outside the frame");
|
||||
}
|
||||
|
||||
std::vector<u8> luma, cb, cr;
|
||||
if (!ReadTiledYCbCr(buffers, width, height, luma, cb, cr)) {
|
||||
const u8 *luma, *cb, *cr;
|
||||
if (!ReadTiledYCbCr(buffers, width, height, &luma, &cb, &cr)) {
|
||||
return hleLogError(Log::Mpeg, -1, "YCbCr buffers not readable");
|
||||
}
|
||||
|
||||
@@ -318,21 +519,8 @@ static int MpegBaseCscRange(u32 bufferRGB, u32 cscAddr, int bufferWidth,
|
||||
return hleLogError(Log::Mpeg, -1, "output buffer not writable");
|
||||
}
|
||||
|
||||
const int width2 = width >> 1;
|
||||
for (int y = 0; y < rangeHeight; y++) {
|
||||
const int sy = rangeY + y;
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
const int sx = rangeX + x;
|
||||
const int ci = (sy >> 1) * width2 + (sx >> 1);
|
||||
const u32 pixel = YCbCrToPixel(luma[sy * width + sx], cb[ci], cr[ci], g_mpegBasePixelMode);
|
||||
if (bpp == 4) {
|
||||
memcpy(dest + (y * bufferWidth + x) * 4, &pixel, 4);
|
||||
} else {
|
||||
const u16 p16 = (u16)pixel;
|
||||
memcpy(dest + (y * bufferWidth + x) * 2, &p16, 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
MpegCscRange(dest, bufferWidth, g_mpegBasePixelMode, luma, cb, cr, width,
|
||||
rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
NotifyMemInfo(MemBlockFlags::WRITE, bufferRGB, destSize, "MpegBaseCsc");
|
||||
// The CPU just wrote a video frame into what is usually a display buffer. The hardware backends
|
||||
// don't see that on their own, so without telling them the screen keeps showing the last frame
|
||||
|
||||
+34
-2
@@ -25,8 +25,9 @@ class PointerWrap;
|
||||
|
||||
void Register_sceMpegbase();
|
||||
|
||||
// Called per boot, from __MpegInit.
|
||||
// Called per boot, from __KernelInit and __KernelShutdown, around sceMpeg's own pair.
|
||||
void __MpegBaseInit();
|
||||
void __MpegBaseShutdown();
|
||||
|
||||
void __MpegBaseDoState(PointerWrap &p);
|
||||
|
||||
@@ -39,5 +40,36 @@ std::vector<u8> MpegBaseTakePESPacket(u32 dest);
|
||||
// Un-tiles a decoded frame from the eight buffers the Media Engine lays it out in into three
|
||||
// planes. The buffers are in sceVideocodec's order: four luma, then four chroma. cb and cr come
|
||||
// out at half width and half height, as YUV420 does.
|
||||
//
|
||||
// The planes point into scratch that is reused by the next call, so read them before calling again.
|
||||
bool ReadTiledYCbCr(const u32 *buffers, int width, int height,
|
||||
std::vector<u8> &luma, std::vector<u8> &cb, std::vector<u8> &cr);
|
||||
const u8 **luma, const u8 **cb, const u8 **cr);
|
||||
|
||||
// The de-tiling on its own, taking the eight buffers already resolved to host pointers, so it can
|
||||
// be measured and checked without a game - see TestMpegCsc. A null entry in src leaves that
|
||||
// buffer's share of the output alone.
|
||||
void UntileYCbCr(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8],
|
||||
int width, int height);
|
||||
|
||||
// Converts a rectangle of a planar YCbCr420 frame to RGB, the way the DMACPLUS does on the way to
|
||||
// the screen. Pure, so it can be measured and checked on its own - see TestMpegCsc.
|
||||
//
|
||||
// luma is width by height; cb and cr are half that in both directions. dest is destStride pixels
|
||||
// wide in the format pixelMode names (a GEBufferFormat), and the range lands at its origin.
|
||||
void MpegCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
|
||||
|
||||
// The two implementations behind it, exposed so TestMpegCsc can measure and compare them.
|
||||
// The scalar one handles anything; the swscale one refuses what it cannot express and is then
|
||||
// not used. They do not agree to the bit - swscale rounds its own way - so the scalar one is
|
||||
// what the reference in the test is checked against.
|
||||
void MpegCscRangeScalar(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
|
||||
bool MpegCscRangeSws(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight);
|
||||
|
||||
// Frees the cached swscale context.
|
||||
void MpegCscShutdown();
|
||||
@@ -26,7 +26,9 @@
|
||||
// by looking at sceMpegBaseYCrCbCopy output on a real PSP.
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <map>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "Common/Serialize/Serializer.h"
|
||||
@@ -682,8 +684,8 @@ static int sceVideocodecCopyYCbCr(u32 ctxAddr, int type) {
|
||||
buffers[fromDescriptor[i]] = Memory::ReadUnchecked_U32(ctxAddr + 0x0c + i * 4);
|
||||
}
|
||||
|
||||
std::vector<u8> luma, cb, cr;
|
||||
if (!ReadTiledYCbCr(buffers, width, height, luma, cb, cr)) {
|
||||
const u8 *luma, *cb, *cr;
|
||||
if (!ReadTiledYCbCr(buffers, width, height, &luma, &cb, &cr)) {
|
||||
return hleLogError(Log::ME, -1, "YCbCr buffers not readable");
|
||||
}
|
||||
|
||||
@@ -692,13 +694,17 @@ static int sceVideocodecCopyYCbCr(u32 ctxAddr, int type) {
|
||||
Memory::ReadUnchecked_U32(ctxAddr + 0x30),
|
||||
Memory::ReadUnchecked_U32(ctxAddr + 0x34),
|
||||
};
|
||||
const std::vector<u8> *planes[3] = { &luma, &cb, &cr };
|
||||
const u8 *planes[3] = { luma, cb, cr };
|
||||
const u32 planeSizes[3] = {
|
||||
(u32)(width * height),
|
||||
(u32)((width >> 1) * (height >> 1)),
|
||||
(u32)((width >> 1) * (height >> 1)),
|
||||
};
|
||||
for (int i = 0; i < 3; i++) {
|
||||
const u32 size = (u32)planes[i]->size();
|
||||
if (!Memory::IsValidRange(dst[i], size)) {
|
||||
return hleLogError(Log::ME, -1, "plane %d (%08x, %d bytes) not writable", i, dst[i], size);
|
||||
if (!Memory::IsValidRange(dst[i], planeSizes[i])) {
|
||||
return hleLogError(Log::ME, -1, "plane %d (%08x, %d bytes) not writable", i, dst[i], planeSizes[i]);
|
||||
}
|
||||
Memory::MemcpyUnchecked(dst[i], planes[i]->data(), size);
|
||||
Memory::MemcpyUnchecked(dst[i], planes[i], planeSizes[i]);
|
||||
}
|
||||
return hleLogDebug(Log::ME, 0, "%dx%d -> %08x %08x %08x", width, height, dst[0], dst[1], dst[2]);
|
||||
}
|
||||
|
||||
@@ -1060,6 +1060,7 @@ ifeq ($(UNITTEST),1)
|
||||
$(SRC)/unittest/TestVFS.cpp \
|
||||
$(SRC)/unittest/TestDemangle.cpp \
|
||||
$(SRC)/unittest/TestLzrc.cpp \
|
||||
$(SRC)/unittest/TestMpegCsc.cpp \
|
||||
$(SRC)/unittest/TestZipSlip.cpp \
|
||||
$(SRC)/unittest/UnitTest.cpp
|
||||
|
||||
|
||||
+14
-3
@@ -16,16 +16,24 @@ An agent can drive the VS solution non-interactively with `MSBuild.exe` instead
|
||||
```powershell
|
||||
$installPath = & "C:\Program Files (x86)\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath
|
||||
$msbuild = "$installPath\MSBuild\Current\Bin\MSBuild.exe"
|
||||
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=x64 /m
|
||||
& $msbuild "Windows\PPSSPP.sln" /t:UnitTest /p:Configuration=Debug /p:Platform=<platform> /m
|
||||
```
|
||||
|
||||
(swap `/t:UnitTest` for `/t:PPSSPPWindows` or another project name as needed; drop it entirely to build the whole solution).
|
||||
|
||||
`<platform>` is `ARM64` or `x64` - whichever the machine actually is, so look it up rather than
|
||||
picking a default. The output directory follows it (`Windows\<platform>\<configuration>\`), which
|
||||
makes building one and running the other an easy mistake. It is easiest to make on Windows-on-ARM,
|
||||
where an x64 build runs anyway under emulation: everything appears to work, but it is slower than
|
||||
the native build and any performance measurement from it describes the emulator rather than the
|
||||
code. `platform.machine()` in Python reports the host; `$PROCESSOR_ARCHITECTURE` reports the shell,
|
||||
which is `AMD64` in an emulated shell even on an ARM64 machine.
|
||||
|
||||
In addition to the pspautotests runner (test.py), there is a separate binary with C++ unit tests
|
||||
in the /unittest subdirectory. After substantial changes (at the end of a chunk of work, not
|
||||
necessarily after every edit), run these too:
|
||||
|
||||
- Windows: build the `UnitTest` project (unittest/UnitTests.vcxproj), then run `Windows/x64/Debug/UnitTest.exe all`
|
||||
- Windows: build the `UnitTest` project (unittest/UnitTests.vcxproj), then run `Windows/<platform>/Debug/UnitTest.exe all` (`<platform>` being `ARM64` or `x64`, whichever you built)
|
||||
- Linux/Mac: configure with `-DUNITTEST=ON`, then run `build/PPSSPPUnitTest all`
|
||||
|
||||
This runs all tests in `availableTests` in unittest/UnitTest.cpp. You can run one or more
|
||||
@@ -127,7 +135,10 @@ main functions (and also stub out most of the System_ functions as needed). Take
|
||||
when making cross platform changes.
|
||||
|
||||
New unit tests are added by listing them in availableTests in unittest.cpp. If they are large, put them in
|
||||
separate files in the unittest subdirectory. Remember to update both CMakeLists.txt and the visual studio project.
|
||||
separate files in the unittest subdirectory. A new file has to be listed in three build files, not two:
|
||||
`CMakeLists.txt`, `unittest/UnitTests.vcxproj` (and its `.filters`), and `android/jni/Android.mk`, which
|
||||
builds a unit test executable of its own. The Android one is the easiest to forget, since missing it builds
|
||||
fine everywhere you are likely to try it and only fails on Android CI.
|
||||
|
||||
A unit test is often the first thing to call a given function from outside its own .cpp, which makes the
|
||||
`ppsspp_unittest` target in the legacy Android build (`android/jni/Android.mk`, see above) the strictest check
|
||||
|
||||
@@ -7,12 +7,18 @@ import os
|
||||
import subprocess
|
||||
import threading
|
||||
import glob
|
||||
import platform
|
||||
|
||||
|
||||
PPSSPP_EXECUTABLES = [
|
||||
# Windows
|
||||
# Windows. The machine's own architecture comes first: an x64 build runs on Windows-on-ARM too,
|
||||
# under emulation, so looking for it first would quietly test the emulated build instead.
|
||||
"Windows\\Debug\\PPSSPPHeadless.exe",
|
||||
"Windows\\Release\\PPSSPPHeadless.exe",
|
||||
] + ([
|
||||
"Windows\\ARM64\\Debug\\PPSSPPHeadless.exe",
|
||||
"Windows\\ARM64\\Release\\PPSSPPHeadless.exe",
|
||||
] if platform.machine().lower() in ("arm64", "aarch64") else []) + [
|
||||
"Windows\\x64\\Debug\\PPSSPPHeadless.exe",
|
||||
"Windows\\x64\\Release\\PPSSPPHeadless.exe",
|
||||
"build*/PPSSPPHeadless.exe",
|
||||
|
||||
@@ -0,0 +1,379 @@
|
||||
// Copyright (c) 2012- PPSSPP Project.
|
||||
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU General Public License as published by
|
||||
// the Free Software Foundation, version 2.0 or later versions.
|
||||
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU General Public License 2.0 for more details.
|
||||
|
||||
// A copy of the GPL 2.0 should have been included with the program.
|
||||
// If not, see http://www.gnu.org/licenses/
|
||||
|
||||
// Official git repository and contact information can be found at
|
||||
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
|
||||
|
||||
// Correctness and speed of the Media Engine's colour conversion, which is the hottest thing in
|
||||
// video playback once sceMpeg runs the real mpeg.prx - sceMpegBaseCscAvc sits at the top of a
|
||||
// profile of a movie.
|
||||
//
|
||||
// The correctness half is a reference implementation written out longhand, so an optimized
|
||||
// MpegCscRange has something to be wrong against that isn't itself. The speed half reports
|
||||
// megapixels per second for a 480x272 frame, the size a PSP movie actually is.
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "Common/CommonTypes.h"
|
||||
#include "Common/TimeUtil.h"
|
||||
#include "Core/HLE/sceMpegbase.h"
|
||||
#include "Core/HLE/sceVideocodec.h"
|
||||
#include "GPU/ge_constants.h"
|
||||
|
||||
#include "unittest/UnitTest.h"
|
||||
|
||||
// The conversion, spelled out. Deliberately the slowest, most obvious thing that could work: it is
|
||||
// here to disagree with MpegCscRange when MpegCscRange is wrong, so it must not share any of its
|
||||
// cleverness.
|
||||
static u32 ReferencePixel(int y, int cbv, int crv, int pixelMode) {
|
||||
const int c = y - 16, d = cbv - 128, e = crv - 128;
|
||||
int r = (298 * c + 409 * e + 128) >> 8;
|
||||
int g = (298 * c - 100 * d - 208 * e + 128) >> 8;
|
||||
int b = (298 * c + 516 * d + 128) >> 8;
|
||||
r = r < 0 ? 0 : (r > 255 ? 255 : r);
|
||||
g = g < 0 ? 0 : (g > 255 ? 255 : g);
|
||||
b = b < 0 ? 0 : (b > 255 ? 255 : b);
|
||||
// Alpha zero, matching the hardware - see the note in sceMpegbase.cpp.
|
||||
switch (pixelMode) {
|
||||
case GE_CMODE_16BIT_BGR5650:
|
||||
return ((b >> 3) << 11) | ((g >> 2) << 5) | (r >> 3);
|
||||
case GE_CMODE_16BIT_ABGR5551:
|
||||
return ((b >> 3) << 10) | ((g >> 3) << 5) | (r >> 3);
|
||||
case GE_CMODE_16BIT_ABGR4444:
|
||||
return ((b >> 4) << 8) | ((g >> 4) << 4) | (r >> 4);
|
||||
default:
|
||||
return (b << 16) | (g << 8) | r;
|
||||
}
|
||||
}
|
||||
|
||||
static void ReferenceCscRange(u8 *dest, int destStride, int pixelMode,
|
||||
const u8 *luma, const u8 *cb, const u8 *cr, int width,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight) {
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
const int width2 = width >> 1;
|
||||
for (int y = 0; y < rangeHeight; y++) {
|
||||
for (int x = 0; x < rangeWidth; x++) {
|
||||
const int sx = rangeX + x, sy = rangeY + y;
|
||||
const int ci = (sy >> 1) * width2 + (sx >> 1);
|
||||
const u32 pixel = ReferencePixel(luma[sy * width + sx], cb[ci], cr[ci], pixelMode);
|
||||
u8 *out = dest + (y * destStride + x) * bpp;
|
||||
if (bpp == 4) {
|
||||
memcpy(out, &pixel, 4);
|
||||
} else {
|
||||
const u16 p16 = (u16)pixel;
|
||||
memcpy(out, &p16, 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A frame with something in every direction: a gradient so neighbouring pixels differ, plus values
|
||||
// that drive the conversion past both ends of the 0..255 clamp, since that is where an optimized
|
||||
// version is most likely to disagree.
|
||||
struct TestFrame {
|
||||
int width = 0;
|
||||
int height = 0;
|
||||
std::vector<u8> luma, cb, cr;
|
||||
|
||||
TestFrame(int w, int h) : width(w), height(h) {
|
||||
luma.resize((size_t)w * h);
|
||||
cb.resize((size_t)(w / 2) * (h / 2));
|
||||
cr.resize((size_t)(w / 2) * (h / 2));
|
||||
for (int y = 0; y < h; y++) {
|
||||
for (int x = 0; x < w; x++) {
|
||||
luma[(size_t)y * w + x] = (u8)((x * 3 + y * 5) & 0xFF);
|
||||
}
|
||||
}
|
||||
for (int y = 0; y < h / 2; y++) {
|
||||
for (int x = 0; x < w / 2; x++) {
|
||||
const size_t i = (size_t)y * (w / 2) + x;
|
||||
cb[i] = (u8)((x * 7 + y * 2) & 0xFF);
|
||||
cr[i] = (u8)((x * 2 + y * 11) & 0xFF);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static const int pixelModes[4] = {
|
||||
GE_CMODE_16BIT_BGR5650,
|
||||
GE_CMODE_16BIT_ABGR5551,
|
||||
GE_CMODE_16BIT_ABGR4444,
|
||||
GE_CMODE_32BIT_ABGR8888,
|
||||
};
|
||||
|
||||
static const char *PixelModeName(int mode) {
|
||||
switch (mode) {
|
||||
case GE_CMODE_16BIT_BGR5650: return "5650";
|
||||
case GE_CMODE_16BIT_ABGR5551: return "5551";
|
||||
case GE_CMODE_16BIT_ABGR4444: return "4444";
|
||||
default: return "8888";
|
||||
}
|
||||
}
|
||||
|
||||
static bool CompareAgainstReference(const TestFrame &frame, int pixelMode,
|
||||
int rangeX, int rangeY, int rangeWidth, int rangeHeight, int destStride) {
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
// Padded, and prefilled with a value neither implementation would write, so that writing
|
||||
// outside the range - or short of it - is a failure rather than a coincidence.
|
||||
const size_t destSize = (size_t)(rangeHeight + 2) * destStride * bpp;
|
||||
std::vector<u8> got(destSize, 0xCD), want(destSize, 0xCD);
|
||||
|
||||
MpegCscRangeScalar(got.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
ReferenceCscRange(want.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, rangeX, rangeY, rangeWidth, rangeHeight);
|
||||
|
||||
for (size_t i = 0; i < destSize; i++) {
|
||||
if (got[i] != want[i]) {
|
||||
printf(" %s %dx%d at %d,%d stride %d: byte %d is %02x, should be %02x\n",
|
||||
PixelModeName(pixelMode), rangeWidth, rangeHeight, rangeX, rangeY, destStride,
|
||||
(int)i, got[i], want[i]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
typedef void (*CscFunc)(u8 *, int, int, const u8 *, const u8 *, const u8 *, int, int, int, int, int);
|
||||
|
||||
static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode, std::vector<u8> &dest,
|
||||
CscFunc fn = &MpegCscRange) {
|
||||
const int destStride = 512;
|
||||
// Long enough to swamp the clock's own resolution, short enough not to pad the test run.
|
||||
const double seconds = 0.2;
|
||||
int frames = 0;
|
||||
const double start = time_now_d();
|
||||
do {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, 0, 0, frame.width, frame.height);
|
||||
frames++;
|
||||
}
|
||||
} while (time_now_d() - start < seconds);
|
||||
const double elapsed = time_now_d() - start;
|
||||
return (double)frames * frame.width * frame.height / elapsed / 1000000.0;
|
||||
}
|
||||
|
||||
// The de-tiling as it was originally written, straight from the description of the layout: bounds
|
||||
// checked per pixel, chroma pulled apart one byte at a time. Kept as the thing UntileYCbCr has to
|
||||
// agree with, and as something to measure it against.
|
||||
static void ReferenceUntile(u8 *luma, u8 *cb, u8 *cr, const u8 *const src[8], const int sizes[8],
|
||||
int width, int height) {
|
||||
const int width2 = width >> 1, height2 = height >> 1;
|
||||
const int *ySize = sizes;
|
||||
const int *cSize = sizes + 4;
|
||||
for (int b = 0; b < 4; b++) {
|
||||
if (!src[b]) {
|
||||
continue;
|
||||
}
|
||||
const int xOffset = (b & 1) ? 16 : 0;
|
||||
const int yStart = (b >> 1) ? 1 : 0;
|
||||
int j = 0;
|
||||
for (int bandX = xOffset; bandX < width; bandX += 32) {
|
||||
const int run = width - bandX < 16 ? width - bandX : 16;
|
||||
for (int row = yStart; row < height; row += 2, j += 16) {
|
||||
if (run <= 0 || j + run > ySize[b]) {
|
||||
continue;
|
||||
}
|
||||
memcpy(luma + (size_t)row * width + bandX, src[b] + j, run);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int b = 0; b < 4; b++) {
|
||||
if (!src[b + 4]) {
|
||||
continue;
|
||||
}
|
||||
const int xOffset = (b & 1) ? 8 : 0;
|
||||
const int yStart = (b >> 1) ? 1 : 0;
|
||||
int j = 0;
|
||||
for (int bandX = xOffset; bandX < width2; bandX += 16) {
|
||||
for (int row = yStart; row < height2; row += 2) {
|
||||
for (int k = 0; k < 8; k++, j += 2) {
|
||||
const int x = bandX + k;
|
||||
if (x >= width2 || j + 1 >= cSize[b]) {
|
||||
continue;
|
||||
}
|
||||
const size_t i = (size_t)row * width2 + x;
|
||||
cb[i] = src[b + 4][j];
|
||||
cr[i] = src[b + 4][j + 1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The eight buffers the Media Engine would have produced, laid out as UntileYCbCr expects. The
|
||||
// contents do not matter for speed and the de-tiling is a pure shuffle, so any pattern will do -
|
||||
// but make it vary so a broken copy is visible.
|
||||
struct TiledFrame {
|
||||
std::vector<u8> storage[8];
|
||||
const u8 *src[8]{};
|
||||
int sizes[8]{};
|
||||
|
||||
TiledFrame(int width, int height) {
|
||||
VideocodecFrameBufferLayout(width, height, sizes, nullptr);
|
||||
for (int i = 0; i < 8; i++) {
|
||||
storage[i].resize(sizes[i] > 0 ? sizes[i] : 1);
|
||||
for (int j = 0; j < sizes[i]; j++) {
|
||||
storage[i][j] = (u8)((j * 7 + i * 31) & 0xFF);
|
||||
}
|
||||
src[i] = sizes[i] > 0 ? storage[i].data() : nullptr;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
typedef void (*UntileFunc)(u8 *, u8 *, u8 *, const u8 *const[8], const int[8], int, int);
|
||||
|
||||
static double MeasureUntileMegapixelsPerSecond(const TiledFrame &tiled, int width, int height,
|
||||
std::vector<u8> &planes, UntileFunc fn = &UntileYCbCr) {
|
||||
u8 *luma = planes.data();
|
||||
u8 *cb = luma + (size_t)width * height;
|
||||
u8 *cr = cb + (size_t)(width / 2) * (height / 2);
|
||||
const double seconds = 0.2;
|
||||
int frames = 0;
|
||||
const double start = time_now_d();
|
||||
do {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
fn(luma, cb, cr, tiled.src, tiled.sizes, width, height);
|
||||
frames++;
|
||||
}
|
||||
} while (time_now_d() - start < seconds);
|
||||
const double elapsed = time_now_d() - start;
|
||||
return (double)frames * width * height / elapsed / 1000000.0;
|
||||
}
|
||||
|
||||
bool TestMpegCsc() {
|
||||
// The size a PSP movie is, so the speed below is the speed that matters.
|
||||
TestFrame frame(480, 272);
|
||||
|
||||
for (int pixelMode : pixelModes) {
|
||||
// A whole frame, which is what sceMpegBaseCscAvc asks for.
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 0, 480, 272, 512));
|
||||
// Partial ranges, as sceMpegBaseCscAvcRange asks for. Odd offsets and sizes on purpose:
|
||||
// chroma is half resolution, so an odd left edge starts mid-chroma-sample, and an odd
|
||||
// width leaves a pixel that a two-at-a-time inner loop would have to handle separately.
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 16, 16, 64, 32, 512));
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 1, 1, 63, 31, 512));
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 33, 7, 17, 5, 128));
|
||||
// A range reaching the far edge, where reading one sample too far would go off the frame.
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 464, 256, 16, 16, 64));
|
||||
// One pixel, one row, one column.
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 5, 9, 1, 1, 16));
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 0, 100, 480, 1, 512));
|
||||
EXPECT_TRUE(CompareAgainstReference(frame, pixelMode, 100, 0, 1, 272, 16));
|
||||
}
|
||||
|
||||
// De-tiling, which runs once per frame ahead of the conversion.
|
||||
{
|
||||
const size_t planeBytes = (size_t)480 * 272 + (size_t)240 * 136 * 2;
|
||||
// Odd frame sizes as well as the real one: the guards in here are about buffers that don't
|
||||
// divide evenly into bands, which is the only thing that makes them fire.
|
||||
for (auto dims : { std::make_pair(480, 272), std::make_pair(64, 32), std::make_pair(48, 16) }) {
|
||||
const int w = dims.first, h = dims.second;
|
||||
TiledFrame tiled(w, h);
|
||||
std::vector<u8> got((size_t)w * h + (size_t)(w / 2) * (h / 2) * 2, 0xCD);
|
||||
std::vector<u8> want(got.size(), 0xCD);
|
||||
u8 *gl = got.data(), *gb = gl + (size_t)w * h, *gr = gb + (size_t)(w / 2) * (h / 2);
|
||||
u8 *wl = want.data(), *wb = wl + (size_t)w * h, *wr = wb + (size_t)(w / 2) * (h / 2);
|
||||
UntileYCbCr(gl, gb, gr, tiled.src, tiled.sizes, w, h);
|
||||
ReferenceUntile(wl, wb, wr, tiled.src, tiled.sizes, w, h);
|
||||
EXPECT_TRUE(got == want);
|
||||
// Nothing left at the fill value: the de-tiling covers every byte of a frame, which is
|
||||
// what lets the caller skip clearing the planes first. A real frame could contain the
|
||||
// fill byte by chance, so this is looking for whole rows left behind, not exact cover.
|
||||
size_t untouched = 0;
|
||||
for (u8 v : got) {
|
||||
if (v == 0xCD) {
|
||||
untouched++;
|
||||
}
|
||||
}
|
||||
EXPECT_TRUE(untouched < got.size() / 100);
|
||||
}
|
||||
|
||||
TiledFrame tiled(480, 272);
|
||||
std::vector<u8> planes(planeBytes, 0);
|
||||
const double mps = MeasureUntileMegapixelsPerSecond(tiled, 480, 272, planes);
|
||||
const double refMps = MeasureUntileMegapixelsPerSecond(tiled, 480, 272, planes, ReferenceUntile);
|
||||
printf("UntileYCbCr, 480x272: %6.1f MPix/s (%5.2f ms/frame), was %6.1f (%5.2f ms)\n",
|
||||
mps, 480.0 * 272.0 / mps / 1000.0, refMps, 480.0 * 272.0 / refMps / 1000.0);
|
||||
}
|
||||
|
||||
std::vector<u8> dest((size_t)512 * 272 * 4, 0);
|
||||
// How far swscale lands from the conversion written out longhand. It rounds its own way, so
|
||||
// this is not expected to be zero - the question is whether it is close enough to use.
|
||||
printf("swscale against the reference, per channel:\n");
|
||||
for (int pixelMode : pixelModes) {
|
||||
const int bpp = pixelMode == GE_CMODE_32BIT_ABGR8888 ? 4 : 2;
|
||||
std::vector<u8> sws((size_t)512 * 272 * 4, 0), ref((size_t)512 * 272 * 4, 0);
|
||||
if (!MpegCscRangeSws(sws.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), 480, 0, 0, 480, 272)) {
|
||||
printf(" %s: declined\n", PixelModeName(pixelMode));
|
||||
continue;
|
||||
}
|
||||
ReferenceCscRange(ref.data(), 512, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), 480, 0, 0, 480, 272);
|
||||
// Per channel, because that is what "how different does it look" means - a byte-wise diff
|
||||
// on a packed 16-bit pixel could be one step in one channel or a disaster in three.
|
||||
int shifts[3], masks[3];
|
||||
if (bpp == 4) {
|
||||
shifts[0] = 0; shifts[1] = 8; shifts[2] = 16;
|
||||
masks[0] = masks[1] = masks[2] = 0xFF;
|
||||
} else if (pixelMode == GE_CMODE_16BIT_BGR5650) {
|
||||
shifts[0] = 0; shifts[1] = 5; shifts[2] = 11;
|
||||
masks[0] = 0x1F; masks[1] = 0x3F; masks[2] = 0x1F;
|
||||
} else if (pixelMode == GE_CMODE_16BIT_ABGR5551) {
|
||||
shifts[0] = 0; shifts[1] = 5; shifts[2] = 10;
|
||||
masks[0] = masks[1] = masks[2] = 0x1F;
|
||||
} else {
|
||||
shifts[0] = 0; shifts[1] = 4; shifts[2] = 8;
|
||||
masks[0] = masks[1] = masks[2] = 0x0F;
|
||||
}
|
||||
int worst = 0;
|
||||
double total = 0.0;
|
||||
int count = 0;
|
||||
for (int y = 0; y < 272; y++) {
|
||||
for (int x = 0; x < 480; x++) {
|
||||
const size_t off = ((size_t)y * 512 + x) * bpp;
|
||||
u32 a = 0, b = 0;
|
||||
memcpy(&a, &sws[off], bpp);
|
||||
memcpy(&b, &ref[off], bpp);
|
||||
for (int ch = 0; ch < 3; ch++) {
|
||||
const int va = (int)((a >> shifts[ch]) & masks[ch]);
|
||||
const int vb = (int)((b >> shifts[ch]) & masks[ch]);
|
||||
const int d = va > vb ? va - vb : vb - va;
|
||||
worst = worst > d ? worst : d;
|
||||
total += d;
|
||||
count++;
|
||||
}
|
||||
}
|
||||
}
|
||||
printf(" %s: worst channel step %d, mean %.3f\n", PixelModeName(pixelMode), worst,
|
||||
total / count);
|
||||
}
|
||||
|
||||
printf("MpegCscRange, 480x272:\n");
|
||||
for (int pixelMode : pixelModes) {
|
||||
const double mps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRangeScalar);
|
||||
const double swsMps = MeasureMegapixelsPerSecond(frame, pixelMode, dest, &MpegCscRange);
|
||||
// A movie is 480*272 at ~30fps, so 3.9 MPix/s is what playback needs of it.
|
||||
printf(" %s: scalar %6.1f MPix/s (%5.2f ms), swscale %6.1f MPix/s (%5.2f ms)\n",
|
||||
PixelModeName(pixelMode), mps, 480.0 * 272.0 / mps / 1000.0,
|
||||
swsMps, 480.0 * 272.0 / swsMps / 1000.0);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
@@ -2981,6 +2981,7 @@ bool TestThreadManager();
|
||||
bool TestVFS();
|
||||
bool TestZipSlip();
|
||||
bool TestLzrc();
|
||||
bool TestMpegCsc();
|
||||
bool TestDemangle();
|
||||
|
||||
// The 8.3 short names games read out of d_private. These aren't verified against hardware yet (no
|
||||
@@ -3200,6 +3201,7 @@ TestItem availableTests[] = {
|
||||
TEST_ITEM(CmdLine),
|
||||
TEST_ITEM(ZipSlip),
|
||||
TEST_ITEM(Lzrc),
|
||||
TEST_ITEM(MpegCsc),
|
||||
TEST_ITEM(Demangle),
|
||||
TEST_ITEM(TextureReplacer),
|
||||
TEST_ITEM(UITabOrder),
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
#pragma once
|
||||
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <algorithm>
|
||||
|
||||
inline bool rel_equal(float a, float b, float precision) {
|
||||
|
||||
@@ -294,6 +294,7 @@
|
||||
<ClCompile Include="TestSoftwareGPUJit.cpp" />
|
||||
<ClCompile Include="TestDemangle.cpp" />
|
||||
<ClCompile Include="TestLzrc.cpp" />
|
||||
<ClCompile Include="TestMpegCsc.cpp" />
|
||||
<ClCompile Include="TestThreadManager.cpp" />
|
||||
<ClCompile Include="TestTextureReplacer.cpp" />
|
||||
<ClCompile Include="TestVertexJit.cpp" />
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
<ClCompile Include="TestIRPassSimplify.cpp" />
|
||||
<ClCompile Include="TestDemangle.cpp" />
|
||||
<ClCompile Include="TestLzrc.cpp" />
|
||||
<ClCompile Include="TestMpegCsc.cpp" />
|
||||
<ClCompile Include="TestTextureReplacer.cpp" />
|
||||
<ClCompile Include="TestRiscVEmitter.cpp" />
|
||||
<ClCompile Include="TestVFS.cpp" />
|
||||
|
||||
Reference in new issue
Block a user