mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Unit tests: Share the benchmark timing loop
Four tests had their own copy of "call this until N seconds have passed, then divide". CallsPerSecond in UnitTest.h does it; each keeps its old duration and batch size. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
1 parent
7a06e25aa0
commit
e8c39ed1a7
4 files changed
+36
-48
No files matched your search
+8
-14
@@ -37,6 +37,7 @@
|
||||
#include "Core/CoreTiming.h"
|
||||
#include "Core/Config.h"
|
||||
#include "Core/HLE/HLE.h"
|
||||
#include "unittest/UnitTest.h"
|
||||
|
||||
void UnitTestTerminator() {
|
||||
// Bails out of jit so we can time things.
|
||||
@@ -50,27 +51,20 @@ HLEFunction UnitTestFakeSyscalls[] = {
|
||||
|
||||
double ExecCPUTest(bool clearCache = true) {
|
||||
int blockTicks = 1000000;
|
||||
int total = 0;
|
||||
|
||||
if (MIPSComp::jit) {
|
||||
currentMIPS->pc = PSP_GetUserMemoryBase();
|
||||
MIPSComp::JitAt(currentMIPS);
|
||||
}
|
||||
|
||||
double st = time_now_d();
|
||||
do {
|
||||
for (int j = 0; j < 1000; ++j) {
|
||||
currentMIPS->pc = PSP_GetUserMemoryBase();
|
||||
coreState = CORE_RUNNING_CPU;
|
||||
const double callsPerSecond = CallsPerSecond([&] {
|
||||
currentMIPS->pc = PSP_GetUserMemoryBase();
|
||||
coreState = CORE_RUNNING_CPU;
|
||||
|
||||
while (coreState == CORE_RUNNING_CPU) {
|
||||
mipsr4k.RunLoopUntil(blockTicks);
|
||||
}
|
||||
++total;
|
||||
while (coreState == CORE_RUNNING_CPU) {
|
||||
mipsr4k.RunLoopUntil(blockTicks);
|
||||
}
|
||||
}
|
||||
while (time_now_d() - st < 0.5);
|
||||
double elapsed = time_now_d() - st;
|
||||
}, 0.5, 1000);
|
||||
|
||||
if (MIPSComp::jit) {
|
||||
JitBlockCacheDebugInterface *cache = MIPSComp::jit->GetBlockCacheDebugInterface();
|
||||
@@ -83,7 +77,7 @@ double ExecCPUTest(bool clearCache = true) {
|
||||
MIPSComp::jit->ClearCache();
|
||||
}
|
||||
|
||||
return total / elapsed;
|
||||
return callsPerSecond;
|
||||
}
|
||||
|
||||
static void SetupJitHarness() {
|
||||
|
||||
@@ -154,18 +154,11 @@ static double MeasureMegapixelsPerSecond(const TestFrame &frame, int pixelMode,
|
||||
CscFunc fn = &MpegCscRange) {
|
||||
const int destStride = 512;
|
||||
// Long enough to swamp the clock's own resolution, short enough not to pad the test run.
|
||||
const double seconds = 0.2;
|
||||
int frames = 0;
|
||||
const double start = time_now_d();
|
||||
do {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, 0, 0, frame.width, frame.height);
|
||||
frames++;
|
||||
}
|
||||
} while (time_now_d() - start < seconds);
|
||||
const double elapsed = time_now_d() - start;
|
||||
return (double)frames * frame.width * frame.height / elapsed / 1000000.0;
|
||||
const double framesPerSecond = CallsPerSecond([&] {
|
||||
fn(dest.data(), destStride, pixelMode, frame.luma.data(), frame.cb.data(),
|
||||
frame.cr.data(), frame.width, 0, 0, frame.width, frame.height);
|
||||
}, 0.2, 4);
|
||||
return framesPerSecond * frame.width * frame.height / 1000000.0;
|
||||
}
|
||||
|
||||
// The de-tiling as it was originally written, straight from the description of the layout: bounds
|
||||
@@ -243,17 +236,10 @@ static double MeasureUntileMegapixelsPerSecond(const TiledFrame &tiled, int widt
|
||||
u8 *luma = planes.data();
|
||||
u8 *cb = luma + (size_t)width * height;
|
||||
u8 *cr = cb + (size_t)(width / 2) * (height / 2);
|
||||
const double seconds = 0.2;
|
||||
int frames = 0;
|
||||
const double start = time_now_d();
|
||||
do {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
fn(luma, cb, cr, tiled.src, tiled.sizes, width, height);
|
||||
frames++;
|
||||
}
|
||||
} while (time_now_d() - start < seconds);
|
||||
const double elapsed = time_now_d() - start;
|
||||
return (double)frames * width * height / elapsed / 1000000.0;
|
||||
const double framesPerSecond = CallsPerSecond([&] {
|
||||
fn(luma, cb, cr, tiled.src, tiled.sizes, width, height);
|
||||
}, 0.2, 4);
|
||||
return framesPerSecond * width * height / 1000000.0;
|
||||
}
|
||||
|
||||
bool TestMpegCsc() {
|
||||
|
||||
@@ -81,17 +81,7 @@ public:
|
||||
double ExecuteTimed(int vtype, int count, bool useJit) {
|
||||
SetupExecute(vtype, useJit);
|
||||
|
||||
int total = 0;
|
||||
double st = time_now_d();
|
||||
do {
|
||||
for (int j = 0; j < ROUNDS; ++j) {
|
||||
dec_->DecodeVerts(dst_, src_, &g_uvScale, count);
|
||||
++total;
|
||||
}
|
||||
} while (time_now_d() - st < 0.5);
|
||||
double elapsed = time_now_d() - st;
|
||||
|
||||
return total / elapsed;
|
||||
return CallsPerSecond([&] { dec_->DecodeVerts(dst_, src_, &g_uvScale, count); }, 0.5, ROUNDS);
|
||||
}
|
||||
|
||||
void Add8(u8 x) {
|
||||
|
||||
@@ -5,6 +5,24 @@
|
||||
#include <cstring>
|
||||
#include <algorithm>
|
||||
|
||||
#include "Common/TimeUtil.h"
|
||||
|
||||
// For benchmarks: calls fn over and over, callsPerBatch at a time between reads of the clock, for at
|
||||
// least the given number of seconds, and returns how many calls per second that came to. Multiply by
|
||||
// the work one call does (pixels, vertices) for a throughput.
|
||||
template <typename Func>
|
||||
double CallsPerSecond(Func fn, double seconds, int callsPerBatch) {
|
||||
int calls = 0;
|
||||
const double start = time_now_d();
|
||||
do {
|
||||
for (int i = 0; i < callsPerBatch; i++) {
|
||||
fn();
|
||||
}
|
||||
calls += callsPerBatch;
|
||||
} while (time_now_d() - start < seconds);
|
||||
return calls / (time_now_d() - start);
|
||||
}
|
||||
|
||||
inline bool rel_equal(float a, float b, float precision) {
|
||||
float diff = fabsf(a - b);
|
||||
if (diff == 0.0f) {
|
||||
|
||||
Reference in new issue
Block a user