Vertex decoder: make the JIT match the step functions

Add a unit test that decodes every vertex format through both the step
functions and the JIT and requires identical output, side effects included.
Only skinning may differ by rounding, since arm64 accumulates the bone
matrices with fused multiply-adds.

What it found, and fixed:
- x86 morph colors rounded to nearest where the steps truncate, and applied
  the scale before the weight, which rounds differently.
- x86 through-mode u16 UV bounds compared signed, so texcoords above 32767
  scrambled the bounds for everyone running the x86 JIT.
- x86 morph sums could produce -0 where the steps produce +0.
- Step_NormalS16Morph scaled by 1/32768 twice, giving near-zero normals.
- Step_PosFloatThrough lost the truncation of Z to an integer that the JITs
  and the vertex reader did before it moved into the decoder.
- SetVertexType never reset skinInDecode.

Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
This commit is contained in:
Henrik RydgårdandClaude Opus 5 committed 2026-09-21 11:29:40 -06:00
1 parent 12e7d222a7
commit 63a6c11214
3 files changed
+406 -18

No files matched your search

+5 -6
View File
@@ -745,9 +745,7 @@ void VertexDecoder::Step_NormalS16Morph(const VertexDecoder *dec, const u8 *ptr,
acc[j] += sv[j] * multiplier;
}
float *normal = (float *)(decoded + dec->decFmt.nrmoff);
normal[0] = acc[0] * (1.0f / 32768.0f);
normal[1] = acc[1] * (1.0f / 32768.0f);
normal[2] = acc[2] * (1.0f / 32768.0f);
memcpy(normal, acc, sizeof(float) * 3);
}
void VertexDecoder::Step_NormalFloatMorph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
@@ -888,7 +886,9 @@ void VertexDecoder::Step_PosFloatThrough(const VertexDecoder *dec, const u8 *ptr
float *v = (float *)(decoded + dec->decFmt.posoff);
const float *fv = (const float *)(ptr + dec->posoff);
memcpy(v, fv, 8);
v[2] = fv[2] > 65535.0f ? 65535.0f : (fv[2] < 0.0f ? 0.0f : fv[2]);
// Depth is an integer in through mode: truncate, and clamp to 16 bits (NaN becomes 0).
const float z = fv[2];
v[2] = z >= 65535.0f ? 65535.0f : (z > 0.0f ? (float)(int)z : 0.0f);
}
void VertexDecoder::Step_PosS8Morph(const VertexDecoder *dec, const u8 *ptr, u8 *decoded) {
@@ -1191,9 +1191,8 @@ void VertexDecoder::SetVertexType(u32 fmt, const VertexDecoderOptions &options,
DEBUG_LOG(Log::G3D, "VTYPE: THRU=%i TC=%i COL=%i POS=%i NRM=%i WT=%i NW=%i IDX=%i MC=%i", (int)throughmode, tc, col, pos, nrm, weighttype, nweights, idx, morphcount);
}
skinInDecode = weighttype != 0;
if (weighttype) { // && nweights?
skinInDecode = true;
weightoff = size;
//size = align(size, wtalign[weighttype]); unnecessary
size += wtsize[weighttype] * nweights;
+26 -12
View File
@@ -691,6 +691,10 @@ void VertexDecoderJitCache::Jit_TcAnyMorph(int bits) {
first = false;
}
}
// The steps sum onto +0, which turns a sum of -0s into +0. Adding +0 at the end does the same.
XORPS(fpScratchReg2, R(fpScratchReg2));
ADDPS(fpScratchReg, R(fpScratchReg2));
}
void VertexDecoderJitCache::Jit_TcU8MorphToFloat() {
@@ -770,10 +774,11 @@ void VertexDecoderJitCache::Jit_TcU16ThroughToFloat() {
SetJumpTarget(skip);
};
// TODO: Can this actually be fast? Hmm, floats aren't better.
updateSide(tempReg1, CC_GE, offsetof(KnownVertexBounds, minU));
updateSide(tempReg1, CC_LE, offsetof(KnownVertexBounds, maxU));
updateSide(tempReg2, CC_GE, offsetof(KnownVertexBounds, minV));
updateSide(tempReg2, CC_LE, offsetof(KnownVertexBounds, maxV));
// The bounds are unsigned, so unsigned conditions.
updateSide(tempReg1, CC_AE, offsetof(KnownVertexBounds, minU));
updateSide(tempReg1, CC_BE, offsetof(KnownVertexBounds, maxU));
updateSide(tempReg2, CC_AE, offsetof(KnownVertexBounds, minV));
updateSide(tempReg2, CC_BE, offsetof(KnownVertexBounds, maxV));
}
void VertexDecoderJitCache::Jit_TcFloatThrough() {
@@ -994,12 +999,12 @@ void VertexDecoderJitCache::Jit_Color4444Morph() {
}
CVTDQ2PS(reg, R(reg));
MULPS(reg, R(XMM6));
// And now the weight.
// The weight goes on before the scale, same order as the steps (it rounds differently).
MOVSS(fpScratchReg3, MDisp(tempReg1, n * sizeof(float)));
SHUFPS(fpScratchReg3, R(fpScratchReg3), _MM_SHUFFLE(0, 0, 0, 0));
MULPS(reg, R(fpScratchReg3));
MULPS(reg, R(XMM6));
if (!first) {
ADDPS(fpScratchReg, R(fpScratchReg2));
@@ -1049,12 +1054,12 @@ void VertexDecoderJitCache::Jit_Color565Morph() {
MOVSS(reg, R(fpScratchReg2));
CVTDQ2PS(reg, R(reg));
MULPS(reg, R(XMM6));
// And now the weight.
// The weight goes on before the scale, same order as the steps (it rounds differently).
MOVSS(fpScratchReg2, MDisp(tempReg1, n * sizeof(float)));
SHUFPS(fpScratchReg2, R(fpScratchReg2), _MM_SHUFFLE(0, 0, 0, 0));
MULPS(reg, R(fpScratchReg2));
MULPS(reg, R(XMM6));
if (!first) {
ADDPS(fpScratchReg, R(fpScratchReg3));
@@ -1107,12 +1112,12 @@ void VertexDecoderJitCache::Jit_Color5551Morph() {
MOVSS(reg, R(fpScratchReg2));
CVTDQ2PS(reg, R(reg));
MULPS(reg, R(XMM6));
// And now the weight.
// The weight goes on before the scale, same order as the steps (it rounds differently).
MOVSS(fpScratchReg2, MDisp(tempReg1, n * sizeof(float)));
SHUFPS(fpScratchReg2, R(fpScratchReg2), _MM_SHUFFLE(0, 0, 0, 0));
MULPS(reg, R(fpScratchReg2));
MULPS(reg, R(XMM6));
if (!first) {
ADDPS(fpScratchReg, R(fpScratchReg3));
@@ -1125,8 +1130,8 @@ void VertexDecoderJitCache::Jit_Color5551Morph() {
}
void VertexDecoderJitCache::Jit_WriteMorphColor(int outOff, bool checkAlpha) {
// Pack back into a u32, with saturation.
CVTPS2DQ(fpScratchReg, R(fpScratchReg));
// Pack back into a u32, with saturation. Truncate like the other color paths.
CVTTPS2DQ(fpScratchReg, R(fpScratchReg));
PACKSSDW(fpScratchReg, R(fpScratchReg));
PACKUSWB(fpScratchReg, R(fpScratchReg));
MOVD_xmm(R(tempReg1), fpScratchReg);
@@ -1467,6 +1472,9 @@ void VertexDecoderJitCache::Jit_AnyS8Morph(int srcoff, int dstoff) {
}
}
// The steps sum onto +0, which turns a sum of -0s into +0. Adding +0 at the end does the same.
XORPS(fpScratchReg2, R(fpScratchReg2));
ADDPS(fpScratchReg, R(fpScratchReg2));
MOVUPS(MDisp(dstReg, dstoff), fpScratchReg);
}
@@ -1506,6 +1514,9 @@ void VertexDecoderJitCache::Jit_AnyS16Morph(int srcoff, int dstoff) {
}
}
// The steps sum onto +0, which turns a sum of -0s into +0. Adding +0 at the end does the same.
XORPS(fpScratchReg2, R(fpScratchReg2));
ADDPS(fpScratchReg, R(fpScratchReg2));
MOVUPS(MDisp(dstReg, dstoff), fpScratchReg);
}
@@ -1527,6 +1538,9 @@ void VertexDecoderJitCache::Jit_AnyFloatMorph(int srcoff, int dstoff) {
}
}
// The steps sum onto +0, which turns a sum of -0s into +0. Adding +0 at the end does the same.
XORPS(fpScratchReg2, R(fpScratchReg2));
ADDPS(fpScratchReg, R(fpScratchReg2));
MOVUPS(MDisp(dstReg, dstoff), fpScratchReg);
}
+375
View File
@@ -16,11 +16,17 @@
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
#include <math.h>
#include <cstring>
#include <map>
#include <string>
#include "Common/CommonTypes.h"
#include "Common/MemoryUtil.h"
#include "Common/StringUtils.h"
#include "Common/TimeUtil.h"
#include "Core/Config.h"
#include "Core/ConfigValues.h"
#include "Core/HDRemaster.h"
#include "GPU/Common/VertexDecoderCommon.h"
#include "GPU/ge_constants.h"
#include "GPU/GPUState.h"
@@ -628,6 +634,373 @@ static bool TestVertexFloatSkin() {
// TODO: Morph (col, pos, nrm), weights (no skin), morph + weights?
// Everything below checks the JIT (and the handwritten SIMD decoders) against the step functions,
// across the whole space of vertex formats. The two must agree bit for bit - not just closely -
// since tests and frame dumps are recorded with one and games run with the other.
namespace {
struct JitMatchRng {
uint32_t state = 0x12345678;
uint32_t Next() {
state ^= state << 13;
state ^= state >> 17;
state ^= state << 5;
return state;
}
float Float(float lo, float hi) {
return lo + (hi - lo) * (float)(Next() & 0xFFFFFF) / (float)0xFFFFFF;
}
// The GE loads morph weights, bone matrices and the UV scale as 24-bit floats.
float Float24(float lo, float hi) {
float f = Float(lo, hi);
uint32_t bits;
memcpy(&bits, &f, 4);
bits &= 0xFFFFFF00;
memcpy(&f, &bits, 4);
return f;
}
// Vertex data floats: mostly arbitrary, with some exact values mixed in.
float VertexFloat() {
static const float nice[] = { 0.0f, 1.0f, -1.0f, 0.5f, -0.5f, 2.0f, 127.0f / 128.0f, 256.0f };
if ((Next() & 3) == 0) {
return nice[Next() % ARRAY_SIZE(nice)];
}
return Float(-8.0f, 8.0f);
}
};
// Output size of the formats the decoder can produce.
int DecodedComponentSize(u8 fmt) {
switch (fmt) {
case DEC_FLOAT_2: return 8;
case DEC_FLOAT_3: return 12;
case DEC_S8_3: return 3;
case DEC_S16_3: return 6;
case DEC_U8_4: return 4;
default: return -1;
}
}
void FillVertexData(JitMatchRng &rng, const VertexDecoder &dec, u8 *src, int count) {
static const int wtSize[] = { 0, 1, 2, 4 };
static const int tcSize[] = { 0, 1, 2, 4 };
static const int nrmPosSize[] = { 0, 1, 2, 4 };
auto fillScalars = [&](u8 *p, int n, int elemSize) {
for (int i = 0; i < n; i++) {
if (elemSize == 4) {
float_le f = rng.VertexFloat();
memcpy(p + i * 4, &f, 4);
} else if (elemSize == 2) {
u16_le v = (u16)rng.Next();
memcpy(p + i * 2, &v, 2);
} else {
p[i] = (u8)rng.Next();
}
}
};
for (int v = 0; v < count; v++) {
for (int m = 0; m < dec.morphcount; m++) {
u8 *p = src + v * dec.size + m * dec.onesize_;
if (dec.weighttype) {
int sz = wtSize[dec.weighttype];
for (int w = 0; w < dec.nweights; w++) {
// Weights are normally 0..1ish; keep floats there so skinning stays sane.
if (sz == 4) {
float_le f = rng.Float(0.0f, 1.5f);
memcpy(p + dec.weightoff + w * 4, &f, 4);
} else {
fillScalars(p + dec.weightoff + w * sz, 1, sz);
}
}
}
if (dec.tc) {
fillScalars(p + dec.tcoff, 2, tcSize[dec.tc]);
}
if (dec.col) {
fillScalars(p + dec.coloff, dec.col == (GE_VTYPE_COL_8888 >> GE_VTYPE_COL_SHIFT) ? 4 : 2, 1);
}
if (dec.nrm) {
fillScalars(p + dec.nrmoff, 3, nrmPosSize[dec.nrm]);
}
if (dec.pos) {
fillScalars(p + dec.posoff, 3, nrmPosSize[dec.pos]);
}
}
}
}
// Skinning is the one place the JITs may round differently: arm64 accumulates the bone matrices
// with fused multiply-adds, the steps multiply and add separately. The difference comes from
// rounding the intermediate terms, which can be far larger than the result when they cancel, so
// the allowed error scales with the sum of the terms' magnitudes rather than with the result.
void SkinTermMagnitudes(const VertexDecoder &dec, const u8 *vtx, bool pos, float out[3]) {
auto readScalar = [](const u8 *p, int type, int i) -> float {
switch (type) {
case 1: return fabsf((float)(s8)p[i] * (1.0f / 128.0f));
case 2: { s16_le s; memcpy(&s, p + i * 2, 2); return fabsf((float)(s16)s * (1.0f / 32768.0f)); }
default: { float_le f; memcpy(&f, p + i * 4, 4); return fabsf((float)f); }
}
};
auto readWeight = [](const u8 *p, int type, int i) -> float {
switch (type) {
case 1: return p[i] * (1.0f / 128.0f);
case 2: { u16_le w; memcpy(&w, p + i * 2, 2); return (u16)w * (1.0f / 32768.0f); }
default: { float_le f; memcpy(&f, p + i * 4, 4); return fabsf((float)f); }
}
};
// Weights come from the first morph frame only, same as the steps.
float absMatrix[12]{};
for (int j = 0; j < dec.nweights; j++) {
const float w = readWeight(vtx + dec.weightoff, dec.weighttype, j);
for (int k = 0; k < 12; k++) {
absMatrix[k] += w * fabsf(gstate.boneMatrix[j * 12 + k]);
}
}
const int type = pos ? dec.pos : dec.nrm;
const int off = pos ? dec.posoff : dec.nrmoff;
float absVec[3]{};
for (int m = 0; m < dec.morphcount; m++) {
const float mw = dec.morphcount > 1 ? fabsf(gstate_c.morphWeights[m]) : 1.0f;
for (int k = 0; k < 3; k++) {
absVec[k] += mw * readScalar(vtx + m * dec.onesize_ + off, type, k);
}
}
for (int i = 0; i < 3; i++) {
out[i] = absVec[0] * absMatrix[i] + absVec[1] * absMatrix[3 + i] + absVec[2] * absMatrix[6 + i] + (pos ? absMatrix[9 + i] : 0.0f);
}
}
struct JitMismatch {
int formats = 0;
int verts = 0;
std::string example;
};
} // namespace
static bool TestVertexJitMatchesSteps() {
constexpr int VERTS = 32;
constexpr int BUF_SIZE = 64 * 1024;
// Decode may overrun by a vertex plus 16 bytes, see DecodeVerts.
u8 *src = (u8 *)AllocateAlignedMemory(BUF_SIZE, 16);
u8 *refOut = (u8 *)AllocateAlignedMemory(BUF_SIZE, 16);
u8 *jitOut = (u8 *)AllocateAlignedMemory(BUF_SIZE, 16);
VertexDecoderJitCache *cache = new VertexDecoderJitCache();
JitMatchRng rng;
std::map<std::string, JitMismatch> mismatches;
int formatsTested = 0;
int formatsJitted = 0;
const bool savedDoubleTexCoords = g_DoubleTextureCoordinates;
auto testFormat = [&](u32 vtype, int uvGenMode, bool doubleTexCoords, bool expand8BitNormals) {
g_DoubleTextureCoordinates = doubleTexCoords;
VertexDecoderOptions opts{};
opts.expand8BitNormalsToFloat = expand8BitNormals;
const u32 vertTypeID = GetVertTypeID(vtype, uvGenMode);
VertexDecoder ref{};
ref.SetVertexType(vertTypeID, opts, nullptr);
// Without a cache, SetVertexType still installs the handwritten decoders for a couple of
// formats. The reference has to be the steps.
ref.jitted_ = nullptr;
if (cache->GetSpaceLeft() < 16384) {
cache->Clear();
}
VertexDecoder jit{};
jit.SetVertexType(vertTypeID, opts, cache);
formatsTested++;
if (!jit.jitted_) {
return;
}
// TODO: The handwritten decoders don't match the steps yet.
if (!cache->IsInSpace((const u8 *)jit.jitted_)) {
return;
}
formatsJitted++;
for (int i = 0; i < 8; i++) {
gstate_c.morphWeights[i] = rng.Float24(-0.5f, 1.5f);
}
for (int i = 0; i < 8 * 12; i++) {
gstate.boneMatrix[i] = rng.Float24(-2.0f, 2.0f);
}
const UVScale uvScale{ rng.Float24(-2.0f, 2.0f), rng.Float24(-2.0f, 2.0f), rng.Float24(-1.0f, 1.0f), rng.Float24(-1.0f, 1.0f) };
const int srcBytes = ref.VertexSize() * VERTS;
_assert_(srcBytes + 256 <= BUF_SIZE && ref.decFmt.stride * (VERTS + 1) + 16 <= BUF_SIZE);
FillVertexData(rng, ref, src, VERTS);
const KnownVertexBounds initialBounds{ 0xFFFF, 0xFFFF, 0, 0 };
memset(refOut, 0, BUF_SIZE);
gstate_c.vertexFullAlpha = true;
gstate_c.vertBounds = initialBounds;
ref.DecodeVerts(refOut, src, &uvScale, VERTS);
const bool refFullAlpha = gstate_c.vertexFullAlpha;
const KnownVertexBounds refBounds = gstate_c.vertBounds;
memset(jitOut, 0, BUF_SIZE);
gstate_c.vertexFullAlpha = true;
gstate_c.vertBounds = initialBounds;
jit.DecodeVerts(jitOut, src, &uvScale, VERTS);
const bool jitFullAlpha = gstate_c.vertexFullAlpha;
const KnownVertexBounds jitBounds = gstate_c.vertBounds;
// Which step writes each component: weights (if any) first, then tc, col, nrm, pos.
int stepIndex = ref.weighttype ? 1 : 0;
StepFunction tcStep = ref.tc ? ref.steps_[stepIndex++] : nullptr;
StepFunction colStep = ref.col ? ref.steps_[stepIndex++] : nullptr;
StepFunction nrmStep = ref.nrm ? ref.steps_[stepIndex++] : nullptr;
StepFunction posStep = ref.steps_[stepIndex];
char fmtDesc[256]{};
ref.ToString(fmtDesc, sizeof(fmtDesc), true);
const char *kind = cache->IsInSpace((const u8 *)jit.jitted_) ? "jit" : "handwritten";
auto record = [&](const char *what, StepFunction step, int badVerts, const std::string &detail) {
std::string key = StringFromFormat("%-4s %-11s %s", what, kind, step ? GetStepFunctionName(step) : "-");
JitMismatch &m = mismatches[key];
if (m.formats == 0) {
m.example = StringFromFormat("%08x %s morph=%d uvgen=%d dbl=%d expand8=%d: %s", vertTypeID, fmtDesc, ref.morphcount, uvGenMode, (int)doubleTexCoords, (int)expand8BitNormals, detail.c_str());
}
m.formats++;
m.verts += badVerts;
};
auto compareComponent = [&](const char *what, StepFunction step, u8 fmt, int off, bool skinnedPos = false, bool skinnedNrm = false) {
if (fmt == DEC_NONE) {
return;
}
const bool skinned = skinnedPos || skinnedNrm;
const int sz = DecodedComponentSize(fmt);
if (sz < 0) {
record(what, step, 0, StringFromFormat("unexpected decoded format %d", fmt));
return;
}
int badVerts = 0;
std::string detail;
for (int v = 0; v < VERTS; v++) {
const u8 *r = refOut + v * ref.decFmt.stride + off;
const u8 *j = jitOut + v * ref.decFmt.stride + off;
if (memcmp(r, j, sz) == 0) {
continue;
}
if (skinned && fmt == DEC_FLOAT_3) {
float terms[3];
SkinTermMagnitudes(ref, src + v * ref.VertexSize(), skinnedPos, terms);
bool close = true;
for (int c = 0; c < 3; c++) {
float fr, fj;
memcpy(&fr, r + c * 4, 4);
memcpy(&fj, j + c * 4, 4);
// About 32 ULPs of the largest term, to cover a rounding per accumulated bone.
if (!(fabsf(fr - fj) <= terms[c] * (1.0f / (1 << 19)))) {
close = false;
}
}
if (close) {
continue;
}
}
if (badVerts++ == 0) {
detail = StringFromFormat("vert %d: steps", v);
const bool isFloat = fmt == DEC_FLOAT_2 || fmt == DEC_FLOAT_3;
for (int pass = 0; pass < 2; pass++) {
const u8 *p = pass == 0 ? r : j;
if (pass == 1) {
detail += " vs jit";
}
for (int c = 0; c < (isFloat ? sz / 4 : sz); c++) {
if (isFloat) {
float f;
memcpy(&f, p + c * 4, 4);
detail += StringFromFormat(" %.9g", f);
} else if (fmt == DEC_S16_3) {
if (c < 3) {
s16 s;
memcpy(&s, p + c * 2, 2);
detail += StringFromFormat(" %d", s);
}
} else {
detail += StringFromFormat(" %d", fmt == DEC_S8_3 ? (int)(s8)p[c] : (int)p[c]);
}
}
}
}
}
if (badVerts) {
record(what, step, badVerts, detail);
}
};
compareComponent("uv", tcStep, ref.decFmt.uvfmt, ref.decFmt.uvoff);
compareComponent("col", colStep, ref.decFmt.c0fmt, ref.decFmt.c0off);
compareComponent("nrm", nrmStep, ref.decFmt.nrmfmt, ref.decFmt.nrmoff, false, ref.skinInDecode);
compareComponent("pos", posStep, DEC_FLOAT_3, ref.decFmt.posoff, ref.skinInDecode && !ref.throughmode, false);
if (refFullAlpha != jitFullAlpha) {
record("fullAlpha", colStep, 0, StringFromFormat("steps %d vs jit %d", (int)refFullAlpha, (int)jitFullAlpha));
}
// TODO: Only the steps track bounds for float UVs in through mode. Undecided which way to go.
if (tcStep != &VertexDecoder::Step_TcFloatThrough && memcmp(&refBounds, &jitBounds, sizeof(refBounds)) != 0) {
record("bounds", tcStep, 0, StringFromFormat("steps %d,%d-%d,%d vs jit %d,%d-%d,%d",
refBounds.minU, refBounds.minV, refBounds.maxU, refBounds.maxV, jitBounds.minU, jitBounds.minV, jitBounds.maxU, jitBounds.maxV));
}
};
static const int colFormats[] = { 0, 4, 5, 6, 7 };
for (int through = 0; through <= 1; through++) {
for (int tc = 0; tc < 4; tc++) {
for (int col : colFormats) {
for (int nrm = 0; nrm < 4; nrm++) {
for (int pos = 1; pos < 4; pos++) {
for (int wt = 0; wt < 4; wt++) {
for (int nweights = 1; nweights <= (wt ? 8 : 1); nweights++) {
for (int morph = 1; morph <= 8; morph++) {
u32 vtype = (tc << GE_VTYPE_TC_SHIFT) | (col << GE_VTYPE_COL_SHIFT) | (nrm << GE_VTYPE_NRM_SHIFT) |
(pos << GE_VTYPE_POS_SHIFT) | (wt << GE_VTYPE_WEIGHT_SHIFT) |
((nweights - 1) << GE_VTYPE_WEIGHTCOUNT_SHIFT) | ((morph - 1) << GE_VTYPE_MORPHCOUNT_SHIFT) |
(through ? GE_VTYPE_THROUGH : 0);
// The UV gen mode and double texcoords only change anything with texcoords, and
// only the TEXTURE_COORDS / TEXTURE_MATRIX split matters (the other two modes
// decode like these). The expand option only applies to plain 8-bit normals.
const int uvGenModes = (tc && !through) ? 2 : 1;
const int doubleModes = tc ? 2 : 1;
const int expandModes = (nrm == 1 && morph == 1 && !wt) ? 2 : 1;
for (int uvGen = 0; uvGen < uvGenModes; uvGen++) {
for (int dbl = 0; dbl < doubleModes; dbl++) {
for (int expand = 0; expand < expandModes; expand++) {
testFormat(vtype, uvGen == 0 ? GE_TEXMAP_TEXTURE_COORDS : GE_TEXMAP_TEXTURE_MATRIX, dbl != 0, expand != 0);
}
}
}
}
}
}
}
}
}
}
}
g_DoubleTextureCoordinates = savedDoubleTexCoords;
delete cache;
FreeAlignedMemory(src);
FreeAlignedMemory(refOut);
FreeAlignedMemory(jitOut);
printf("VertexJitMatchesSteps: %d formats, %d jitted, %d mismatch kinds\n", formatsTested, formatsJitted, (int)mismatches.size());
for (auto &iter : mismatches) {
printf(" %s: %d formats, %d verts\n e.g. %s\n", iter.first.c_str(), iter.second.formats, iter.second.verts, iter.second.example.c_str());
}
return mismatches.empty();
}
typedef bool (*VertexTestFunc)();
static VertexTestFunc vertdecTestFuncs[] = {
@@ -647,6 +1020,8 @@ static VertexTestFunc vertdecTestFuncs[] = {
&TestVertex8Skin,
&TestVertex16Skin,
&TestVertexFloatSkin,
&TestVertexJitMatchesSteps,
};
bool TestVertexJit() {