Lighting: Compute the GE's pow from the float's bits

Read as an integer, a float's bits are its log2 with the mantissa a
straight line between powers of two, scaled by 2^23, and writing an
integer back is the matching exp2. That's exactly the GE's
approximation, without log2/exp2/floor. pspPow now also returns 1 for
e <= 0 itself, so the callers drop their checks. Shader languages
without integers fall back to a true pow.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
Henrik RydgårdandClaude Opus 5.5 committed 2026-09-30 15:47:16 -06:00
1 parent 9162592493
commit 333035df03
4 files changed
+41 -44

No files matched your search

+3 -4
View File
@@ -123,7 +123,7 @@ void Lighter::Light(float colorOut0[4], float colorOut1[4], const float colorIn[
if (dot < 0.0f) dot = 0.0f;
if (poweredDiffuse)
dot = specCoef_ <= 0.0f ? 1.0f : PSPLightPow(dot, specCoef_);
dot = PSPLightPow(dot, specCoef_);
// Attenuation
switch (type) {
@@ -138,7 +138,7 @@ void Lighter::Light(float colorOut0[4], float colorOut1[4], const float colorIn[
lightDir = ldir[l];
angle = Dot(toLight.NormalizedOr001(cpu_info.bSSE4_1), lightDir.NormalizedOr001(cpu_info.bSSE4_1));
if (angle >= lcutoff[l])
lightScale = clamp(1.0f / (latt[l].x + latt[l].y * distanceToLight + latt[l].z * distanceToLight*distanceToLight), 0.0f, 1.0f) * (lconv[l] <= 0.0f ? 1.0f : PSPLightPow(angle, lconv[l]));
lightScale = clamp(1.0f / (latt[l].x + latt[l].y * distanceToLight + latt[l].z * distanceToLight*distanceToLight), 0.0f, 1.0f) * PSPLightPow(angle, lconv[l]);
break;
default:
// ILLEGAL
@@ -154,8 +154,7 @@ void Lighter::Light(float colorOut0[4], float colorOut1[4], const float colorIn[
dot = Dot(halfVec, norm);
if (dot > 0.0f) {
Color4 lightSpec(lcolor[2][l], 0.0f);
float specFactor = specCoef_ <= 0.0f ? 1.0f : PSPLightPow(dot, specCoef_);
lightSum1 += (lightSpec * *specular * (specFactor * lightScale));
lightSum1 += (lightSpec * *specular * (PSPLightPow(dot, specCoef_) * lightScale));
}
}
+14 -10
View File
@@ -62,19 +62,23 @@ struct Color4 {
}
};
// The GE's pow() for specular, powered diffuse and the spot exponent: exp2(e * log2(x)), with log2
// and exp2 each a straight line between powers of two (Mitchell's approximation). Matches hardware
// within one step of 255 (gpu/lighting/specular).
// The GE's pow() for specular, powered diffuse and the spot exponent: 1 for e <= 0, else 0 for
// x <= 0. Otherwise exp2(e * log2(x)) with log2 and exp2 each a straight line between powers of two
// (Mitchell's approximation), which is what reading a float's bits as an integer gives: exponent
// plus mantissa, scaled by 2^23. Matches hardware within one step of 255 (gpu/lighting/specular).
inline float PSPLightPow(float x, float e) {
if (!(x > 0.0f)) {
return 0.0f;
return e > 0.0f ? 0.0f : 1.0f;
}
int ex;
float m = frexpf(x, &ex); // x = m * 2^ex, m in [0.5, 1)
float y = e * ((float)(ex - 1) + (2.0f * m - 1.0f));
y = y < -64.0f ? -64.0f : (y > 64.0f ? 64.0f : y);
float fl = floorf(y);
return ldexpf(1.0f + (y - fl), (int)fl);
int32_t ix;
memcpy(&ix, &x, sizeof(ix));
float t = (e > 0.0f ? e : 0.0f) * (float)(ix - 0x3F800000) + 1065353216.0f;
// Also turns NaN into 0, and stays below infinity's bits.
t = t >= 0.0f ? (t < 2139095039.0f ? t : 2139095039.0f) : 0.0f;
int32_t iy = (int32_t)t;
float y;
memcpy(&y, &iy, sizeof(y));
return y;
}
// The GE only uses the top 4 bits of the specular coefficient's mantissa.
+20 -30
View File
@@ -434,15 +434,17 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag
WRITE(p, " float len2 = dot(v, v);\n");
WRITE(p, " return len2 == 0.0 ? splat3(0.0) : (v * inversesqrt(len2));\n");
WRITE(p, "}\n");
// The GE's pow for lighting: exp2(e * log2(x)), with log2 and exp2 each a straight line
// between powers of two. Continuous, so floor() landing on the wrong side of a power of
// two is harmless.
// The GE's pow for lighting: 1 for e <= 0, else 0 for x <= 0. Otherwise exp2(e * log2(x)) with
// log2 and exp2 each a straight line between powers of two, which is what reading a float's
// bits as an integer gives: exponent plus mantissa, scaled by 2^23. Without integers, a true
// pow is close enough.
WRITE(p, "float pspPow(float x, float e) {\n");
WRITE(p, " if (x <= 0.0) return 0.0;\n");
WRITE(p, " float ex = floor(log2(x));\n");
WRITE(p, " float y = e * (ex + x * exp2(-ex) - 1.0);\n");
WRITE(p, " float fl = floor(y);\n");
WRITE(p, " return exp2(fl) * (1.0 + y - fl);\n");
if (compat.bitwiseOps) {
WRITE(p, " float t = max(e, 0.0) * float(floatBitsToInt(max(x, 1e-30)) - 0x3F800000) + 1065353216.0;\n");
WRITE(p, " return x > 0.0 || e <= 0.0 ? intBitsToFloat(int(max(t, 0.0))) : 0.0;\n");
} else {
WRITE(p, " return e <= 0.0 ? 1.0 : pow(max(x, 0.0), e);\n");
}
WRITE(p, "}\n");
}
@@ -667,7 +669,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag
p.C(" } else {\n"); // type must be 0x02 - GE_LIGHTTYPE_SPOT
p.F(" angle = dot(u_lightdir%s, toLight);\n", iStr);
p.F(" if (angle >= u_lightangle_spotCoef%s.x) {\n", iStr);
p.F(" lightScale = attenuation * (u_lightangle_spotCoef%s.y <= 0.0 ? 1.0 : pspPow(angle, u_lightangle_spotCoef%s.y));\n", iStr, iStr, iStr);
p.F(" lightScale = attenuation * pspPow(angle, u_lightangle_spotCoef%s.y);\n", iStr, iStr);
p.C(" } else {\n");
p.C(" lightScale = 0.0;\n");
p.C(" }\n");
@@ -677,17 +679,13 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag
p.C(" }\n");
p.C(" ldot = dot(toLight, worldnormal);\n");
p.C(" if (comp == 0x2u) {\n"); // GE_LIGHTCOMP_ONLYPOWDIFFUSE
p.C(" ldot = u_matspecular.a > 0.0 ? pspPow(ldot, u_matspecular.a) : 1.0;\n");
p.C(" ldot = pspPow(ldot, u_matspecular.a);\n");
p.C(" }\n");
p.F(" diffuse = (u_lightdiffuse%s * diffuseColor) * max(ldot, 0.0);\n", iStr);
p.C(" if (comp == 0x1u && ldot >= 0.0) {\n"); // do specular. note - must allow for the >= case, since the u_matspecular.a <= 0.0 case relies on it.
p.C(" if (u_matspecular.a > 0.0) {\n");
p.C(" vec3 halfVec = toLight + viewDir;\n");
p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n");
p.C(" ldot = pspPow(dot(halfVec, worldnormal) * halfInvLen, u_matspecular.a);\n");
p.C(" } else {\n");
p.C(" ldot = 1.0;\n");
p.C(" }\n");
p.C(" vec3 halfVec = toLight + viewDir;\n");
p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n");
p.C(" ldot = pspPow(dot(halfVec, worldnormal) * halfInvLen, u_matspecular.a);\n");
p.F(" lightSum1 += u_lightspecular%s * specularColor * ldot * lightScale;\n", iStr);
p.C(" }\n");
p.F(" lightSum0.rgb += (u_lightambient%s * ambientColor.rgb + diffuse) * lightScale;\n", iStr);
@@ -727,11 +725,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag
if (poweredDiffuse) {
// pow(0.0, 0.0) may be undefined, but the PSP seems to treat it as 1.0.
// Seen in Tales of the World: Radiant Mythology (#2424.)
p.C(" if (u_matspecular.a > 0.0) {\n");
p.C(" ldot = pspPow(ldot, u_matspecular.a);\n");
p.C(" } else {\n");
p.C(" ldot = 1.0;\n");
p.C(" }\n");
p.C(" ldot = pspPow(ldot, u_matspecular.a);\n");
}
const char *timesLightScale = " * lightScale";
@@ -748,7 +742,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag
case GE_LIGHTTYPE_UNKNOWN:
p.F(" angle = dot(u_lightdir%s, toLight);\n", iStr, iStr);
p.F(" if (angle >= u_lightangle_spotCoef%s.x) {\n", iStr);
p.F(" lightScale = clamp(1.0 / dot(u_lightatt%s, vec3(1.0, distance, distSq)), 0.0, 1.0) * (u_lightangle_spotCoef%s.y <= 0.0 ? 1.0 : pspPow(angle, u_lightangle_spotCoef%s.y));\n", iStr, iStr, iStr);
p.F(" lightScale = clamp(1.0 / dot(u_lightatt%s, vec3(1.0, distance, distSq)), 0.0, 1.0) * pspPow(angle, u_lightangle_spotCoef%s.y);\n", iStr, iStr);
p.C(" } else {\n");
p.C(" lightScale = 0.0;\n");
p.C(" }\n");
@@ -761,13 +755,9 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag
p.F(" diffuse = (u_lightdiffuse%s * diffuseColor) * max(ldot, 0.0);\n", iStr);
if (doSpecular) {
p.C(" if (ldot >= 0.0) {\n");
p.C(" if (u_matspecular.a > 0.0) {\n");
p.C(" vec3 halfVec = toLight + viewDir;\n");
p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n");
p.C(" ldot = pspPow(dot(halfVec, worldnormal) * halfInvLen, u_matspecular.a);\n");
p.C(" } else {\n");
p.C(" ldot = 1.0;\n");
p.C(" }\n");
p.C(" vec3 halfVec = toLight + viewDir;\n");
p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n");
p.C(" ldot = pspPow(dot(halfVec, worldnormal) * halfInvLen, u_matspecular.a);\n");
p.C(" if (ldot > 0.0)\n");
p.F(" lightSum1 += u_lightspecular%s * specularColor * ldot %s;\n", iStr, timesLightScale);
p.C(" }\n");