Math3D: Make float Length/Normalize exact and identical on every platform

The SSE, NEON and scalar versions summed the components in three different
orders, so Length() alone differed by architecture. Worse, the SSE normalize
used _mm_rsqrt_ps with no Newton step - about 12 bits - where NEON used vrsqrte
plus two refinement steps and the generic path used exact sqrtf. Three accuracy
tiers for one function, and the SSE2-vs-SSE4.1 choice was made from cpu_info at
runtime, so one binary gave different answers on two different x86 CPUs.

This isn't shading-only: the results feed environment-map texture coordinates via
GE_PROJMAP_NORMALIZED_NORMAL and Lighting's GenerateLightST, so it moves UVs.

sqrtf and division are correctly rounded per IEEE-754, so a single scalar version
is bit-identical everywhere. The horizontal sums being removed were never much
faster than the three multiplies they replaced. The useSSE4 parameter is kept so
the call sites don't churn, but it no longer selects anything.

Co-Authored-By: Claude Opus 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_01Vd8ntC2brCUtCrDJMqLbs8
This commit is contained in:
Henrik RydgårdandClaude Opus 5 committed 2026-08-29 12:05:50 +02:00
1 parent a007976258
commit e4337eed28
1 file changed
+11 -129
+11 -129
View File
@@ -26,28 +26,18 @@
namespace Math3D {
// NOTE: The float Length/Normalize functions below are deliberately scalar on every platform.
// They used to have SSE and NEON specializations, but the three summed the components in three
// different orders (and the SSE normalize used _mm_rsqrt_ps with no Newton step, roughly 12 bits),
// so the same vertex normal came out different per architecture, per build, and - since the SSE4
// variant was picked from cpu_info at runtime - per CPU within one binary. That fed environment-map
// texture coordinates and specular, i.e. actual pixels. sqrtf and division are correctly rounded by
// IEEE-754, so a single scalar version is bit-identical everywhere; the horizontal sum these were
// doing was never much faster than the three multiplies it replaced.
template<>
float Vec2<float>::Length() const
{
// Doubt this is worth it for a vec2 :/
#if defined(_M_SSE)
float ret;
__m128d tmp = _mm_load_sd((const double*)&x);
__m128 xy = _mm_castpd_ps(tmp);
__m128 sq = _mm_mul_ps(xy, xy);
const __m128 r2 = _mm_shuffle_ps(sq, sq, _MM_SHUFFLE(0, 0, 0, 1));
const __m128 res = _mm_add_ss(sq, r2);
_mm_store_ss(&ret, _mm_sqrt_ss(res));
return ret;
#elif PPSSPP_ARCH(ARM64_NEON)
float32x2_t vec = vld1_f32(&x);
float32x2_t sq = vmul_f32(vec, vec);
float32x2_t add2 = vpadd_f32(sq, sq);
float32x2_t res = vsqrt_f32(add2);
return vget_lane_f32(res, 0);
#else
return sqrtf(Length2());
#endif
}
template<>
@@ -84,24 +74,7 @@ float Vec2<float>::Normalize()
template<>
float Vec3<float>::Length() const
{
#if defined(_M_SSE)
float ret;
__m128 xyz = _mm_loadu_ps(&x);
__m128 sq = _mm_mul_ps(xyz, xyz);
const __m128 r2 = _mm_shuffle_ps(sq, sq, _MM_SHUFFLE(0, 0, 0, 1));
const __m128 r3 = _mm_shuffle_ps(sq, sq, _MM_SHUFFLE(0, 0, 0, 2));
const __m128 res = _mm_add_ss(sq, _mm_add_ss(r2, r3));
_mm_store_ss(&ret, _mm_sqrt_ss(res));
return ret;
#elif PPSSPP_ARCH(ARM64_NEON)
float32x4_t sq = vsetq_lane_f32(0.0f, vmulq_f32(vec, vec), 3);
float32x2_t add1 = vget_low_f32(vpaddq_f32(sq, sq));
float32x2_t add2 = vpadd_f32(add1, add1);
float32x2_t res = vsqrt_f32(add2);
return vget_lane_f32(res, 0);
#else
return sqrtf(Length2());
#endif
}
template<>
@@ -121,82 +94,8 @@ float Vec3<float>::Distance2To(const Vec3<float> &other) const {
return Vec3<float>(other-(*this)).Length2();
}
#if defined(_M_SSE)
__m128 SSENormalizeMultiplierSSE2(__m128 v)
{
const __m128 sq = _mm_mul_ps(v, v);
const __m128 r2 = _mm_shuffle_ps(sq, sq, _MM_SHUFFLE(0, 0, 0, 1));
const __m128 r3 = _mm_shuffle_ps(sq, sq, _MM_SHUFFLE(0, 0, 0, 2));
const __m128 res = _mm_add_ss(r3, _mm_add_ss(r2, sq));
const __m128 rt = _mm_rsqrt_ss(res);
return _mm_shuffle_ps(rt, rt, _MM_SHUFFLE(0, 0, 0, 0));
}
#if defined(__GNUC__) || defined(__clang__) || defined(__INTEL_COMPILER)
[[gnu::target("sse4.1")]]
#endif
__m128 SSENormalizeMultiplierSSE4(__m128 v)
{
// This is only used for Vec3f, so ignore the 4th component, might be garbage.
return _mm_rsqrt_ps(_mm_dp_ps(v, v, 0x77));
}
__m128 SSENormalizeMultiplier(bool useSSE4, __m128 v)
{
if (useSSE4)
return SSENormalizeMultiplierSSE4(v);
return SSENormalizeMultiplierSSE2(v);
}
template<>
Vec3<float> Vec3<float>::Normalized(bool useSSE4) const
{
const __m128 normalize = SSENormalizeMultiplier(useSSE4, vec);
return _mm_mul_ps(normalize, vec);
}
template<>
Vec3<float> Vec3<float>::NormalizedOr001(bool useSSE4) const {
const __m128 normalize = SSENormalizeMultiplier(useSSE4, vec);
const __m128 result = _mm_mul_ps(normalize, vec);
const __m128 mask = _mm_cmpunord_ps(result, vec);
const __m128 replace = _mm_and_ps(_mm_set_ps(0.0f, 1.0f, 0.0f, 0.0f), mask);
// Replace with the constant if the mask matched.
return _mm_or_ps(_mm_andnot_ps(mask, result), replace);
}
#elif PPSSPP_ARCH(ARM64_NEON)
template<>
Vec3<float> Vec3<float>::Normalized(bool useSSE4) const {
float32x4_t sq = vsetq_lane_f32(0.0f, vmulq_f32(vec, vec), 3);
float32x2_t add1 = vget_low_f32(vpaddq_f32(sq, sq));
float32x2_t summed = vpadd_f32(add1, add1);
float32x2_t e = vrsqrte_f32(summed);
e = vmul_f32(vrsqrts_f32(vmul_f32(e, e), summed), e);
e = vmul_f32(vrsqrts_f32(vmul_f32(e, e), summed), e);
float32x4_t factor = vdupq_lane_f32(e, 0);
return Vec3<float>(vmulq_f32(vec, factor));
}
template<>
Vec3<float> Vec3<float>::NormalizedOr001(bool useSSE4) const {
float32x4_t sq = vsetq_lane_f32(0.0f, vmulq_f32(vec, vec), 3);
float32x2_t add1 = vget_low_f32(vpaddq_f32(sq, sq));
float32x2_t summed = vpadd_f32(add1, add1);
if (vget_lane_f32(summed, 0) == 0.0f) {
return Vec3<float>(vsetq_lane_f32(1.0f, vdupq_lane_f32(summed, 0), 2));
}
float32x2_t e = vrsqrte_f32(summed);
e = vmul_f32(vrsqrts_f32(vmul_f32(e, e), summed), e);
e = vmul_f32(vrsqrts_f32(vmul_f32(e, e), summed), e);
float32x4_t factor = vdupq_lane_f32(e, 0);
return Vec3<float>(vmulq_f32(vec, factor));
}
#else
// The useSSE4 parameter is kept for source compatibility with the call sites; there is no longer a
// separate path for it to select. See the note above - a runtime-selected variant was part of the bug.
template<>
Vec3<float> Vec3<float>::Normalized(bool useSSE4) const
{
@@ -209,9 +108,8 @@ Vec3<float> Vec3<float>::NormalizedOr001(bool useSSE4) const {
if (len == 0.0f) {
return Vec3<float>(0.0f, 0.0f, 1.0f);
}
return *this / len;
return (*this) / len;
}
#endif
template<>
float Vec3<float>::Normalize()
@@ -300,23 +198,7 @@ float Vec3Packed<float>::Normalize()
template<>
float Vec4<float>::Length() const
{
#if defined(_M_SSE)
float ret;
__m128 xyzw = _mm_loadu_ps(&x);
__m128 sq = _mm_mul_ps(xyzw, xyzw);
const __m128 r2 = _mm_add_ps(sq, _mm_movehl_ps(sq, sq));
const __m128 res = _mm_add_ss(r2, _mm_shuffle_ps(r2, r2, _MM_SHUFFLE(0, 0, 0, 1)));
_mm_store_ss(&ret, _mm_sqrt_ss(res));
return ret;
#elif PPSSPP_ARCH(ARM64_NEON)
float32x4_t sq = vmulq_f32(vec, vec);
float32x2_t add1 = vget_low_f32(vpaddq_f32(sq, sq));
float32x2_t add2 = vpadd_f32(add1, add1);
float32x2_t res = vsqrt_f32(add2);
return vget_lane_f32(res, 0);
#else
return sqrtf(Length2());
#endif
}
template<>