From f41fe77643a506bac890844d6c274416d2edb343 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 29 Aug 2026 12:05:50 +0200 Subject: [PATCH] softgpu: Truncate float->int conversions on SSE, like every other path Vec4::Cast() used _mm_cvtps_epi32 under SSE, which rounds using MXCSR's mode, while NEON's vcvtq_s32_f32 and the scalar (T2)x fallback both truncate. Same split in Rasterizer's InterpolateI. That's on depth interpolation, so the differing values are written to the depth buffer and then compared - a one-LSB difference can flip a later GE_COMP_EQUAL pass and change a whole surface's visibility, not just a shade. Following MXCSR also meant anything that left a non-default rounding mode in the render thread would have changed rasterized output. Truncation is what two of the three paths already did, so SSE moves to match. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Vd8ntC2brCUtCrDJMqLbs8 --- GPU/Math3D.h | 4 +++- GPU/Software/Rasterizer.cpp | 3 ++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/GPU/Math3D.h b/GPU/Math3D.h index f14aa6e1c5..e15ddd7ae6 100644 --- a/GPU/Math3D.h +++ b/GPU/Math3D.h @@ -595,7 +595,9 @@ public: Vec4 Cast() const { if constexpr (std::is_same::value && std::is_same::value) { #if defined(_M_SSE) - return _mm_cvtps_epi32(SAFE_M128(vec)); + // Truncate, don't round. NEON's vcvtq_s32_f32 and the scalar (T2)x below both truncate, + // and _mm_cvtps_epi32 would additionally follow MXCSR's rounding mode. + return _mm_cvttps_epi32(SAFE_M128(vec)); #elif PPSSPP_ARCH(ARM_NEON) return vcvtq_s32_f32(vec); #endif diff --git a/GPU/Software/Rasterizer.cpp b/GPU/Software/Rasterizer.cpp index 56abea7897..8dd12cf5e8 100644 --- a/GPU/Software/Rasterizer.cpp +++ b/GPU/Software/Rasterizer.cpp @@ -56,7 +56,8 @@ static inline __m128 InterpolateF(const __m128 &c0, const __m128 &c1, const __m1 } static inline __m128i InterpolateI(const __m128i &c0, const __m128i &c1, const __m128i &c2, int w0, int w1, int w2, float wsum) { - return _mm_cvtps_epi32(InterpolateF(_mm_cvtepi32_ps(c0), _mm_cvtepi32_ps(c1), _mm_cvtepi32_ps(c2), w0, w1, w2, wsum)); + // Truncate to match the NEON and scalar paths, see Vec4::Cast(). + return _mm_cvttps_epi32(InterpolateF(_mm_cvtepi32_ps(c0), _mm_cvtepi32_ps(c1), _mm_cvtepi32_ps(c2), w0, w1, w2, wsum)); } #elif PPSSPP_ARCH(ARM64_NEON) static inline float32x4_t InterpolateF(const float32x4_t &c0, const float32x4_t &c1, const float32x4_t &c2, int w0, int w1, int w2, float wsum) {