softgpu: Truncate float->int conversions on SSE, like every other path

Vec4<float>::Cast<int>() used _mm_cvtps_epi32 under SSE, which rounds using
MXCSR's mode, while NEON's vcvtq_s32_f32 and the scalar (T2)x fallback both
truncate. Same split in Rasterizer's InterpolateI.

That's on depth interpolation, so the differing values are written to the depth
buffer and then compared - a one-LSB difference can flip a later GE_COMP_EQUAL
pass and change a whole surface's visibility, not just a shade. Following MXCSR
also meant anything that left a non-default rounding mode in the render thread
would have changed rasterized output.

Truncation is what two of the three paths already did, so SSE moves to match.

Co-Authored-By: Claude Opus 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_01Vd8ntC2brCUtCrDJMqLbs8
This commit is contained in:
Henrik RydgårdandClaude Opus 5 committed 2026-08-29 12:05:50 +02:00
1 parent e4337eed28
commit f41fe77643
2 files changed
+5 -2

No files matched your search

+3 -1
View File
@@ -595,7 +595,9 @@ public:
Vec4<T2> Cast() const {
if constexpr (std::is_same<T, float>::value && std::is_same<T2, int>::value) {
#if defined(_M_SSE)
return _mm_cvtps_epi32(SAFE_M128(vec));
// Truncate, don't round. NEON's vcvtq_s32_f32 and the scalar (T2)x below both truncate,
// and _mm_cvtps_epi32 would additionally follow MXCSR's rounding mode.
return _mm_cvttps_epi32(SAFE_M128(vec));
#elif PPSSPP_ARCH(ARM_NEON)
return vcvtq_s32_f32(vec);
#endif
+2 -1
View File
@@ -56,7 +56,8 @@ static inline __m128 InterpolateF(const __m128 &c0, const __m128 &c1, const __m1
}
static inline __m128i InterpolateI(const __m128i &c0, const __m128i &c1, const __m128i &c2, int w0, int w1, int w2, float wsum) {
return _mm_cvtps_epi32(InterpolateF(_mm_cvtepi32_ps(c0), _mm_cvtepi32_ps(c1), _mm_cvtepi32_ps(c2), w0, w1, w2, wsum));
// Truncate to match the NEON and scalar paths, see Vec4<float>::Cast<int>().
return _mm_cvttps_epi32(InterpolateF(_mm_cvtepi32_ps(c0), _mm_cvtepi32_ps(c1), _mm_cvtepi32_ps(c2), w0, w1, w2, wsum));
}
#elif PPSSPP_ARCH(ARM64_NEON)
static inline float32x4_t InterpolateF(const float32x4_t &c0, const float32x4_t &c1, const float32x4_t &c2, int w0, int w1, int w2, float wsum) {