mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
softgpu: Truncate float->int conversions on SSE, like every other path
Vec4<float>::Cast<int>() used _mm_cvtps_epi32 under SSE, which rounds using MXCSR's mode, while NEON's vcvtq_s32_f32 and the scalar (T2)x fallback both truncate. Same split in Rasterizer's InterpolateI. That's on depth interpolation, so the differing values are written to the depth buffer and then compared - a one-LSB difference can flip a later GE_COMP_EQUAL pass and change a whole surface's visibility, not just a shade. Following MXCSR also meant anything that left a non-default rounding mode in the render thread would have changed rasterized output. Truncation is what two of the three paths already did, so SSE moves to match. Co-Authored-By: Claude Opus 5 <[email protected]> Claude-Session: https://claude.ai/code/session_01Vd8ntC2brCUtCrDJMqLbs8
This commit is contained in:
1 parent
e4337eed28
commit
f41fe77643
2 files changed
+5
-2
No files matched your search
+3
-1
@@ -595,7 +595,9 @@ public:
|
||||
Vec4<T2> Cast() const {
|
||||
if constexpr (std::is_same<T, float>::value && std::is_same<T2, int>::value) {
|
||||
#if defined(_M_SSE)
|
||||
return _mm_cvtps_epi32(SAFE_M128(vec));
|
||||
// Truncate, don't round. NEON's vcvtq_s32_f32 and the scalar (T2)x below both truncate,
|
||||
// and _mm_cvtps_epi32 would additionally follow MXCSR's rounding mode.
|
||||
return _mm_cvttps_epi32(SAFE_M128(vec));
|
||||
#elif PPSSPP_ARCH(ARM_NEON)
|
||||
return vcvtq_s32_f32(vec);
|
||||
#endif
|
||||
|
||||
@@ -56,7 +56,8 @@ static inline __m128 InterpolateF(const __m128 &c0, const __m128 &c1, const __m1
|
||||
}
|
||||
|
||||
static inline __m128i InterpolateI(const __m128i &c0, const __m128i &c1, const __m128i &c2, int w0, int w1, int w2, float wsum) {
|
||||
return _mm_cvtps_epi32(InterpolateF(_mm_cvtepi32_ps(c0), _mm_cvtepi32_ps(c1), _mm_cvtepi32_ps(c2), w0, w1, w2, wsum));
|
||||
// Truncate to match the NEON and scalar paths, see Vec4<float>::Cast<int>().
|
||||
return _mm_cvttps_epi32(InterpolateF(_mm_cvtepi32_ps(c0), _mm_cvtepi32_ps(c1), _mm_cvtepi32_ps(c2), w0, w1, w2, wsum));
|
||||
}
|
||||
#elif PPSSPP_ARCH(ARM64_NEON)
|
||||
static inline float32x4_t InterpolateF(const float32x4_t &c0, const float32x4_t &c1, const float32x4_t &c2, int w0, int w1, int w2, float wsum) {
|
||||
|
||||
Reference in new issue
Block a user