From 6fc4eb19df2b3751e4979c24e67abb428c3bc261 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 24 Sep 2026 15:00:19 -0600 Subject: [PATCH] VFPU: Fix vrot with the angle in a destination lane The cosine is then taken of what vrot wrote to that lane: the sine, or zero. The IR looked at the sine lane instead of the lane holding the angle, and the legacy JITs ignored the overlap. The assembler refuses such a vrot, so those now leave it to the interpreter, and don't pair one with the vrot before it. Covered by the new cpu/vfpu/vrot test. Co-Authored-By: Claude Opus 5.5 (1M context) --- Core/MIPS/ARM/ArmCompVFPU.cpp | 10 ++++++++++ Core/MIPS/ARM64/Arm64CompVFPU.cpp | 10 ++++++++++ Core/MIPS/IR/IRCompVFPU.cpp | 18 +++++++++++++----- Core/MIPS/x86/CompVFPU.cpp | 10 ++++++++++ pspautotests | 2 +- test.py | 1 + 6 files changed, 45 insertions(+), 6 deletions(-) diff --git a/Core/MIPS/ARM/ArmCompVFPU.cpp b/Core/MIPS/ARM/ArmCompVFPU.cpp index 52a7273adc..c4b51f7b13 100644 --- a/Core/MIPS/ARM/ArmCompVFPU.cpp +++ b/Core/MIPS/ARM/ArmCompVFPU.cpp @@ -2250,6 +2250,16 @@ namespace MIPSComp if (vd2 >= 0) GetVectorRegs(dregs2, sz, vd2); GetVectorRegs(&sreg, V_Single, vs); + // With the angle in a destination lane, the cosine is taken of what was written there. + // The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot. + for (int i = 0; i < n; i++) { + if (dregs[i] == sreg) { + DISABLE; + } + if (vd2 >= 0 && dregs2[i] == sreg) { + vd2 = -1; + } + } int imm = (op >> 16) & 0x1f; diff --git a/Core/MIPS/ARM64/Arm64CompVFPU.cpp b/Core/MIPS/ARM64/Arm64CompVFPU.cpp index 791e25fa7e..a0424a3b24 100644 --- a/Core/MIPS/ARM64/Arm64CompVFPU.cpp +++ b/Core/MIPS/ARM64/Arm64CompVFPU.cpp @@ -2231,6 +2231,16 @@ namespace MIPSComp { if (vd2 >= 0) GetVectorRegs(dregs2, sz, vd2); GetVectorRegs(&sreg, V_Single, vs); + // With the angle in a destination lane, the cosine is taken of what was written there. + // The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot. + for (int i = 0; i < n; i++) { + if (dregs[i] == sreg) { + DISABLE; + } + if (vd2 >= 0 && dregs2[i] == sreg) { + vd2 = -1; + } + } int imm = (op >> 16) & 0x1f; diff --git a/Core/MIPS/IR/IRCompVFPU.cpp b/Core/MIPS/IR/IRCompVFPU.cpp index 480f31231f..42779c479c 100644 --- a/Core/MIPS/IR/IRCompVFPU.cpp +++ b/Core/MIPS/IR/IRCompVFPU.cpp @@ -2335,12 +2335,20 @@ namespace MIPSComp { } break; case 'c': - if (IsOverlapSafe(n, dregs, 1, sreg)) + if (IsOverlapSafe(n, dregs, 1, sreg)) { ir.Write(IROp::FCos, dregs[i], sreg[0]); - else if (dregs[sineLane] == sreg[0]) - ir.Write(IROp::FCos, dregs[i], IRVTEMP_0); - else - ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f); + } else { + // The cosine is taken of what the source lane got: the sine, or zero. + int srcLane = 0; + while (dregs[srcLane] != sreg[0]) { + srcLane++; + } + if (broadcastSine || srcLane == sineLane) { + ir.Write(IROp::FCos, dregs[i], IRVTEMP_0); + } else { + ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f); + } + } break; } } diff --git a/Core/MIPS/x86/CompVFPU.cpp b/Core/MIPS/x86/CompVFPU.cpp index 8729c1b7eb..123da1dfea 100644 --- a/Core/MIPS/x86/CompVFPU.cpp +++ b/Core/MIPS/x86/CompVFPU.cpp @@ -3759,6 +3759,16 @@ void Jit::Comp_VRot(MIPSOpcode op) { if (vd2 >= 0) GetVectorRegs(dregs2, sz, vd2); GetVectorRegs(&sreg, V_Single, vs); + // With the angle in a destination lane, the cosine is taken of what was written there. + // The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot. + for (int i = 0; i < n; i++) { + if (dregs[i] == sreg) { + DISABLE; + } + if (vd2 >= 0 && dregs2[i] == sreg) { + vd2 = -1; + } + } // Flush SIMD. fpr.SimpleRegsV(&sreg, V_Single, 0); diff --git a/pspautotests b/pspautotests index 80eb3ae038..6f03ee6457 160000 --- a/pspautotests +++ b/pspautotests @@ -1 +1 @@ -Subproject commit 80eb3ae038c17aafd28ba04dc8491a20c087f4eb +Subproject commit 6f03ee6457804144add92bfc47563cf60c443e4f diff --git a/test.py b/test.py index 6cd1da599d..99bc7cf629 100755 --- a/test.py +++ b/test.py @@ -102,6 +102,7 @@ tests_good = [ "cpu/vfpu/matrix", "cpu/vfpu/vavg", "cpu/vfpu/exact", + "cpu/vfpu/vrot", "cpu/icache/icache", "cpu/lsu/lsu", "cpu/lsu/llsc",