From 3fa67f22eddf3e7a35503578789d3f8def71e97c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 27 Aug 2026 16:42:26 +0200 Subject: [PATCH 1/6] Interpreter: honor the guest's FPU rounding mode and flush-to-zero Every JIT backend puts the host FPU into the mode fcr31 asks for (bits 0-1 and 24) before running emulated code, and takes it back out before calling any host code. The plain interpreter did none of that, so all its float math rounded to nearest with denormals intact no matter what the game had set - cpu/fpu/fpu fails under -i and passes under the JIT on exactly this. Move the helpers the IR interpreter already had for this out of IRInterpreter and into MIPS.cpp as ApplyHostRoundingMode/RestoreHostRoundingMode, and use them around the interpreter's run loop and single step, restoring around syscalls and replacement functions, which are host code. ctc1 re-applies immediately, since the interpreter has no block boundary to defer it to. round.w.s changes with it: it was floorf(x + 0.5f), which is half-away-from-zero rather than the half-to-even every JIT produces, and the add would now pick up the guest's rounding mode on top of that. round_ieee_754 is both correct and mode-independent, and is what cvt.w.s already used for the same rounding. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SfY7iFJEjmRXf1XGrTs4MF --- Core/MIPS/IR/IRInterpreter.cpp | 85 +--------------------------------- Core/MIPS/IR/IRInterpreter.h | 3 -- Core/MIPS/Interpreter.cpp | 20 +++++++- Core/MIPS/MIPS.cpp | 82 ++++++++++++++++++++++++++++++++ Core/MIPS/MIPS.h | 9 ++++ Core/MIPS/MIPSTables.cpp | 8 ++++ 6 files changed, 120 insertions(+), 87 deletions(-) diff --git a/Core/MIPS/IR/IRInterpreter.cpp b/Core/MIPS/IR/IRInterpreter.cpp index 27465a1616..7a6a764951 100644 --- a/Core/MIPS/IR/IRInterpreter.cpp +++ b/Core/MIPS/IR/IRInterpreter.cpp @@ -3,10 +3,6 @@ #include "ppsspp_config.h" -#if PPSSPP_PLATFORM(WINDOWS) && PPSSPP_ARCH(ARM64) -#include -#endif - #include "Common/BitSet.h" #include "Common/BitScan.h" #include "Common/Common.h" @@ -28,32 +24,6 @@ #include "Core/System.h" #include "Core/MIPS/MIPSTracer.h" -#if PPSSPP_ARCH(ARM64) - -// TODO: This should be put in some common header. -static inline u64 ARM64ReadFPCR() { -#if PPSSPP_PLATFORM(WINDOWS) - return _ReadStatusReg(ARM64_FPCR); -#else - // TODO: Try __builtin_arm_get_fpcr() - u64 fpcr; // not really 64-bit, just to match the register size. - asm volatile ("mrs %0, fpcr" : "=r" (fpcr)); - return fpcr; -#endif -} - -static inline void ARM64WriteFPCR(u64 fpcr) { -#if PPSSPP_PLATFORM(WINDOWS) - _WriteStatusReg(ARM64_FPCR, fpcr); -#else - // TODO: Try __builtin_arm_set_fpcr() - // Write back the modified FPCR - asm volatile ("msr fpcr, %0" : : "r" (fpcr)); -#endif -} - -#endif - #ifdef mips // Why do MIPS compilers define something so generic? Try to keep defined, at least... #undef mips @@ -110,57 +80,6 @@ u32 IRRunMemCheck(u32 pc, u32 addr) { return coreState != CORE_RUNNING_CPU ? 1 : 0; } -void IRApplyRounding(MIPSState *mips) { - u32 fcr1Bits = mips->fcr31 & 0x01000003; - // If these are 0, we just leave things as they are. - if (fcr1Bits) { - int rmode = fcr1Bits & 3; - bool ftz = (fcr1Bits & 0x01000000) != 0; -#if PPSSPP_ARCH(SSE2) - u32 csr = _mm_getcsr() & ~0x6000; - // Translate the rounding mode bits to X86, the same way as in Asm.cpp. - if (rmode & 1) { - rmode ^= 2; - } - csr |= rmode << 13; - - if (ftz) { - // Flush to zero - csr |= 0x8000; - } - _mm_setcsr(csr); -#elif PPSSPP_ARCH(ARM64) - u64 fpcr = ARM64ReadFPCR(); - // Translate MIPS to ARM rounding mode - static const u8 lookup[4] = {0, 3, 1, 2}; - - fpcr &= ~(3 << 22); // Clear bits [23:22] - fpcr |= ((u64)lookup[rmode] << 22); - - if (ftz) { - fpcr |= 1 << 24; - } - - ARM64WriteFPCR(fpcr); -#endif - } -} - -void IRRestoreRounding() { -#if PPSSPP_ARCH(SSE2) - // TODO: We should avoid this if we didn't apply rounding in the first place. - // In the meantime, clear out FTZ and rounding mode bits. - u32 csr = _mm_getcsr(); - csr &= ~(7 << 13); - _mm_setcsr(csr); -#elif PPSSPP_ARCH(ARM64) - u64 fpcr = ARM64ReadFPCR(); // not really 64-bit, just to match the regsiter size. - fpcr &= ~(7 << 22); // Clear bits [23:22] for rounding, 24 for FTZ - // Write back the modified FPCR - ARM64WriteFPCR(fpcr); -#endif -} - u32 IRInterpret(MIPSState *mips, const IRInst *inst) { while (true) { switch (inst->op) { @@ -1257,10 +1176,10 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) { break; case IROp::ApplyRoundingMode: - IRApplyRounding(mips); + ApplyHostRoundingMode(mips); break; case IROp::RestoreRoundingMode: - IRRestoreRounding(); + RestoreHostRoundingMode(); break; case IROp::UpdateRoundingMode: // TODO: Implement diff --git a/Core/MIPS/IR/IRInterpreter.h b/Core/MIPS/IR/IRInterpreter.h index 85e3795069..f555552f49 100644 --- a/Core/MIPS/IR/IRInterpreter.h +++ b/Core/MIPS/IR/IRInterpreter.h @@ -11,9 +11,6 @@ u32 IRRunBreakpoint(u32 pc); u32 IRRunMemCheck(u32 pc, u32 addr); u32 IRInterpret(MIPSState *ms, const IRInst *inst); -void IRApplyRounding(); -void IRRestoreRounding(); - template u32 RunValidateAddress(u32 pc, u32 addr, u32 isWrite) { static_assert(alignment <= 16); diff --git a/Core/MIPS/Interpreter.cpp b/Core/MIPS/Interpreter.cpp index 8c8af8c931..7fd2be1d70 100644 --- a/Core/MIPS/Interpreter.cpp +++ b/Core/MIPS/Interpreter.cpp @@ -82,6 +82,8 @@ int MIPS_InterpretSingleStep(MIPSState *mips) { return 0; } MIPSOpcode op = Memory::Read_Opcode_JIT(mips->pc); // now unchecked + // Same reason as the run loop in MIPSInterpret_RunUntil - see ApplyHostRoundingMode. + ApplyHostRoundingMode(mips); if (mips->inDelaySlot) { MIPSInterpret(mips, op); if (mips->inDelaySlot) { @@ -91,6 +93,7 @@ int MIPS_InterpretSingleStep(MIPSState *mips) { } else { MIPSInterpret(mips, op); } + RestoreHostRoundingMode(); return 1; } @@ -235,7 +238,10 @@ namespace MIPSInt { mips->pc += 4; } mips->inDelaySlot = false; + // HLE code is host code - it must not run under the guest's rounding mode. + RestoreHostRoundingMode(); CallSyscallWithPC(op, syscallPC); + ApplyHostRoundingMode(mips); } void Int_Sync(MIPSState *mips, MIPSOpcode op) { @@ -718,6 +724,13 @@ namespace MIPSInt { if (MIPSComp::jit) { // In case of DISABLE, we need to tell jit we updated FCR31. MIPSComp::jit->UpdateFCR31(); + } else { + // The interpreter emulates the rounding mode by putting the host FPU in it, + // so it has to switch right here rather than at the next block boundary. + // Restore first: the new value may be back to the default, which Apply + // deliberately doesn't write. + RestoreHostRoundingMode(); + ApplyHostRoundingMode(mips); } } else { WARN_LOG_REPORT(Log::CPU, "WriteFCR: Unexpected reg %d (value %08x)", fs, value); @@ -1100,7 +1113,9 @@ namespace MIPSInt { } switch (op & 0x3f) { - case 12: FsI(fd) = (int)floorf(F(fs)+0.5f); break; //round.w.s + // round.w.s is round-half-to-even, not half-away-from-zero - and its mode is fixed, + // so unlike cvt.w.s below it must not follow fcr31. round_ieee_754 is both. + case 12: FsI(fd) = (int)round_ieee_754(F(fs)); break; //round.w.s case 13: //trunc.w.s if (F(fs) >= 0.0f) { FsI(fd) = (int)floorf(F(fs)); @@ -1247,7 +1262,10 @@ namespace MIPSInt { int index = op.encoding & 0xFFFFFF; const ReplacementTableEntry *entry = GetReplacementFunc(index); if (entry && entry->replaceFunc && (entry->flags & REPFLAG_DISABLED) == 0) { + // Like a syscall, a replacement function is host code - see Int_Syscall. + RestoreHostRoundingMode(); int cycles = entry->replaceFunc(); + ApplyHostRoundingMode(mips); if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) { // Interpret the original instruction under the hook. diff --git a/Core/MIPS/MIPS.cpp b/Core/MIPS/MIPS.cpp index c733131cad..8ba3733d0d 100644 --- a/Core/MIPS/MIPS.cpp +++ b/Core/MIPS/MIPS.cpp @@ -20,8 +20,14 @@ #include #include +#include "ppsspp_config.h" + +#if PPSSPP_PLATFORM(WINDOWS) && PPSSPP_ARCH(ARM64) +#include +#endif #include "Common/CommonTypes.h" +#include "Common/Math/SIMDHeaders.h" #include "Common/Serialize/Serializer.h" #include "Common/Serialize/SerializeFuncs.h" #include "Core/ConfigValues.h" @@ -42,6 +48,82 @@ MIPSState *currentMIPS = &mipsr4k; MIPSDebugInterface debugr4k(&mipsr4k); MIPSDebugInterface *currentDebugMIPS = &debugr4k; +#if PPSSPP_ARCH(ARM64) + +static inline u64 ARM64ReadFPCR() { +#if PPSSPP_PLATFORM(WINDOWS) + return _ReadStatusReg(ARM64_FPCR); +#else + // TODO: Try __builtin_arm_get_fpcr() + u64 fpcr; // not really 64-bit, just to match the register size. + asm volatile ("mrs %0, fpcr" : "=r" (fpcr)); + return fpcr; +#endif +} + +static inline void ARM64WriteFPCR(u64 fpcr) { +#if PPSSPP_PLATFORM(WINDOWS) + _WriteStatusReg(ARM64_FPCR, fpcr); +#else + // TODO: Try __builtin_arm_set_fpcr() + // Write back the modified FPCR + asm volatile ("msr fpcr, %0" : : "r" (fpcr)); +#endif +} + +#endif + +void ApplyHostRoundingMode(const MIPSState *mips) { + u32 fcr1Bits = mips->fcr31 & 0x01000003; + // If these are 0, we just leave things as they are. + if (fcr1Bits) { + int rmode = fcr1Bits & 3; + bool ftz = (fcr1Bits & 0x01000000) != 0; +#if PPSSPP_ARCH(SSE2) + u32 csr = _mm_getcsr() & ~0x6000; + // Translate the rounding mode bits to X86, the same way as in Asm.cpp. + if (rmode & 1) { + rmode ^= 2; + } + csr |= rmode << 13; + + if (ftz) { + // Flush to zero + csr |= 0x8000; + } + _mm_setcsr(csr); +#elif PPSSPP_ARCH(ARM64) + u64 fpcr = ARM64ReadFPCR(); + // Translate MIPS to ARM rounding mode + static const u8 lookup[4] = {0, 3, 1, 2}; + + fpcr &= ~(3 << 22); // Clear bits [23:22] + fpcr |= ((u64)lookup[rmode] << 22); + + if (ftz) { + fpcr |= 1 << 24; + } + + ARM64WriteFPCR(fpcr); +#endif + } +} + +void RestoreHostRoundingMode() { + // TODO: We should avoid this if we didn't apply rounding in the first place. + // In the meantime, clear out FTZ and rounding mode bits. +#if PPSSPP_ARCH(SSE2) + u32 csr = _mm_getcsr(); + csr &= ~(7 << 13); + _mm_setcsr(csr); +#elif PPSSPP_ARCH(ARM64) + u64 fpcr = ARM64ReadFPCR(); // not really 64-bit, just to match the register size. + fpcr &= ~(7 << 22); // Clear bits [23:22] for rounding, 24 for FTZ + // Write back the modified FPCR + ARM64WriteFPCR(fpcr); +#endif +} + u8 voffset[128]; u8 fromvoffset[128]; diff --git a/Core/MIPS/MIPS.h b/Core/MIPS/MIPS.h index 9ae7dd87ca..2c5cc2847e 100644 --- a/Core/MIPS/MIPS.h +++ b/Core/MIPS/MIPS.h @@ -293,3 +293,12 @@ extern MIPSDebugInterface *currentDebugMIPS; extern MIPSState mipsr4k; extern const float cst_constants[32]; + +// The guest's rounding mode and flush-to-zero flag (fcr31 bits 0-1 and 24) are emulated by putting +// the *host* FPU into the matching mode, since we do the arithmetic with plain host float ops. +// That mode must not be left on while running anything that isn't emulating a guest instruction - +// HLE syscalls, replacement functions, the GPU - so an emulation loop applies it on entry and +// restores it before calling out, the way the JITs do (see Jit::ApplyRoundingMode). +// Apply is cheap when the guest is in the default mode (by far the common case): it does nothing. +void ApplyHostRoundingMode(const MIPSState *mips); +void RestoreHostRoundingMode(); diff --git a/Core/MIPS/MIPSTables.cpp b/Core/MIPS/MIPSTables.cpp index f84b4b3fe1..0ad08e5556 100644 --- a/Core/MIPS/MIPSTables.cpp +++ b/Core/MIPS/MIPSTables.cpp @@ -1272,6 +1272,12 @@ int MIPSInterpret_RunUntil(MIPSState *mips, u64 globalTicks) { while (coreState == CORE_RUNNING_CPU) { CoreTiming::Advance(mips); + // The emulated instructions below do their float math with plain host float ops, so the + // host FPU has to be in the guest's rounding mode while they run - and back in the normal + // one whenever we're not running them, which is what the JITs do too. Calls out from + // inside the run loops (syscalls, replacement functions) restore it themselves. + ApplyHostRoundingMode(mips); + uint64_t ticksLeft = globalTicks - CoreTiming::GetTicks(mips); if (g_breakpoints.HasBreakPoints() || g_breakpoints.HasMemChecks() || g_breakpoints.HasRegBreakpoints() || ticksLeft <= mips->downcount) { RunUntilDowncountZeroWithChecks(mips, globalTicks); @@ -1279,6 +1285,8 @@ int MIPSInterpret_RunUntil(MIPSState *mips, u64 globalTicks) { RunUntilDowncountZeroFast(mips); } + RestoreHostRoundingMode(); + if (CoreTiming::GetTicks(mips) > globalTicks) { // DEBUG_LOG(Log::CPU, "Hit the max ticks, bailing 1 : %llu, %llu", globalTicks, CoreTiming::GetTicks(mips)); return 1; From 484898385ffbea45e3f0ede6f7a893a4842ceefe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 27 Aug 2026 16:43:43 +0200 Subject: [PATCH 2/6] Interpreter: don't spin forever on a memory access that's set to be ignored The alignment checks added in d8edeb7649 return out of the instruction handler without advancing PC, so with IgnoreBadMemAccess - which is the default, and which makes Core_MemoryException log and return - the run loop comes straight back to the same instruction and never gets past it. cpu/crash/crash_read_u32 under -i logged the same SIGSEGV 585096 times in 30 seconds before being killed; the JIT runs it to completion. Continue instead, the way Memory::Read_U32 did before those checks existed and the way the JIT's safe-memory path still does: loads produce zero, stores are dropped, PC advances. When the exception is set to break rather than ignore, Core_Break has already stopped the core by the time we get here, so nothing changes for that case. lv.q/sv.q are left alone - they already fall through and do the access. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SfY7iFJEjmRXf1XGrTs4MF --- Core/MIPS/Interpreter.cpp | 43 ++++++++++++++++++++--------------- Core/MIPS/InterpreterVFPU.cpp | 13 ++++++----- 2 files changed, 32 insertions(+), 24 deletions(-) diff --git a/Core/MIPS/Interpreter.cpp b/Core/MIPS/Interpreter.cpp index 7fd2be1d70..2c94330f8c 100644 --- a/Core/MIPS/Interpreter.cpp +++ b/Core/MIPS/Interpreter.cpp @@ -420,9 +420,10 @@ namespace MIPSInt { if (rt != 0) { if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "ll"); - return; + R(rt) = 0; + } else { + R(rt) = Memory::ReadUnchecked_U32(addr); } - R(rt) = Memory::ReadUnchecked_U32(addr); } mips->llBit = 1; break; @@ -430,9 +431,9 @@ namespace MIPSInt { if (mips->llBit) { if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::WRITE_WORD, "sc"); - return; + } else { + Memory::WriteUnchecked_U32(R(rt), addr); } - Memory::WriteUnchecked_U32(R(rt), addr); if (rt != 0) { R(rt) = 1; } @@ -501,7 +502,8 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 1, PC, MemoryExceptionType::READ_WORD, "lb"); - return; + R(rt) = 0; + break; } R(rt) = SignExtend8ToU32(Memory::ReadUnchecked_U8(addr)); break; //lb @@ -513,7 +515,8 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 2, PC, MemoryExceptionType::READ_WORD, "lh"); - return; + R(rt) = 0; + break; } R(rt) = SignExtend16ToU32(Memory::ReadUnchecked_U16(addr)); break; //lh @@ -525,7 +528,8 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lw"); - return; + R(rt) = 0; + break; } R(rt) = Memory::ReadUnchecked_U32(addr); break; //lw @@ -537,7 +541,8 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 1, PC, MemoryExceptionType::READ_WORD, "lbu"); - return; + R(rt) = 0; + break; } R(rt) = Memory::ReadUnchecked_U8(addr); break; //lbu @@ -549,7 +554,8 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 2, PC, MemoryExceptionType::READ_WORD, "lhu"); - return; + R(rt) = 0; + break; } R(rt) = Memory::ReadUnchecked_U16(addr); break; //lhu @@ -561,7 +567,7 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 1, PC, MemoryExceptionType::WRITE_WORD, "sb"); - return; + break; } Memory::WriteUnchecked_U8(R(rt), addr); break; //sb @@ -573,7 +579,7 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 2, PC, MemoryExceptionType::WRITE_WORD, "sh"); - return; + break; } Memory::WriteUnchecked_U16(R(rt), addr); break; //sh @@ -585,7 +591,7 @@ namespace MIPSInt { break; } Core_MemoryException(addr, 4, PC, MemoryExceptionType::WRITE_WORD, "sw"); - return; + break; } Memory::WriteUnchecked_U32(R(rt), addr); break; //sw @@ -597,7 +603,7 @@ namespace MIPSInt { // Not checking for alignment here - the actual read will be aligned. if (!Memory::IsValidAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lwl"); - return; + break; } u32 shift = (addr & 3) * 8; u32 mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); @@ -611,7 +617,7 @@ namespace MIPSInt { // Not checking for alignment here - the actual read will be aligned. if (!Memory::IsValidAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lwr"); - return; + break; } u32 shift = (addr & 3) * 8; u32 mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); @@ -626,7 +632,7 @@ namespace MIPSInt { // Not checking for alignment here - the actual read/write will be aligned. if (!Memory::IsValidAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::WRITE_WORD, "swl"); - return; + break; } u32 shift = (addr & 3) * 8; u32 mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); @@ -640,7 +646,7 @@ namespace MIPSInt { // Not checking for alignment here - the actual read/write will be aligned. if (!Memory::IsValidAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::WRITE_WORD, "swr"); - return; + break; } u32 shift = (addr & 3) << 3; u32 mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); @@ -666,14 +672,15 @@ namespace MIPSInt { case 49: if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lwc1"); - return; + FI(ft) = 0; + break; } FI(ft) = Memory::ReadUnchecked_U32(addr); break; //lwc1 case 57: if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::WRITE_WORD, "swc1"); - return; + break; } Memory::WriteUnchecked_U32(FI(ft), addr); break; //swc1 diff --git a/Core/MIPS/InterpreterVFPU.cpp b/Core/MIPS/InterpreterVFPU.cpp index d378158cf8..70dca92e3d 100644 --- a/Core/MIPS/InterpreterVFPU.cpp +++ b/Core/MIPS/InterpreterVFPU.cpp @@ -219,7 +219,7 @@ namespace MIPSInt if ((op & 2) == 0) { if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::READ_WORD, "lvl.q"); - return; + break; } // It's an LVL for (int i = 0; i < offset + 1; i++) { @@ -228,7 +228,7 @@ namespace MIPSInt } else { if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::READ_WORD, "lvr.q"); - return; + break; } // It's an LVR for (int i = 0; i < (3 - offset) + 1; i++) { @@ -268,7 +268,7 @@ namespace MIPSInt if ((op & 2) == 0) { if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::WRITE_WORD, "svl.q"); - return; + break; } // It's an SVL for (int i = 0; i < offset + 1; i++) @@ -278,7 +278,7 @@ namespace MIPSInt } else { if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::WRITE_WORD, "svr.q"); - return; + break; } // It's an SVR for (int i = 0; i < (3 - offset) + 1; i++) { @@ -1757,14 +1757,15 @@ namespace MIPSInt case 50: //lv.s if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lv.s"); - return; + VI(vt) = 0; + break; } VI(vt) = Memory::ReadUnchecked_U32(addr); break; case 58: //sv.s if (!Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 4, PC, MemoryExceptionType::WRITE_WORD, "sv.s"); - return; + break; } Memory::WriteUnchecked_U32(VI(vt), addr); break; From d79117146ddfad4c8c4f473342d64d977d5d3017 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 27 Aug 2026 19:01:23 +0200 Subject: [PATCH 3/6] Interpreter: fix the ins mask when the encoded msb is below pos ins derived its width as (_SIZE + 1) - pos, which is zero or negative when the encoded msb is below pos: the following shift is then 32 or more, undefined, and on x86 produces an all-ones mask that writes bits the JITs don't touch. Build the mask from msb and shift it down instead, which is what the JITs do and can't shift out of range. Hardware calls that encoding unpredictable, so consistency is all that's wanted here. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SfY7iFJEjmRXf1XGrTs4MF --- Core/MIPS/Interpreter.cpp | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/Core/MIPS/Interpreter.cpp b/Core/MIPS/Interpreter.cpp index 2c94330f8c..68f2a55e60 100644 --- a/Core/MIPS/Interpreter.cpp +++ b/Core/MIPS/Interpreter.cpp @@ -1088,10 +1088,16 @@ namespace MIPSInt { break; case 0x4: //ins { - int size = (_SIZE + 1) - pos; - u32 sourcemask = 0xFFFFFFFFUL >> (32 - size); - u32 destmask = sourcemask << pos; - R(rt) = (R(rt) & ~destmask) | ((R(rs)&sourcemask) << pos); + // The size field actually holds msb (= pos + size - 1), so build the mask from + // that and shift it down, the way the JITs do. Computing the width as + // (_SIZE + 1) - pos instead would shift by 32 or more when msb < pos - undefined + // behavior, and on x86 it yields an all-ones mask that writes bits the JITs leave + // alone. Hardware calls that encoding unpredictable, so all we need is to be + // consistent and not invoke UB. + const u32 mask = 0xFFFFFFFFUL >> (31 - _SIZE); + const u32 sourcemask = mask >> pos; + const u32 destmask = sourcemask << pos; + R(rt) = (R(rt) & ~destmask) | ((R(rs) & sourcemask) << pos); } break; } From 53f616faa3789cf3e17e90c9b905b04ea41ee8f9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 27 Aug 2026 19:01:23 +0200 Subject: [PATCH 4/6] Interpreter: vrot cleared the wrong lane's D prefix saturation vrot clears the D prefix for the cosine lane, since the prefix doesn't apply there, but shifted the saturation mask by cosineLane rather than cosineLane * 2. That field is two bits per element - ApplyPrefixD reads it as (data >> (i * 2)) & 3, and every other site in the file shifts accordingly - so for lanes 1 and up it cleared the wrong lane's saturation and left the cosine lane's in place. The mask field next to it is one bit per element and was already right. Only reachable through the interpreter, but that includes the JITs, which fall back here for any prefixed vrot. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SfY7iFJEjmRXf1XGrTs4MF --- Core/MIPS/InterpreterVFPU.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Core/MIPS/InterpreterVFPU.cpp b/Core/MIPS/InterpreterVFPU.cpp index 70dca92e3d..defd19540b 100644 --- a/Core/MIPS/InterpreterVFPU.cpp +++ b/Core/MIPS/InterpreterVFPU.cpp @@ -1642,7 +1642,8 @@ namespace MIPSInt } // D prefix works, just not for the cosine lane. - uint32_t dprefixRemove = (3 << cosineLane) | (1 << (8 + cosineLane)); + // The saturation field is two bits per element (see ApplyPrefixD), the mask field one. + uint32_t dprefixRemove = (3 << (cosineLane * 2)) | (1 << (8 + cosineLane)); mips->vfpuCtrl[VFPU_CTRL_DPREFIX] &= 0xFFFFF ^ dprefixRemove; ApplyPrefixD(mips, d, sz); WriteVector(mips, d, sz, vd); From 991d43a71337b5970bf2794a595b0845ddc0b1a1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 27 Aug 2026 19:18:30 +0200 Subject: [PATCH 5/6] Interpreter: reject misaligned lv.q/sv.q instead of carrying them out These two raised the memory exception and then went ahead and did the access anyway, unlike every other load/store here. A quadword access that isn't 16-byte aligned isn't valid, so there's nothing to carry out - and on 64-bit, where GetPointerUnchecked is base + address with no masking, an address that failed the validity check meant dereferencing whatever that landed on. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SfY7iFJEjmRXf1XGrTs4MF --- Core/MIPS/InterpreterVFPU.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/Core/MIPS/InterpreterVFPU.cpp b/Core/MIPS/InterpreterVFPU.cpp index defd19540b..40c8393863 100644 --- a/Core/MIPS/InterpreterVFPU.cpp +++ b/Core/MIPS/InterpreterVFPU.cpp @@ -240,8 +240,11 @@ namespace MIPSInt break; case 54: //lv.q + // A quadword access has to be 16-byte aligned, so a misaligned one is simply + // rejected - we don't try to carry it out anyway, same as every other path here. if ((addr & 0xF) || !Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::READ_WORD, "lv.q"); + break; } #ifndef COMMON_BIG_ENDIAN @@ -289,8 +292,10 @@ namespace MIPSInt } case 62: //sv.q + // See lv.q above. if ((addr & 0xF) || !Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::WRITE_WORD, "sv.q"); + break; } #ifndef COMMON_BIG_ENDIAN f = reinterpret_cast(Memory::GetPointerWriteUnchecked(addr)); From 3a1162475bf8422a2995c01df29d50a2195bdc7e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Fri, 28 Aug 2026 14:32:03 +0200 Subject: [PATCH 6/6] Interpreter: Return 0 on all types of bad memory reads - probably best for IgnoreBadMemoryAccess But ideally I want to get rid of this at some point. --- Core/MIPS/Interpreter.cpp | 25 +++++++++++++++++-------- Core/MIPS/InterpreterVFPU.cpp | 16 ++++++++-------- 2 files changed, 25 insertions(+), 16 deletions(-) diff --git a/Core/MIPS/Interpreter.cpp b/Core/MIPS/Interpreter.cpp index 68f2a55e60..163398324d 100644 --- a/Core/MIPS/Interpreter.cpp +++ b/Core/MIPS/Interpreter.cpp @@ -116,7 +116,7 @@ static u8 ReadMMIO_U8(MIPSState *mips, u32 addr) { static u16 ReadMMIO_U16(MIPSState *mips, u32 addr) { if (!Memory::IsKernelCodeAddress(mips->pc)) { Core_MemoryException(addr, 2, mips->pc, MemoryExceptionType::READ_WORD, "Kernel mode only"); - return (u8)UNKNOWN_MMIO_POISON; + return (u16)UNKNOWN_MMIO_POISON; } WARN_LOG(Log::CPU, "Unhandled MMIO Read16 at %08x", addr); return (u16)UNKNOWN_MMIO_POISON; @@ -125,7 +125,7 @@ static u16 ReadMMIO_U16(MIPSState *mips, u32 addr) { static u32 ReadMMIO_U32(MIPSState *mips, u32 addr) { if (!Memory::IsKernelCodeAddress(mips->pc)) { Core_MemoryException(addr, 4, mips->pc, MemoryExceptionType::READ_WORD, "Kernel mode only"); - return (u8)UNKNOWN_MMIO_POISON; + return UNKNOWN_MMIO_POISON; } if (GpioMMIO::IsGpioAddress(addr)) { return GpioMMIO::Read32(addr); @@ -434,6 +434,8 @@ namespace MIPSInt { } else { Memory::WriteUnchecked_U32(R(rt), addr); } + // Report success even if the store got dropped - reporting failure just makes + // the usual retry loop spin on the same bad address forever. if (rt != 0) { R(rt) = 1; } @@ -481,6 +483,11 @@ namespace MIPSInt { PC += 4; } + // On a bad access that's set to be ignored (the default), we mirror what + // Memory::ReadOrException_*/WriteOrException_* do, which is also what the JIT's slow + // path calls: loads produce zero, stores are dropped, and PC advances either way. + // Returning without advancing PC instead would just re-execute the same instruction forever. + // When the access is set to break, Core_MemoryException has already stopped the core. void Int_ITypeMem(MIPSState *mips, MIPSOpcode op) { int imm = (signed short)(op&0xFFFF); int rt = _RT; @@ -601,12 +608,13 @@ namespace MIPSInt { case 34: //lwl { // Not checking for alignment here - the actual read will be aligned. - if (!Memory::IsValidAddress(addr)) { + u32 mem = 0; + if (Memory::IsValidAddress(addr)) { + mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); + } else { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lwl"); - break; } u32 shift = (addr & 3) * 8; - u32 mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); u32 result = ( u32(R(rt)) & (0x00ffffff >> shift) ) | ( mem << (24 - shift) ); R(rt) = result; } @@ -615,12 +623,13 @@ namespace MIPSInt { case 38: //lwr { // Not checking for alignment here - the actual read will be aligned. - if (!Memory::IsValidAddress(addr)) { + u32 mem = 0; + if (Memory::IsValidAddress(addr)) { + mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); + } else { Core_MemoryException(addr, 4, PC, MemoryExceptionType::READ_WORD, "lwr"); - break; } u32 shift = (addr & 3) * 8; - u32 mem = Memory::ReadUnchecked_U32(addr & 0xfffffffc); u32 regval = R(rt); u32 result = ( regval & (0xffffff00 << (24 - shift)) ) | ( mem >> shift ); R(rt) = result; diff --git a/Core/MIPS/InterpreterVFPU.cpp b/Core/MIPS/InterpreterVFPU.cpp index 40c8393863..5947ba6dc5 100644 --- a/Core/MIPS/InterpreterVFPU.cpp +++ b/Core/MIPS/InterpreterVFPU.cpp @@ -216,23 +216,22 @@ namespace MIPSInt float d[4]; ReadVector(mips, d, V_Quad, vt); int offset = (addr >> 2) & 3; + const bool valid = Memory::IsValid4AlignedAddress(addr); if ((op & 2) == 0) { - if (!Memory::IsValid4AlignedAddress(addr)) { + if (!valid) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::READ_WORD, "lvl.q"); - break; } // It's an LVL for (int i = 0; i < offset + 1; i++) { - d[3 - i] = Memory::ReadUnchecked_Float(addr - 4 * i); + d[3 - i] = valid ? Memory::ReadUnchecked_Float(addr - 4 * i) : 0.0f; } } else { - if (!Memory::IsValid4AlignedAddress(addr)) { + if (!valid) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::READ_WORD, "lvr.q"); - break; } // It's an LVR for (int i = 0; i < (3 - offset) + 1; i++) { - d[i] = Memory::ReadUnchecked_Float(addr + 4 * i); + d[i] = valid ? Memory::ReadUnchecked_Float(addr + 4 * i) : 0.0f; } } WriteVector(mips, d, V_Quad, vt); @@ -244,13 +243,14 @@ namespace MIPSInt // rejected - we don't try to carry it out anyway, same as every other path here. if ((addr & 0xF) || !Memory::IsValid4AlignedAddress(addr)) { Core_MemoryException(addr, 16, PC, MemoryExceptionType::READ_WORD, "lv.q"); + const float zero[4]{}; + WriteVector(mips, zero, V_Quad, vt); break; } #ifndef COMMON_BIG_ENDIAN cf = reinterpret_cast(Memory::GetPointerUnchecked(addr)); - if (cf) - WriteVector(mips, cf, V_Quad, vt); + WriteVector(mips, cf, V_Quad, vt); #else float lvqd[4];