From 9f90512ef6d3ae538bab43d0c03885a5c7bd0376 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sun, 16 Aug 2026 12:07:32 +0200 Subject: [PATCH] Make instruction cache invalidation (for us, jit cache invalidation) clearer --- Core/Core.cpp | 34 ++++++-- Core/Core.h | 5 +- Core/CwCheat.cpp | 25 +++--- Core/Debugger/Breakpoints.cpp | 4 +- Core/Debugger/WebSocket/MemorySubscriber.cpp | 11 +-- Core/HLE/HLEHelperThread.cpp | 4 +- Core/HLE/HLEHelperThread.h | 6 +- Core/HLE/ReplaceTables.cpp | 8 +- Core/HLE/sceDmac.cpp | 12 +-- Core/HLE/sceIo.cpp | 4 +- Core/HLE/sceKernel.cpp | 30 +++---- Core/HLE/sceKernel.h | 2 + Core/HLE/sceKernelInterrupt.cpp | 5 +- Core/HLE/sceKernelModule.cpp | 18 ++-- Core/HW/AsyncIOManager.cpp | 2 +- Core/MIPS/Interpreter.cpp | 2 +- Core/MIPS/Interpreter.h | 2 +- Core/MIPS/MIPS.cpp | 87 ++++++++++++-------- Core/MIPS/MIPS.h | 31 +++++-- Core/MIPS/MIPSAsm.cpp | 2 +- Core/MIPS/x86/RegCacheFPU.cpp | 6 +- Core/MIPS/x86/RegCacheFPU.h | 2 +- Core/System.cpp | 12 +-- UI/EmuScreen.cpp | 2 +- UI/ImDebugger/ImDisasmView.cpp | 4 +- Windows/Debugger/CtrlDisAsmView.cpp | 13 +-- 26 files changed, 198 insertions(+), 135 deletions(-) diff --git a/Core/Core.cpp b/Core/Core.cpp index 9ce6cefcfe..76ebd19689 100644 --- a/Core/Core.cpp +++ b/Core/Core.cpp @@ -30,7 +30,6 @@ #include "Common/Profiler/Profiler.h" #include "Common/GPU/GraphicsContext.h" -#include "Common/Thread/ThreadUtil.h" #include "Common/Log.h" #include "Common/StringUtils.h" #include "Core/Core.h" @@ -43,9 +42,10 @@ #include "Core/Debugger/Breakpoints.h" #include "Core/MIPS/MIPS.h" #include "Core/MIPS/MIPSAnalyst.h" -#include "Core/HLE/sceNetAdhoc.h" #include "Core/HLE/sceKernelModule.h" +#include "Core/HLE/sceKernelThread.h" #include "Core/MIPS/MIPSTracer.h" +#include "Core/CoreTiming.h" #include "GPU/Debugger/Stepping.h" #include "GPU/GPU.h" @@ -299,8 +299,19 @@ bool Core_GetPowerSaving() { return powerSaving; } +void Core_ReenterDispatcher() { + if (coreState == CORE_RUNNING_CPU) { + // This will flip back into CORE_RUNNING_CPU. + coreState = CORE_REENTER_DISPATCH; + } +} + void Core_RunLoopUntil(u64 globalticks) { while (true) { + // Drain any functions queued up by Core_RunOnCPUThread() from other threads. Doing this at the + // top of this loop means it's reached at least once per call (i.e. about once per host frame) + // even while the CPU is fully running, and continuously (in a tight spin) while it's stepping/paused. + Core_ProcessCPUQueue(); switch (coreState) { case CORE_POWERDOWN: case CORE_RUNTIME_ERROR: @@ -308,12 +319,25 @@ void Core_RunLoopUntil(u64 globalticks) { return; case CORE_STEPPING_CPU: case CORE_STEPPING_GE: + { + CoreState preState = coreState; if (Core_ProcessStepping(currentDebugMIPS)) { + if (coreState == CORE_REENTER_DISPATCH) { + coreState = preState; + } return; } break; + } case CORE_RUNNING_CPU: mipsr4k.RunLoopUntil(globalticks); + if (coreState == CORE_RUNNING_CPU) { + // If we are still running, we must have reached the end of a frame. + coreState = CORE_NEXTFRAME; + } else if (coreState == CORE_REENTER_DISPATCH) { + // Back to running right away. + coreState = CORE_RUNNING_CPU; + } if (g_breakAfterFrame && coreState == CORE_NEXTFRAME) { g_breakAfterFrame = false; g_breakReason = BreakReason::AfterFrame; @@ -383,6 +407,7 @@ static void Core_PerformCPUStep(MIPSDebugInterface *cpu, CPUStepType stepType, i for (int i = 0; i < stepSize; i++) { currentMIPS->SingleStep(); } + CoreTiming::Advance(currentMIPS); break; } case CPUStepType::Over: @@ -462,11 +487,6 @@ static void Core_PerformCPUStep(MIPSDebugInterface *cpu, CPUStepType stepType, i static bool Core_ProcessStepping(MIPSDebugInterface *cpu) { Core_StateProcessed(); - // Drain any functions queued up by Core_RunOnCPUThread() from other threads. Doing this at the - // top of this loop means it's reached at least once per call (i.e. about once per host frame) - // even while the CPU is fully running, and continuously (in a tight spin) while it's stepping/paused. - Core_ProcessCPUQueue(); - // Check if there's any pending save state actions. SaveState::Process(); diff --git a/Core/Core.h b/Core/Core.h index 2bd2ee72f6..0d7d371907 100644 --- a/Core/Core.h +++ b/Core/Core.h @@ -118,7 +118,9 @@ enum CoreState { // Emulation is running normally. CORE_RUNNING_CPU = 0, // Emulation was running normally, just reached the end of a frame. - CORE_NEXTFRAME = 1, + CORE_NEXTFRAME, + // Set this when running to bounce out from the dispatcher and just go back in again. Useful for things like cache clears. + CORE_REENTER_DISPATCH, // Emulation is paused, CPU thread is sleeping. CORE_STEPPING_CPU, // Can be used for recoverable runtime errors (ignored memory exceptions) // Core is not running. @@ -150,6 +152,7 @@ void Core_SetPowerSaving(bool mode); bool Core_GetPowerSaving(); void Core_RunLoopUntil(u64 globalticks); +void Core_ReenterDispatcher(); // If you've done things that mess with caches, call this so we can run deferred operations. // Runs a function on the CPU thread - the thread that calls Core_RunLoopUntil (and thus, indirectly, // NativeFrame). Useful for code running on unrelated threads (like the WebSocket debugger) that needs to diff --git a/Core/CwCheat.cpp b/Core/CwCheat.cpp index 8fa4ca2c32..584bfd9ae2 100644 --- a/Core/CwCheat.cpp +++ b/Core/CwCheat.cpp @@ -303,24 +303,25 @@ void hleCheat(u64 userdata, int cyclesLate) { // Horrible hack for Tony Hawk - Underground 2. See #3854. Avoids crashing somehow // but still causes regular JIT invalidations which causes stutters. if (gameTitle == "ULUS10014") { - currentMIPS->InvalidateICache(0x08865600, 72); - currentMIPS->InvalidateICache(0x08865690, 4); + currentMIPS->InvalidateICacheRangeDeferred(0x08865600, 72); + currentMIPS->InvalidateICacheRangeDeferred(0x08865690, 4); } else if (gameTitle == "ULES00033" || gameTitle == "ULES00034" || gameTitle == "ULES00035") { // euro, also 34 and 35 - currentMIPS->InvalidateICache(0x088655D8, 72); - currentMIPS->InvalidateICache(0x08865668, 4); + currentMIPS->InvalidateICacheRangeDeferred(0x088655D8, 72); + currentMIPS->InvalidateICacheRangeDeferred(0x08865668, 4); } else if (gameTitle == "ULUS10138") { // MTX MotoTrax US - currentMIPS->InvalidateICache(0x0886DCC0, 72); - currentMIPS->InvalidateICache(0x0886DC20, 4); - currentMIPS->InvalidateICache(0x0886DD40, 4); + currentMIPS->InvalidateICacheRangeDeferred(0x0886DCC0, 72); + currentMIPS->InvalidateICacheRangeDeferred(0x0886DC20, 4); + currentMIPS->InvalidateICacheRangeDeferred(0x0886DD40, 4); } else if (gameTitle == "ULES00581") { // MTX MotoTrax EU (ported from US cwcheat codes) - currentMIPS->InvalidateICache(0x0886E1D8, 72); - currentMIPS->InvalidateICache(0x0886E138, 4); - currentMIPS->InvalidateICache(0x0886E258, 4); + currentMIPS->InvalidateICacheRangeDeferred(0x0886E1D8, 72); + currentMIPS->InvalidateICacheRangeDeferred(0x0886E138, 4); + currentMIPS->InvalidateICacheRangeDeferred(0x0886E258, 4); } } - if (!cheatEngine || !cheatsEnabled) + if (!cheatEngine || !cheatsEnabled) { return; + } if (g_Config.bReloadCheats) { // Checks if the "reload cheats" button has been pressed. cheatEngine->ParseCheats(); @@ -371,7 +372,7 @@ void CWCheatEngine::InvalidateICache(u32 addr, int size) const { // Round start down and size up to the nearest word. u32 aligned = addr & ~3; int alignedSize = (addr + size - aligned + 3) & ~3; - currentMIPS->InvalidateICache(aligned, alignedSize); + currentMIPS->InvalidateICacheRangeDeferred(aligned, alignedSize); } enum class CheatOp { diff --git a/Core/Debugger/Breakpoints.cpp b/Core/Debugger/Breakpoints.cpp index b2e18f5855..6d2539458f 100644 --- a/Core/Debugger/Breakpoints.cpp +++ b/Core/Debugger/Breakpoints.cpp @@ -759,9 +759,9 @@ void BreakpointManager::Frame() { if (MIPSComp::jit && updateAddr_ != INVALID_ADDRESS) { // In case this is a delay slot, clear the previous instruction too. if (updateAddr_ != 0) - mipsr4k.InvalidateICache(updateAddr_ - 4, 8); + mipsr4k.InvalidateICacheRangeDeferred(updateAddr_ - 4, 8); else - mipsr4k.ClearJitCache(); + mipsr4k.ClearJitCacheDeferred(); } if (anyMemChecks_ && updateAddr_ != INVALID_ADDRESS) diff --git a/Core/Debugger/WebSocket/MemorySubscriber.cpp b/Core/Debugger/WebSocket/MemorySubscriber.cpp index ce56f2eeff..6243ba2594 100644 --- a/Core/Debugger/WebSocket/MemorySubscriber.cpp +++ b/Core/Debugger/WebSocket/MemorySubscriber.cpp @@ -321,8 +321,8 @@ void WebSocketMemoryWriteU8(DebuggerRequest &req) { // from this WebSocket handler thread - see Core_RunOnCPUThread() in Core.h. Core_RunOnCPUThread([&] { AutoDisabledReplacements memLock = LockMemory(true); - currentMIPS->InvalidateICache(addr, 1); Memory::WriteUnchecked_U8(val, addr); + currentMIPS->InvalidateICacheRangeDeferred(addr, 1); Reporting::NotifyDebugger(); JsonWriter &json = req.Respond(); @@ -358,8 +358,8 @@ void WebSocketMemoryWriteU16(DebuggerRequest &req) { // from this WebSocket handler thread - see Core_RunOnCPUThread() in Core.h. Core_RunOnCPUThread([&] { AutoDisabledReplacements memLock = LockMemory(true); - currentMIPS->InvalidateICache(addr, 2); Memory::WriteUnchecked_U16(val, addr); + currentMIPS->InvalidateICacheRangeDeferred(addr, 2); Reporting::NotifyDebugger(); JsonWriter &json = req.Respond(); @@ -395,8 +395,8 @@ void WebSocketMemoryWriteU32(DebuggerRequest &req) { // from this WebSocket handler thread - see Core_RunOnCPUThread() in Core.h. Core_RunOnCPUThread([&] { AutoDisabledReplacements memLock = LockMemory(true); - currentMIPS->InvalidateICache(addr, 4); Memory::WriteUnchecked_U32(val, addr); + currentMIPS->InvalidateICacheRangeDeferred(addr, 4); Reporting::NotifyDebugger(); JsonWriter &json = req.Respond(); @@ -436,9 +436,10 @@ void WebSocketMemoryWrite(DebuggerRequest &req) { // from this WebSocket handler thread - see Core_RunOnCPUThread() in Core.h. Core_RunOnCPUThread([&] { AutoDisabledReplacements memLock = LockMemory(true); - currentMIPS->InvalidateICache(addr, size); - if (size != 0) + currentMIPS->InvalidateICacheRangeDeferred(addr, size); + if (size != 0) { Memory::MemcpyUnchecked(addr, &value[0], size); + } Reporting::NotifyDebugger(); req.Respond(); }); diff --git a/Core/HLE/HLEHelperThread.cpp b/Core/HLE/HLEHelperThread.cpp index 7b78215a8c..939a910c46 100644 --- a/Core/HLE/HLEHelperThread.cpp +++ b/Core/HLE/HLEHelperThread.cpp @@ -26,8 +26,6 @@ #include "Core/HLE/sceKernelMemory.h" #include "Core/MIPS/MIPSCodeUtils.h" -HLEHelperThread::HLEHelperThread() : id_(0), entry_(0) {} - HLEHelperThread::HLEHelperThread(const char *threadName, const u32 instructions[], u32 instrCount, u32 prio, int stacksize) { u32 instrBytes = instrCount * sizeof(u32); u32 totalBytes = instrBytes + sizeof(u32) * 2; @@ -63,7 +61,7 @@ void HLEHelperThread::AllocEntry(u32 size) { entry_ = kernelMemory.Alloc(size, false, "HLEHelper"); _dbg_assert_(Memory::IsValid4AlignedAddress(entry_)); // after AllocEntry. Memory::Memset(entry_, 0, size, "HLEHelperClear"); - currentMIPS->InvalidateICache(entry_, size); + currentMIPS->InvalidateICacheRangeDeferred(entry_, size); } void HLEHelperThread::Create(const char *threadName, u32 prio, int stacksize) { diff --git a/Core/HLE/HLEHelperThread.h b/Core/HLE/HLEHelperThread.h index d932c2ea23..e5d3d2e889 100644 --- a/Core/HLE/HLEHelperThread.h +++ b/Core/HLE/HLEHelperThread.h @@ -24,7 +24,7 @@ enum WaitType : int; class HLEHelperThread { public: // For savestates. - HLEHelperThread(); + HLEHelperThread() = default; HLEHelperThread(const char *threadName, const u32 instructions[], u32 instrCount, u32 prio, int stacksize); HLEHelperThread(const char *threadName, const char *module, const char *func, u32 prio, int stacksize); ~HLEHelperThread(); @@ -47,6 +47,6 @@ private: void AllocEntry(u32 size); void Create(const char *threadName, u32 prio, int stacksize); - SceUID id_; - u32 entry_; + SceUID id_ = 0; + u32 entry_ = 0; }; diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index aadff67a91..2a14406eb3 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -126,7 +126,7 @@ static int Replace_memcpy() { } // Some games use memcpy on executable code. We need to flush emuhack ops. - currentMIPS->InvalidateICache(srcPtr, bytes); + currentMIPS->InvalidateICacheRangeImmediate(srcPtr, bytes); if ((skipGPUReplacements & (int)GPUReplacementSkip::MEMCPY) == 0) { if (Memory::IsVRAMAddress(destPtr) || Memory::IsVRAMAddress(srcPtr)) { skip = gpu->PerformMemoryCopy(destPtr, srcPtr, bytes); @@ -187,7 +187,7 @@ static int Replace_memcpy_jak() { bool sliced = false; static constexpr uint32_t SLICE_SIZE = 32768; - currentMIPS->InvalidateICache(srcPtr, bytes); + currentMIPS->InvalidateICacheRangeImmediate(srcPtr, bytes); if ((skipGPUReplacements & (int)GPUReplacementSkip::MEMCPY) == 0) { if (Memory::IsVRAMAddress(destPtr) || Memory::IsVRAMAddress(srcPtr)) { skip = gpu->PerformMemoryCopy(destPtr, srcPtr, bytes); @@ -258,7 +258,7 @@ static int Replace_memcpy16() { // Some games use memcpy on executable code. We need to flush emuhack ops. if (bytes != 0) - currentMIPS->InvalidateICache(srcPtr, bytes); + currentMIPS->InvalidateICacheRangeImmediate(srcPtr, bytes); if ((skipGPUReplacements & (int)GPUReplacementSkip::MEMCPY) == 0 && bytes != 0) { if (Memory::IsVRAMAddress(destPtr) || Memory::IsVRAMAddress(srcPtr)) { skip = gpu->PerformMemoryCopy(destPtr, srcPtr, bytes); @@ -327,7 +327,7 @@ static int Replace_memmove() { // Some games use memcpy on executable code. We need to flush emuhack ops. if ((skipGPUReplacements & (int)GPUReplacementSkip::MEMMOVE) == 0 && bytes != 0) { - currentMIPS->InvalidateICache(srcPtr, bytes); + currentMIPS->InvalidateICacheRangeImmediate(srcPtr, bytes); if (Memory::IsVRAMAddress(destPtr) || Memory::IsVRAMAddress(srcPtr)) { skip = gpu->PerformMemoryCopy(destPtr, srcPtr, bytes); } diff --git a/Core/HLE/sceDmac.cpp b/Core/HLE/sceDmac.cpp index a6c673c39c..666138208c 100644 --- a/Core/HLE/sceDmac.cpp +++ b/Core/HLE/sceDmac.cpp @@ -29,7 +29,7 @@ #include "GPU/GPUCommon.h" #include "GPU/GPUState.h" -u64 dmacMemcpyDeadline; +static u64 dmacMemcpyDeadline; void __DmacInit() { dmacMemcpyDeadline = 0; @@ -45,21 +45,21 @@ void __DmacDoState(PointerWrap &p) { Do(p, dmacMemcpyDeadline); } -static int __DmacMemcpy(u32 dst, u32 src, u32 size) { +static int __DmacMemcpy(MIPSState *mips, u32 dst, u32 src, u32 size) { bool skip = false; if (Memory::IsVRAMAddress(src) || Memory::IsVRAMAddress(dst)) { // We let the GPU deal with invalid range. skip = gpu->PerformMemoryCopy(dst, src, size); } if (!skip && size != 0) { - currentMIPS->InvalidateICache(src, size); + mips->InvalidateICacheRangeDeferred(src, size); if (Memory::IsValidRange(dst, size) && Memory::IsValidRange(src, size)) { memcpy(Memory::GetPointerWriteUnchecked(dst), Memory::GetPointerUnchecked(src), size); } if (MemBlockInfoDetailed(size)) { NotifyMemInfoCopy(dst, src, size, "DmacMemcpy/"); } - currentMIPS->InvalidateICache(dst, size); + mips->InvalidateICacheRangeDeferred(dst, size); } // This number seems strangely reproducible. @@ -91,7 +91,7 @@ static u32 sceDmacMemcpy(u32 dst, u32 src, u32 size) { // Might matter for overlapping copies. } - int delay = __DmacMemcpy(dst, src, size); + int delay = __DmacMemcpy(currentMIPS, dst, src, size); int result = hleLogDebug(Log::HLE, 0); return delay ? hleDelayResult(result, "dmac-memcpy", delay) : delay; } @@ -111,7 +111,7 @@ static u32 sceDmacTryMemcpy(u32 dst, u32 src, u32 size) { return hleLogDebug(Log::HLE, SCE_KERNEL_ERROR_BUSY, "busy"); } - int delay = __DmacMemcpy(dst, src, size); + int delay = __DmacMemcpy(currentMIPS, dst, src, size); int result = hleLogDebug(Log::HLE, 0); return delay ? hleDelayResult(result, "dmac-memcpy", delay) : delay; } diff --git a/Core/HLE/sceIo.cpp b/Core/HLE/sceIo.cpp index 970949ad3a..eb33c71126 100644 --- a/Core/HLE/sceIo.cpp +++ b/Core/HLE/sceIo.cpp @@ -1041,7 +1041,7 @@ static bool __IoRead(int &result, int id, u32 data_addr, int size, int &us) { u32 validSize = Memory::ClampValidSizeAt(data_addr, size); if (f->npdrm) { result = npdrmRead(f, data, validSize); - currentMIPS->InvalidateICache(data_addr, validSize); + currentMIPS->InvalidateICacheRangeDeferred(data_addr, validSize); return true; } @@ -1067,7 +1067,7 @@ static bool __IoRead(int &result, int id, u32 data_addr, int size, int &us) { } else { result = (int)pspFileSystem.ReadFile(f->handle, data, validSize, us); } - currentMIPS->InvalidateICache(data_addr, validSize); + currentMIPS->InvalidateICacheRangeDeferred(data_addr, validSize); return true; } } else { diff --git a/Core/HLE/sceKernel.cpp b/Core/HLE/sceKernel.cpp index 8b4dd79e70..5968e5fe15 100644 --- a/Core/HLE/sceKernel.cpp +++ b/Core/HLE/sceKernel.cpp @@ -394,21 +394,24 @@ int sceKernelDcacheInvalidateRange(u32 addr, int size) if (size < 0 || (int) addr + size < 0) return hleNoLog(SCE_KERNEL_ERROR_ILLEGAL_ADDR); - if (size > 0) - { - if ((addr % 64) != 0 || (size % 64) != 0) + if (size > 0) { + if ((addr % 64) != 0 || (size % 64) != 0) { return hleNoLog(SCE_KERNEL_ERROR_CACHE_ALIGNMENT); + } - if (addr != 0) + if (addr != 0) { gpu->InvalidateCache(addr, size, GPU_INVALIDATE_HINT); + } } hleEatCycles(190); return hleNoLog(0); } int sceKernelIcacheInvalidateRange(u32 addr, int size) { - if (size != 0) - currentMIPS->InvalidateICache(addr, size); + if (size != 0) { + currentMIPS->InvalidateICacheRangeDeferred(addr, size); + Core_ReenterDispatcher(); + } return hleLogDebug(Log::CPU, 0); } @@ -455,8 +458,7 @@ int sceKernelDcacheWritebackInvalidateRange(u32 addr, int size) return hleNoLog(0); } -int sceKernelDcacheWritebackInvalidateAll() -{ +int sceKernelDcacheWritebackInvalidateAll() { #ifdef LOG_CACHE NOTICE_LOG(Log::CPU,"sceKernelDcacheInvalidateAll()"); #endif @@ -466,23 +468,23 @@ int sceKernelDcacheWritebackInvalidateAll() return hleLogDebug(Log::CPU, 0, "Dcache invalidated"); } -u32 sceKernelIcacheInvalidateAll() -{ +u32 sceKernelIcacheInvalidateAll() { #ifdef LOG_CACHE NOTICE_LOG(Log::CPU, "Icache invalidated - should clear JIT someday"); #endif // Note that this doesn't actually fully invalidate all with such a large range. - currentMIPS->InvalidateICache(0, 0x3FFFFFFF); + currentMIPS->InvalidateICacheRangeDeferred(0, 0x3FFFFFFF); + Core_ReenterDispatcher(); return hleLogDebug(Log::CPU, 0, "Icache invalidated"); } -u32 sceKernelIcacheClearAll() -{ +u32 sceKernelIcacheClearAll() { #ifdef LOG_CACHE NOTICE_LOG(Log::CPU, "Icache cleared - should clear JIT someday"); #endif // Note that this doesn't actually fully invalidate all with such a large range. - currentMIPS->InvalidateICache(0, 0x3FFFFFFF); + currentMIPS->InvalidateICacheRangeDeferred(0, 0x3FFFFFFF); + Core_ReenterDispatcher(); return hleLogDebug(Log::CPU, 0, "Icache cleared"); } diff --git a/Core/HLE/sceKernel.h b/Core/HLE/sceKernel.h index b1d79f9476..1d12a9dee2 100644 --- a/Core/HLE/sceKernel.h +++ b/Core/HLE/sceKernel.h @@ -91,6 +91,8 @@ bool __KernelLoadExec(const char *filename, SceKernelLoadExecParam *param); // For crash reporting. std::string __KernelStateSummary(); +// These HLE functions are declared here to be included in tables that are in other files. + int sceKernelLoadExec(const char *filename, u32 paramPtr); void sceKernelExitGame(); diff --git a/Core/HLE/sceKernelInterrupt.cpp b/Core/HLE/sceKernelInterrupt.cpp index e9f0e3191e..0f5f707c3c 100644 --- a/Core/HLE/sceKernelInterrupt.cpp +++ b/Core/HLE/sceKernelInterrupt.cpp @@ -607,8 +607,9 @@ static u32 sceKernelMemset(u32 addr, u32 fillc, u32 n) { static u32 sceKernelMemcpy(u32 dst, u32 src, u32 size) { // Some games copy from executable code. We need to flush emuhack ops. - if (size != 0) - currentMIPS->InvalidateICache(src, size); + if (size != 0) { + currentMIPS->InvalidateICacheRangeDeferred(src, size); + } bool skip = false; if (Memory::IsVRAMAddress(src) || Memory::IsVRAMAddress(dst)) { diff --git a/Core/HLE/sceKernelModule.cpp b/Core/HLE/sceKernelModule.cpp index da0572445e..cd4a3cace7 100644 --- a/Core/HLE/sceKernelModule.cpp +++ b/Core/HLE/sceKernelModule.cpp @@ -578,7 +578,7 @@ static void WriteVarSymbol(WriteVarSymbolState &state, u32 exportAddress, u32 re // We add 1 in that case so that it ends up the right value. u16 high = (full >> 16) + ((full & 0x8000) ? 1 : 0); Memory::WriteUnchecked_U32((reloc.data & ~0xFFFF) | high, reloc.addr); - currentMIPS->InvalidateICache(reloc.addr, 4); + currentMIPS->InvalidateICacheRangeDeferred(reloc.addr, 4); } state.lastHI16Processed = true; } @@ -593,7 +593,7 @@ static void WriteVarSymbol(WriteVarSymbolState &state, u32 exportAddress, u32 re } Memory::WriteUnchecked_U32(relocData, relocAddress); - currentMIPS->InvalidateICache(relocAddress, 4); + currentMIPS->InvalidateICacheRangeDeferred(relocAddress, 4); } void ImportVarSymbol(WriteVarSymbolState &state, const VarSymbolImport &var) { @@ -692,7 +692,7 @@ void ImportFuncSymbol(const FuncSymbolImport &func, bool reimporting, const char } // TODO: There's some double lookup going on here (we already did the lookup in GetHLEFunc above). WriteHLESyscall(func.moduleName, func.nid, func.stubAddr); - currentMIPS->InvalidateICache(func.stubAddr, 8); + currentMIPS->InvalidateICacheRangeDeferred(func.stubAddr, 8); return; } @@ -710,7 +710,7 @@ void ImportFuncSymbol(const FuncSymbolImport &func, bool reimporting, const char WARN_LOG_REPORT(Log::Loader, "Reimporting: func import %s/%08x changed", func.moduleName, func.nid); } WriteFuncStub(func.stubAddr, it->symAddr); - currentMIPS->InvalidateICache(func.stubAddr, 8); + currentMIPS->InvalidateICacheRangeDeferred(func.stubAddr, 8); return; } } @@ -726,7 +726,7 @@ void ImportFuncSymbol(const FuncSymbolImport &func, bool reimporting, const char if (shouldHLE || !reimporting) { WriteFuncMissingStub(func.stubAddr, func.nid); - currentMIPS->InvalidateICache(func.stubAddr, 8); + currentMIPS->InvalidateICacheRangeDeferred(func.stubAddr, 8); } } @@ -750,7 +750,7 @@ void ExportFuncSymbol(const FuncSymbolExport &func) { if (func.Matches(*it)) { INFO_LOG(Log::Loader, "Resolving function %s/%08x", func.moduleName, func.nid); WriteFuncStub(it->stubAddr, func.symAddr); - currentMIPS->InvalidateICache(it->stubAddr, 8); + currentMIPS->InvalidateICacheRangeDeferred(it->stubAddr, 8); } } } @@ -774,7 +774,7 @@ void UnexportFuncSymbol(const FuncSymbolExport &func) { if (func.Matches(*it)) { INFO_LOG(Log::Loader, "Unresolving function %s/%08x", func.moduleName, func.nid); WriteFuncMissingStub(it->stubAddr, it->nid); - currentMIPS->InvalidateICache(it->stubAddr, 8); + currentMIPS->InvalidateICacheRangeDeferred(it->stubAddr, 8); } } } @@ -826,7 +826,7 @@ void PSPModule::Cleanup() { Memory::Memset(nm.text_addr + nm.text_size, -1, nm.data_size + nm.bss_size, "ModuleClear"); // Let's also invalidate, just to make sure it's cleared out for any future data. - currentMIPS->InvalidateICache(memoryBlockAddr, memoryBlockSize); + currentMIPS->InvalidateICacheRangeDeferred(memoryBlockAddr, memoryBlockSize); } } @@ -1287,7 +1287,7 @@ static PSPModule *__KernelLoadELFFromPtr(const u8 *ptr, size_t elfSize, u32 load module->memoryBlockAddr = reader.GetVaddr(); module->memoryBlockSize = reader.GetTotalSize(); - currentMIPS->InvalidateICache(module->memoryBlockAddr, module->memoryBlockSize); + currentMIPS->InvalidateICacheRangeDeferred(module->memoryBlockAddr, module->memoryBlockSize); SectionID sceModuleInfoSection = reader.GetSectionByName(".rodata.sceModuleInfo"); const PspModuleInfo *modinfo; diff --git a/Core/HW/AsyncIOManager.cpp b/Core/HW/AsyncIOManager.cpp index 51072c100c..75de06ebbf 100644 --- a/Core/HW/AsyncIOManager.cpp +++ b/Core/HW/AsyncIOManager.cpp @@ -67,7 +67,7 @@ bool AsyncIOManager::PopResult(u32 handle, AsyncIOResult &result) { resultsPending_.erase(handle); if (result.invalidateAddr && result.result > 0) { - currentMIPS->InvalidateICache(result.invalidateAddr, (int)result.result); + currentMIPS->InvalidateICacheRangeImmediate(result.invalidateAddr, (int)result.result); } return true; } else { diff --git a/Core/MIPS/Interpreter.cpp b/Core/MIPS/Interpreter.cpp index 2aa75a6f71..d2ad867778 100644 --- a/Core/MIPS/Interpreter.cpp +++ b/Core/MIPS/Interpreter.cpp @@ -75,7 +75,7 @@ static inline void SkipLikely(MIPSState *mips) { } } -int MIPS_SingleStep(MIPSState *mips) { +int MIPS_InterpretSingleStep(MIPSState *mips) { if (!Memory::IsValid4AlignedAddress(mips->pc)) { Core_ExecException(mips->pc, mips-> pc, ExecExceptionType::JUMP); return 0; diff --git a/Core/MIPS/Interpreter.h b/Core/MIPS/Interpreter.h index de1ec80cde..7fb54a45c0 100644 --- a/Core/MIPS/Interpreter.h +++ b/Core/MIPS/Interpreter.h @@ -20,7 +20,7 @@ #include "Common/CommonTypes.h" #include "Core/MIPS/MIPS.h" -int MIPS_SingleStep(MIPSState *mips); +int MIPS_InterpretSingleStep(MIPSState *mips); namespace MIPSInt { diff --git a/Core/MIPS/MIPS.cpp b/Core/MIPS/MIPS.cpp index 8c43e40657..c733131cad 100644 --- a/Core/MIPS/MIPS.cpp +++ b/Core/MIPS/MIPS.cpp @@ -166,6 +166,8 @@ void MIPSState::Shutdown() { MIPSComp::jit = nullptr; delete oldjit; } + pendingInvalidates_.clear(); + invalidateAll_ = false; } void MIPSState::Reset() { @@ -302,8 +304,8 @@ void MIPSState::DoState(PointerWrap &p) { Do(p, fcr31); if (s <= 3) { uint32_t dummy; - Do(p, dummy); // rng.m_w - Do(p, dummy); // rng.m_z + Do(p, dummy); // rng.m_w + Do(p, dummy); // rng.m_z } Do(p, inDelaySlot); @@ -317,9 +319,9 @@ void MIPSState::DoState(PointerWrap &p) { } void MIPSState::SingleStep() { - int cycles = MIPS_SingleStep(this); + ProcessPendingInvalidates(); + int cycles = MIPS_InterpretSingleStep(this); downcount -= cycles; - CoreTiming::Advance(currentMIPS); } // returns 1 if reached ticks limit @@ -331,13 +333,11 @@ int MIPSState::RunLoopUntil(u64 globalTicks) { while (inDelaySlot) { // We must get out of the delay slot before going into jit. // This normally should never take more than one step... - SingleStep(); + int cycles = MIPS_InterpretSingleStep(this); + downcount -= cycles; } - insideJit = true; - if (hasPendingClears) - ProcessPendingClears(); + ProcessPendingInvalidates(); MIPSComp::jit->RunLoopUntil(globalTicks); - insideJit = false; break; case CPUCore::INTERPRETER: @@ -346,36 +346,55 @@ int MIPSState::RunLoopUntil(u64 globalTicks) { return 1; } -// Kept outside MIPSState to avoid header pollution (MIPS.h doesn't even have vector, and is used widely.) -static std::vector> pendingClears; - -void MIPSState::ProcessPendingClears() { - for (auto &p : pendingClears) { - if (p.first == 0 && p.second == 0) - MIPSComp::jit->ClearCache(); - else - MIPSComp::jit->InvalidateCacheAt(p.first, p.second); +void MIPSState::InvalidateICacheRangeImmediate(u32 address, u32 length) { + if (!MIPSComp::jit) { + // Nothing to do. + return; } - pendingClears.clear(); - hasPendingClears = false; -} - -void MIPSState::InvalidateICache(u32 address, int length) { - // Only really applies to jit. - // Note that the backend is responsible for ensuring native code can still be returned to. - if (MIPSComp::jit && length != 0) { + if (length != 0) { MIPSComp::jit->InvalidateCacheAt(address, length); } } -void MIPSState::ClearJitCache() { - if (MIPSComp::jit) { - if (coreState == CORE_RUNNING_CPU || insideJit) { - pendingClears.emplace_back(0, 0); - hasPendingClears = true; - CoreTiming::ForceCheck(this); - } else { - MIPSComp::jit->ClearCache(); +void MIPSState::ProcessPendingInvalidates() { + if (!MIPSComp::jit) { + // Nothing to do. Clear any old state from CPU core switching. + pendingInvalidates_.clear(); + return; + } + + if (invalidateAll_) { + MIPSComp::jit->ClearCache(); + pendingInvalidates_.clear(); + invalidateAll_ = false; + return; + } + + if (!pendingInvalidates_.empty()) { + for (const auto &p : pendingInvalidates_) { + MIPSComp::jit->InvalidateCacheAt(p.addr, p.size); } + pendingInvalidates_.clear(); } } + +void MIPSState::InvalidateICacheRangeDeferred(u32 address, u32 length) { + if (!MIPSComp::jit) { + // Nothing to do. + return; + } + + // TODO: We can try to merge the invalidate with the previous one. + pendingInvalidates_.push_back(PendingCacheOperation{address, length}); + Core_ReenterDispatcher(); +} + +void MIPSState::ClearJitCacheDeferred() { + if (!MIPSComp::jit) { + return; + } + + pendingInvalidates_.clear(); + invalidateAll_ = true; + Core_ReenterDispatcher(); +} diff --git a/Core/MIPS/MIPS.h b/Core/MIPS/MIPS.h index c5ccbdc7b6..9ae7dd87ca 100644 --- a/Core/MIPS/MIPS.h +++ b/Core/MIPS/MIPS.h @@ -20,6 +20,7 @@ #include "ppsspp_config.h" #include +#include #include "Common/CommonTypes.h" #include "Core/Opcode.h" @@ -163,12 +164,13 @@ enum class CPUCore; #endif -enum { - NUM_X86_FPU_TEMPS = 16, +// Only icache invalidations currently, so no type field. +struct PendingCacheOperation { + u32 addr; + u32 size; }; -class MIPSState -{ +class MIPSState { public: MIPSState(); ~MIPSState(); @@ -245,6 +247,9 @@ public: static const u32 FCR0_VALUE = 0x00003351; #if PPSSPP_ARCH(X86) || PPSSPP_ARCH(AMD64) + enum { + NUM_X86_FPU_TEMPS = 16, + }; // FPU TEMP0, etc. are swapped in here if necessary (e.g. on x86.) float tempValues[NUM_X86_FPU_TEMPS]; #endif @@ -260,15 +265,23 @@ public: void SingleStep(); int RunLoopUntil(u64 globalTicks); + // To clear jit caches, etc. - void InvalidateICache(u32 address, int length = 4); - void ClearJitCache(); - void ProcessPendingClears(); + // The immediate functions have the risk of invalidating the currently running block - + // might not behave correctly in all cases. Use Deferred when possible. + void InvalidateICacheRangeImmediate(u32 address, u32 length); + // Actual clearing is deferred until execution begins again. + void InvalidateICacheRangeDeferred(u32 address, u32 length); + void ClearJitCacheDeferred(); + + void ProcessPendingInvalidates(); + +private: // Doesn't need save stating. - volatile bool insideJit = false; - volatile bool hasPendingClears = false; + std::vector pendingInvalidates_; + bool invalidateAll_ = false; }; class MIPSDebugInterface; diff --git a/Core/MIPS/MIPSAsm.cpp b/Core/MIPS/MIPSAsm.cpp index 19b4cc7123..8bdef6ffb4 100644 --- a/Core/MIPS/MIPSAsm.cpp +++ b/Core/MIPS/MIPSAsm.cpp @@ -27,7 +27,7 @@ public: Memory::Memcpy((u32)address, data, (u32)length, "Debugger"); // In case this is a delay slot or combined instruction, clear cache above it too. - mipsr4k.InvalidateICache((u32)(address - 4), (int)length + 4); + mipsr4k.InvalidateICacheRangeDeferred((u32)(address - 4), (int)length + 4); address += length; return true; diff --git a/Core/MIPS/x86/RegCacheFPU.cpp b/Core/MIPS/x86/RegCacheFPU.cpp index 4940bf78c1..567b53f901 100644 --- a/Core/MIPS/x86/RegCacheFPU.cpp +++ b/Core/MIPS/x86/RegCacheFPU.cpp @@ -588,7 +588,7 @@ void FPURegCache::ReleaseSpillLock(int mipsreg) { void FPURegCache::ReleaseSpillLocks() { for (int i = 0; i < NUM_MIPS_FPRS; i++) regs[i].locked = 0; - for (int i = TEMP0; i < TEMP0 + NUM_X86_FPU_TEMPS; ++i) + for (int i = TEMP0; i < TEMP0 + MIPSState::NUM_X86_FPU_TEMPS; ++i) DiscardR(i); } @@ -797,7 +797,7 @@ bool FPURegCache::IsTempX(X64Reg xr) { int FPURegCache::GetTempR() { pendingFlush = true; - for (int r = TEMP0; r < TEMP0 + NUM_X86_FPU_TEMPS; ++r) { + for (int r = TEMP0; r < TEMP0 + MIPSState::NUM_X86_FPU_TEMPS; ++r) { if (!regs[r].away && !regs[r].tempLocked) { regs[r].tempLocked = true; return r; @@ -814,7 +814,7 @@ int FPURegCache::GetTempVS(u8 *v, VectorSize vsz) { // Let's collect regs as we go, but try for n free in a row. int found = 0; - for (int r = TEMP0; r <= TEMP0 + NUM_X86_FPU_TEMPS - n; ++r) { + for (int r = TEMP0; r <= TEMP0 + MIPSState::NUM_X86_FPU_TEMPS - n; ++r) { if (regs[r].away || regs[r].tempLocked) { continue; } diff --git a/Core/MIPS/x86/RegCacheFPU.h b/Core/MIPS/x86/RegCacheFPU.h index 69cc412270..179800357a 100644 --- a/Core/MIPS/x86/RegCacheFPU.h +++ b/Core/MIPS/x86/RegCacheFPU.h @@ -46,7 +46,7 @@ enum { TEMP0 = 32 + 128, - NUM_MIPS_FPRS = 32 + 128 + NUM_X86_FPU_TEMPS, + NUM_MIPS_FPRS = 32 + 128 + MIPSState::NUM_X86_FPU_TEMPS, }; #if PPSSPP_ARCH(AMD64) diff --git a/Core/System.cpp b/Core/System.cpp index 08780ae079..348d951835 100644 --- a/Core/System.cpp +++ b/Core/System.cpp @@ -94,7 +94,7 @@ static std::mutex loadingLock; static std::thread g_loadingThread; bool g_coreCollectDebugStats = false; -static int coreCollectDebugStatsCounter = 0; +static int g_coreCollectDebugStatsCounter = 0; static volatile CPUThreadState cpuThreadState = CPU_THREAD_NOT_RUNNING; @@ -607,10 +607,10 @@ void UpdateLoadedFile(FileLoader *fileLoader) { } void PSP_UpdateDebugStats(bool collectStats) { - bool newState = collectStats || coreCollectDebugStatsCounter > 0; + bool newState = collectStats || g_coreCollectDebugStatsCounter > 0; if (g_coreCollectDebugStats != newState) { g_coreCollectDebugStats = newState; - mipsr4k.ClearJitCache(); + mipsr4k.ClearJitCacheDeferred(); } if (!PSP_CoreParameter().frozen && !Core_IsStepping()) { @@ -621,11 +621,11 @@ void PSP_UpdateDebugStats(bool collectStats) { void PSP_ForceDebugStats(bool enable) { if (enable) { - coreCollectDebugStatsCounter++; + g_coreCollectDebugStatsCounter++; } else { - coreCollectDebugStatsCounter--; + g_coreCollectDebugStatsCounter--; } - _assert_(coreCollectDebugStatsCounter >= 0); + _assert_(g_coreCollectDebugStatsCounter >= 0); } static void InitGPU(std::string *error_string) { diff --git a/UI/EmuScreen.cpp b/UI/EmuScreen.cpp index 5580274d96..618ee0dd72 100644 --- a/UI/EmuScreen.cpp +++ b/UI/EmuScreen.cpp @@ -624,7 +624,7 @@ void EmuScreen::sendMessage(UIMessage message, const char *value) { gpu->DumpNextFrame(); } else if (message == UIMessage::REQUEST_CLEAR_JIT) { if (!bootPending_) { - currentMIPS->ClearJitCache(); + currentMIPS->ClearJitCacheDeferred(); if (PSP_IsInited()) { currentMIPS->UpdateCore((CPUCore)g_Config.iCpuCore); } diff --git a/UI/ImDebugger/ImDisasmView.cpp b/UI/ImDebugger/ImDisasmView.cpp index 91bcfe456a..50645ae762 100644 --- a/UI/ImDebugger/ImDisasmView.cpp +++ b/UI/ImDebugger/ImDisasmView.cpp @@ -787,8 +787,8 @@ void ImDisasmView::PopupMenu(MIPSState *mips, ImControl &control) { Memory::WriteUnchecked_U32(0, addr); } } - if (currentMIPS) { - currentMIPS->InvalidateICache(selectRangeStart_, selectRangeEnd_ - selectRangeStart_); + if (mips) { + mips->InvalidateICacheRangeDeferred(selectRangeStart_, selectRangeEnd_ - selectRangeStart_); } } ImGui::Separator(); diff --git a/Windows/Debugger/CtrlDisAsmView.cpp b/Windows/Debugger/CtrlDisAsmView.cpp index 2fcde81d91..fc0c4872f9 100644 --- a/Windows/Debugger/CtrlDisAsmView.cpp +++ b/Windows/Debugger/CtrlDisAsmView.cpp @@ -954,14 +954,17 @@ void CtrlDisAsmView::NopInstructions(u32 selectRangeStart, u32 selectRangeEnd) { // Route the memory writes to the CPU thread instead of poking at it directly from this GUI // thread - see Core_RunOnCPUThread() in Core.h. Core_RunOnCPUThread([&] { - if (Memory::IsValidRange(selectRangeStart, selectRangeEnd - selectRangeStart)) { - for (u32 addr = selectRangeStart; addr < selectRangeEnd; addr += 4) { - Memory::WriteUnchecked_U32(0, addr); - } + if (!Memory::IsValid4AlignedRange(selectRangeStart, selectRangeEnd - selectRangeStart)) { + ERROR_LOG(Log::Debugger, "NopIntructions: Bad address range %08x->%08x", selectRangeStart, selectRangeEnd); + return; + } + + for (u32 addr = selectRangeStart; addr < selectRangeEnd; addr += 4) { + Memory::WriteUnchecked_U32(0, addr); } if (currentMIPS) { - currentMIPS->InvalidateICache(selectRangeStart, selectRangeEnd - selectRangeStart); + currentMIPS->InvalidateICacheRangeDeferred(selectRangeStart, selectRangeEnd - selectRangeStart); } }); }