diff --git a/Common/Log.h b/Common/Log.h index aa55de1b6d..198c765b6a 100644 --- a/Common/Log.h +++ b/Common/Log.h @@ -98,8 +98,8 @@ struct LogChannel { #endif bool enabled = true; - bool IsEnabled(LogLevel level) const { - if (level > this->level || !this->enabled) + bool IsEnabled(LogLevel logLevel) const { + if (logLevel > level || !enabled) return false; return true; } diff --git a/Common/Math/SIMDHeaders.h b/Common/Math/SIMDHeaders.h index 9dc9a90f83..0619efc0cd 100644 --- a/Common/Math/SIMDHeaders.h +++ b/Common/Math/SIMDHeaders.h @@ -197,17 +197,17 @@ inline __m128i _mm_mullo_epi32_SSE2(const __m128i v0, const __m128i v1) { inline __m128i _mm_max_epu16_SSE2(const __m128i v0, const __m128i v1) { return _mm_xor_si128( _mm_max_epi16( - _mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)), - _mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))), - _mm_set1_epi16((int16_t)0x8000)); + _mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)), + _mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))), + _mm_set1_epi16((int16_t)(uint16_t)0x8000)); } inline __m128i _mm_min_epu16_SSE2(const __m128i v0, const __m128i v1) { return _mm_xor_si128( _mm_min_epi16( - _mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)), - _mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))), - _mm_set1_epi16((int16_t)0x8000)); + _mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)), + _mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))), + _mm_set1_epi16((int16_t)(uint16_t)0x8000)); } // SSE2 replacement for half of a _mm_packus_epi32 but without the saturation. diff --git a/Common/Math/math_util.h b/Common/Math/math_util.h index 89aea117f3..ee68806836 100644 --- a/Common/Math/math_util.h +++ b/Common/Math/math_util.h @@ -34,7 +34,7 @@ inline constexpr uint32_t RoundDownToMultipleOf(uint32_t v, uint32_t multiple) { // TODO: this should just use a bitscan. inline uint32_t log2i(uint32_t val) { - unsigned int ret = -1; + unsigned int ret = (unsigned int)-1; while (val != 0) { val >>= 1; ret++; } diff --git a/Core/HLE/HLE.cpp b/Core/HLE/HLE.cpp index d50a8cf801..a6fe6db123 100644 --- a/Core/HLE/HLE.cpp +++ b/Core/HLE/HLE.cpp @@ -78,7 +78,7 @@ static const HLEFunction *g_stack[MAX_SYSCALL_RECURSION]; u32 g_syscallPC; int g_stackSize; -static int idleOp; +static int g_idleOp; // Split syscall support. NOTE: This needs to be saved in DoState somehow! static int splitSyscallEatCycles = 0; @@ -266,7 +266,7 @@ void HLEInit() { RegisterAllModules(); g_stackSize = 0; delayedResultEvent = CoreTiming::RegisterEvent("HLEDelayedResult", hleDelayResultFinish); - idleOp = GetSyscallOp("FakeSysCalls", NID_IDLE); + g_idleOp = GetSyscallOp("FakeSysCalls", NID_IDLE); } void HLEDoState(PointerWrap &p) { @@ -801,10 +801,10 @@ void hleFinishSyscallAfterGe() { hleFinishSyscall(nullptr); } -static void updateSyscallStats(int modulenum, int funcnum, double total) { +static void UpdateSyscallStats(int modulenum, int funcnum, double total) { const char *name = moduleDB[modulenum].funcTable[funcnum].name; // Ignore this one, especially for msInSyscalls (although that ignores CoreTiming events.) - if (0 == strcmp(name, "_sceKernelIdle")) + if (equals(name, "_sceKernelIdle")) return; if (total > kernelStats.slowestSyscallTime) { @@ -822,7 +822,7 @@ static void updateSyscallStats(int modulenum, int funcnum, double total) { kernelStats.summedSlowestSyscallName = name; } } else { - double newTotal = kernelStats.summedMsInSyscalls[statCall] += total; + const double newTotal = kernelStats.summedMsInSyscalls[statCall] += total; if (newTotal > kernelStats.summedSlowestSyscallTime) { kernelStats.summedSlowestSyscallTime = newTotal; kernelStats.summedSlowestSyscallName = name; @@ -895,66 +895,76 @@ static void CallSyscallWithoutFlags(const HLEFunction *info) { g_stackSize = 0; } -const HLEFunction *GetSyscallFuncPointer(MIPSOpcode op) { +void LogBadSyscallAtPC(u32 pc, bool compilePhase) { + const LogLevel level = compilePhase ? LogLevel::LWARNING : LogLevel::LERROR; + const char *phase = compilePhase ? "compile" : "run"; + std::string importModuleName, importingModuleName; + u32 nid = 0; + if (pc && KernelFindImportByStubAddr(pc, &importModuleName, &nid, &importingModuleName)) { + const char *funcName = GetHLEFuncName(importModuleName, nid); + GENERIC_LOG(Log::HLE, level, "Unknown syscall (%s) at %08x: unresolved import %s/%08x (%s), called from '%s'", phase, pc, importModuleName.c_str(), nid, funcName ? funcName : "(unknown)", importingModuleName.c_str()); + } else { + char buffer[256]; + DescribeAddress(currentDebugMIPS, pc, buffer, sizeof(buffer)); + GENERIC_LOG(Log::HLE, level, "Unknown syscall (%s) at %08x (%s): was unable to determine more information", phase, pc, buffer); + } +} + +const HLEFunction *GetSyscallFunctionData(MIPSOpcode op, u32 pcForDiagnostics) { u32 callno = (op >> 6) & 0xFFFFF; //20 bits int funcnum = callno & 0xFFF; int modulenum = (callno & 0xFF000) >> 12; if (funcnum == 0xfff) { - std::string_view modName = modulenum >= (int)moduleDB.size() ? "(unknown)" : moduleDB[modulenum].name; // This is what a still-unresolved import looks like once written as a syscall opcode - // the original module name/NID aren't recoverable from the opcode itself (see // WriteFuncMissingStub), but the calling address is a stub we may still be tracking. std::string importModuleName, importingModuleName; u32 nid = 0; - if (currentMIPS->pc >= 8 && KernelFindImportByStubAddr(currentMIPS->pc - 8, &importModuleName, &nid, &importingModuleName)) { - const char *funcName = GetHLEFuncName(importModuleName, nid); - ERROR_LOG(Log::HLE, "Unknown syscall: unresolved import %s/%08x (%s), called from '%s'", importModuleName.c_str(), nid, funcName ? funcName : "(unknown)", importingModuleName.c_str()); - } else { - ERROR_LOG(Log::HLE, "Unknown syscall: Module: '%.*s' (module: %d func: %d)", (int)modName.size(), modName.data(), modulenum, funcnum); - } - return NULL; + LogBadSyscallAtPC(pcForDiagnostics, PSP_CoreParameter().cpuCore != CPUCore::INTERPRETER); + return nullptr; } if (modulenum >= (int)moduleDB.size()) { ERROR_LOG(Log::HLE, "Syscall had bad module number %d - probably executing garbage", modulenum); - return NULL; + return nullptr; } if (funcnum >= moduleDB[modulenum].numFunctions) { ERROR_LOG(Log::HLE, "Syscall had bad function number %d in module %d - probably executing garbage", funcnum, modulenum); - return NULL; + return nullptr; } return &moduleDB[modulenum].funcTable[funcnum]; } -void *GetQuickSyscallFunc(MIPSOpcode op) { - if (coreCollectDebugStats) +void *GetQuickSyscallFunc(const HLEFunction *info, MIPSOpcode op) { + if (g_coreCollectDebugStats) return nullptr; - - const HLEFunction *info = GetSyscallFuncPointer(op); if (!info || !info->func) return nullptr; VERBOSE_LOG(Log::HLE, "Compiling syscall to '%s'", info->name); // TODO: Do this with a flag? - if (op == idleOp) + if (op == g_idleOp) { return (void *)info->func; - if (info->flags != 0) + } else if (info->flags != 0) { return (void *)&CallSyscallWithFlags; - return (void *)&CallSyscallWithoutFlags; + } else { + return (void *)&CallSyscallWithoutFlags; + } } void hleSetFlipTime(double t) { hleFlipTime = t; } -void CallSyscall(MIPSOpcode op) { +void CallSyscallWithPC(MIPSOpcode op, u32 pc) { PROFILE_THIS_SCOPE("syscall"); - double start = 0.0; // need to initialize to fix the race condition where coreCollectDebugStats is enabled in the middle of this func. - if (coreCollectDebugStats) { + const bool collectStats = g_coreCollectDebugStats; + double start = 0.0; + if (collectStats) { start = time_now_d(); } - const HLEFunction *info = GetSyscallFuncPointer(op); + const HLEFunction *info = GetSyscallFunctionData(op, pc); if (!info) { // We haven't incremented the stack yet. RETURN(SCE_KERNEL_ERROR_LIBRARY_NOT_YET_LINKED); @@ -962,7 +972,7 @@ void CallSyscall(MIPSOpcode op) { } if (info->func) { - if (op == idleOp) + if (op == g_idleOp) info->func(); else if (info->flags != 0) CallSyscallWithFlags(info); @@ -974,19 +984,27 @@ void CallSyscall(MIPSOpcode op) { ERROR_LOG_REPORT(Log::HLE, "Unimplemented HLE function %s", info->name ? info->name : "(\?\?\?)"); } - if (coreCollectDebugStats) { - u32 callno = (op >> 6) & 0xFFFFF; //20 bits - int funcnum = callno & 0xFFF; - int modulenum = (callno & 0xFF000) >> 12; + if (collectStats) { + const u32 callno = (op >> 6) & 0xFFFFF; // 20 bits + const int funcnum = callno & 0xFFF; // 12 bits + const int modulenum = (callno & 0xFF000) >> 12; double total = time_now_d() - start; if (total >= hleFlipTime) total -= hleFlipTime; _dbg_assert_msg_(total >= 0.0, "Time spent in syscall became negative"); hleFlipTime = 0.0; - updateSyscallStats(modulenum, funcnum, total); + UpdateSyscallStats(modulenum, funcnum, total); } } +void CallSyscall(MIPSOpcode op) { + CallSyscallWithPC(op, 0); +} + +void CallSyscallUnresolvedAtPC(u32 pc) { + LogBadSyscallAtPC(pc, false); +} + void hlePushFuncDesc(std::string_view module, std::string_view funcName) { const HLEModule *mod = GetHLEModuleByName(module); _dbg_assert_(mod != nullptr); diff --git a/Core/HLE/HLE.h b/Core/HLE/HLE.h index a157cf6f80..a0c39650b3 100644 --- a/Core/HLE/HLE.h +++ b/Core/HLE/HLE.h @@ -184,14 +184,16 @@ size_t HLEFormatLogArgs(const MIPSState *mips, char *message, size_t sz, const c u32 GetSyscallOp(std::string_view module, u32 nib); bool WriteHLESyscall(std::string_view module, u32 nib, u32 address); void CallSyscall(MIPSOpcode op); +void CallSyscallWithPC(MIPSOpcode op, u32 pc); // better diagnostics +void CallSyscallUnresolvedAtPC(u32 pc); void WriteFuncStub(u32 stubAddr, u32 symAddr); void WriteFuncMissingStub(u32 stubAddr, u32 nid); void HLEReturnFromMipsCall(); -const HLEFunction *GetSyscallFuncPointer(MIPSOpcode op); -// For jit, takes arg: const HLEFunction * -void *GetQuickSyscallFunc(MIPSOpcode op); +const HLEFunction *GetSyscallFunctionData(MIPSOpcode op, u32 pcForDiagnostics); +// For jit, the returned function takes the arg: const HLEFunction * +void *GetQuickSyscallFunc(const HLEFunction *info, MIPSOpcode op); void hleDoLogInternal(Log t, LogLevel level, u64 res, const char *file, int line, const char *reportTag, const char *reason, const char *formatted_reason); diff --git a/Core/HLE/sceDisplay.cpp b/Core/HLE/sceDisplay.cpp index 20a436358e..decd6a36aa 100644 --- a/Core/HLE/sceDisplay.cpp +++ b/Core/HLE/sceDisplay.cpp @@ -507,7 +507,7 @@ static void DoFrameIdleTiming() { #endif } - if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) { + if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) { DisplayNotifySleep(time_now_d() - before); } } @@ -720,7 +720,7 @@ void __DisplayFlip(int cyclesLate) { CoreTiming::ScheduleEvent(0 - cyclesLate, afterFlipEvent, 0); numVBlanksSinceFlip = 0; - if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) { + if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) { // Track how long we sleep (whether vsync or sleep_ms.) DisplayNotifySleep(time_now_d() - frameSleepStart, frameSleepPos); } @@ -786,7 +786,7 @@ void hleLagSync(u64 userdata, int cyclesLate) { const int over = (int)((now - goal) * 1000000); ScheduleLagSync(over - emuOver); - if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) { + if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) { DisplayNotifySleep(now - before); } } @@ -846,9 +846,9 @@ void __DisplaySetFramebuf(u32 topaddr, int linesize, int pixelFormat, int sync) // Doing it in non-buffered though creates problems (black screen) on occasion though // so let's not. if (!flippedThisFrame && !g_Config.bSkipBufferEffects) { - double before_flip = time_now_d(); + const double before_flip = time_now_d(); __DisplayFlip(0); - double after_flip = time_now_d(); + const double after_flip = time_now_d(); // Ignore for debug stats. hleSetFlipTime(after_flip - before_flip); } diff --git a/Core/HLE/sceKernelModule.cpp b/Core/HLE/sceKernelModule.cpp index a550bdcb12..da0572445e 100644 --- a/Core/HLE/sceKernelModule.cpp +++ b/Core/HLE/sceKernelModule.cpp @@ -780,7 +780,7 @@ void UnexportFuncSymbol(const FuncSymbolExport &func) { } } -// Used to add detail to the "Unknown syscall" log in HLE.cpp's GetSyscallFuncPointer - a call +// Used to add detail to the "Unknown syscall" log in HLE.cpp's GetSyscallFunctionData - a call // through a still-unresolved import ends up as a generic "invalid syscall" opcode that no // longer carries the original module name/NID, but the (fixed, unique) address of the syscall // instruction itself does - it's exactly the stubAddr every pending FuncSymbolImport recorded @@ -793,7 +793,7 @@ bool KernelFindImportByStubAddr(u32 stubAddr, std::string *importModuleName, u32 continue; } for (const auto &func : module->importedFuncs) { - if (func.stubAddr == stubAddr) { + if (Memory::AddressesEqualAfterMask(func.stubAddr, stubAddr)) { *importModuleName = func.moduleName; *nid = func.nid; *importingModuleName = module->GetName(); diff --git a/Core/HW/Display.cpp b/Core/HW/Display.cpp index 9c63e8ee98..6f78bebeeb 100644 --- a/Core/HW/Display.cpp +++ b/Core/HW/Display.cpp @@ -95,7 +95,7 @@ static void CalculateFPS() { } } - if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) { + if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) { frameTimeHistory[frameTimeHistoryPos++] = (float)(now - lastFrameTimeHistory); lastFrameTimeHistory = now; frameTimeHistoryPos = frameTimeHistoryPos % frameTimeHistorySize; diff --git a/Core/MIPS/ARM/ArmCompBranch.cpp b/Core/MIPS/ARM/ArmCompBranch.cpp index 35fb62d541..f1634164d4 100644 --- a/Core/MIPS/ARM/ArmCompBranch.cpp +++ b/Core/MIPS/ARM/ArmCompBranch.cpp @@ -620,17 +620,20 @@ void ArmJit::Comp_Syscall(MIPSOpcode op) QuickCallFunction(R1, (void *)&CallSyscall); #else // Skip the CallSyscall where possible. - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) - { - gpr.SetRegImm(R0, (u32)(intptr_t)GetSyscallFuncPointer(op)); - // Already flushed, so R1 is safe. - QuickCallFunction(R1, quickFunc); - } - else - { - gpr.SetRegImm(R0, op.encoding); - QuickCallFunction(R1, (void *)&CallSyscall); + const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + gpr.SetRegImm(R0, (uintptr_t)func); + // Already flushed, so R1 is safe. + QuickCallFunction(R1, quickFunc); + } else { + gpr.SetRegImm(R0, op.encoding); + QuickCallFunction(R1, (void *)&CallSyscall); + } + } else { + gpr.SetRegImm(R0, js.compilerPC); + QuickCallFunction(R1, (void *)&CallSyscallUnresolvedAtPC); } #endif ApplyRoundingMode(); diff --git a/Core/MIPS/ARM64/Arm64CompBranch.cpp b/Core/MIPS/ARM64/Arm64CompBranch.cpp index 7cd137a2dd..ccdc4913d8 100644 --- a/Core/MIPS/ARM64/Arm64CompBranch.cpp +++ b/Core/MIPS/ARM64/Arm64CompBranch.cpp @@ -593,7 +593,6 @@ void Arm64Jit::Comp_JumpReg(MIPSOpcode op) WriteExitDestInR(destReg); js.compiling = false; } - void Arm64Jit::Comp_Syscall(MIPSOpcode op) { @@ -636,14 +635,20 @@ void Arm64Jit::Comp_Syscall(MIPSOpcode op) QuickCallFunction(X1, (void *)&CallSyscall); #else // Skip the CallSyscall where possible. - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) { - MOVI2R(X0, (uintptr_t)GetSyscallFuncPointer(op)); - // Already flushed, so X1 is safe. - QuickCallFunction(X1, quickFunc); + const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + MOVI2R(X0, (uintptr_t)func); + // Already flushed, so X1 is safe. + QuickCallFunction(X1, quickFunc); + } else { + MOVI2R(W0, op.encoding); + QuickCallFunction(X1, (void *)&CallSyscall); + } } else { - MOVI2R(W0, op.encoding); - QuickCallFunction(X1, (void *)&CallSyscall); + MOVI2R(W0, js.compilerPC); + QuickCallFunction(X1, (void *)&CallSyscallUnresolvedAtPC); } #endif LoadStaticRegisters(); diff --git a/Core/MIPS/ARM64/Arm64IRCompSystem.cpp b/Core/MIPS/ARM64/Arm64IRCompSystem.cpp index 54a759f1bd..b136408c35 100644 --- a/Core/MIPS/ARM64/Arm64IRCompSystem.cpp +++ b/Core/MIPS/ARM64/Arm64IRCompSystem.cpp @@ -219,13 +219,20 @@ void Arm64JitBackend::CompIR_System(IRInst inst) { // Skip the CallSyscall where possible. { MIPSOpcode op(inst.constant); - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) { - MOVP2R(X0, GetSyscallFuncPointer(op)); - QuickCallFunction(SCRATCH2_64, (const u8 *)quickFunc); + const HLEFunction *func = GetSyscallFunctionData(op, 0); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + MOVP2R(X0, func); + QuickCallFunction(SCRATCH2_64, (const u8 *)quickFunc); + } else { + MOVI2R(W0, inst.constant); + QuickCallFunction(SCRATCH2_64, &CallSyscall); + } } else { - MOVI2R(W0, inst.constant); - QuickCallFunction(SCRATCH2_64, &CallSyscall); + // Shouldn't get here. + MOVI2R(W0, 0); + QuickCallFunction(SCRATCH2_64, &CallSyscallUnresolvedAtPC); } } #endif @@ -235,6 +242,16 @@ void Arm64JitBackend::CompIR_System(IRInst inst) { // This is always followed by an ExitToPC, where we check coreState. break; + case IROp::SyscallUnresolved: + FlushAll(); + SaveStaticRegisters(); + WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL); + MOVI2R(W0, inst.constant); + QuickCallFunction(SCRATCH2_64, &CallSyscallUnresolvedAtPC); + WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT); + LoadStaticRegisters(); + break; + case IROp::CallReplacement: FlushAll(); SaveStaticRegisters(); diff --git a/Core/MIPS/ARM64/Arm64IRJit.cpp b/Core/MIPS/ARM64/Arm64IRJit.cpp index 2a5d38d5be..6f81f4fdc5 100644 --- a/Core/MIPS/ARM64/Arm64IRJit.cpp +++ b/Core/MIPS/ARM64/Arm64IRJit.cpp @@ -65,6 +65,7 @@ static void NoBlockExits() { _assert_msg_(false, "Never exited block, invalid IR?"); } +// TODO: Much of this function should be merged with the same function for the other backends. bool Arm64JitBackend::CompileBlock(IRBlockCache *irBlockCache, int block_num) { if (GetSpaceLeft() < 0x800) return false; diff --git a/Core/MIPS/IR/IRAnalysis.cpp b/Core/MIPS/IR/IRAnalysis.cpp index 8dfc6aab11..8ce6388de8 100644 --- a/Core/MIPS/IR/IRAnalysis.cpp +++ b/Core/MIPS/IR/IRAnalysis.cpp @@ -77,7 +77,7 @@ static int IRReadsFromList(const IRInstMeta &inst, IRReg regs[4], char type) { if ((inst.m.flags & (IRFLAG_SRC3 | IRFLAG_SRC3DST)) != 0 && inst.m.types[0] == type) regs[c++] = inst.src3; - if (inst.op == IROp::Interpret || inst.op == IROp::CallReplacement || inst.op == IROp::Syscall || inst.op == IROp::Break) + if (inst.op == IROp::Interpret || inst.op == IROp::CallReplacement || inst.op == IROp::Syscall || inst.op == IROp::SyscallUnresolved ||inst.op == IROp::Break) return -1; if (inst.op == IROp::Breakpoint || inst.op == IROp::MemoryCheck) return -1; diff --git a/Core/MIPS/IR/IRCompALU.cpp b/Core/MIPS/IR/IRCompALU.cpp index d3813bacad..bcceaf4a33 100644 --- a/Core/MIPS/IR/IRCompALU.cpp +++ b/Core/MIPS/IR/IRCompALU.cpp @@ -61,19 +61,19 @@ void IRFrontend::Comp_IType(MIPSOpcode op) { switch (op >> 26) { case 8: // same as addiu? case 9: // R(rt) = R(rs) + simm; break; //addiu - ir.Write(IROp::AddConst, rt, rs, ir.AddConstant(simm)); + ir.Write(IROp::AddConst, rt, rs, 0, (u32)simm); break; - case 12: ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(uimm)); break; - case 13: ir.Write(IROp::OrConst, rt, rs, ir.AddConstant(uimm)); break; - case 14: ir.Write(IROp::XorConst, rt, rs, ir.AddConstant(uimm)); break; + case 12: ir.Write(IROp::AndConst, rt, rs, 0, uimm); break; + case 13: ir.Write(IROp::OrConst, rt, rs, 0, uimm); break; + case 14: ir.Write(IROp::XorConst, rt, rs, 0, uimm); break; case 10: // R(rt) = (s32)R(rs) < simm; break; //slti - ir.Write(IROp::SltConst, rt, rs, ir.AddConstant(simm)); + ir.Write(IROp::SltConst, rt, rs, 0, (u32)simm); break; case 11: // R(rt) = R(rs) < suimm; break; //sltiu - ir.Write(IROp::SltUConst, rt, rs, ir.AddConstant(suimm)); + ir.Write(IROp::SltUConst, rt, rs, 0, suimm); break; case 15: // R(rt) = uimm << 16; //lui @@ -197,7 +197,7 @@ void IRFrontend::CompShiftVar(MIPSOpcode op, IROp shiftOp) { // The interpreter already masks where needed, don't need to generate extra ops. ir.Write(shiftOp, rd, rt, rs); } else { - ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(31)); + ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, (u32)31); ir.Write(shiftOp, rd, rt, IRTEMP_0); } } @@ -244,9 +244,9 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) { case 0x0: // ext if (pos != 0) { ir.Write(IROp::ShrImm, rt, rs, pos); - ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(mask)); + ir.Write(IROp::AndConst, rt, rt, 0, mask); } else { - ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(mask)); + ir.Write(IROp::AndConst, rt, rs, 0, mask); } break; @@ -259,7 +259,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) { if (size != 32) { // Need to use the sourcemask. - ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(sourcemask)); + ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, sourcemask); if (pos != 0) { ir.Write(IROp::ShlImm, IRTEMP_0, IRTEMP_0, pos); } @@ -271,7 +271,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) { ir.Write(IROp::Mov, IRTEMP_0, rs); } } - ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(destmask)); + ir.Write(IROp::AndConst, rt, rt, 0, destmask); ir.Write(IROp::Or, rt, rt, IRTEMP_0); } break; diff --git a/Core/MIPS/IR/IRCompBranch.cpp b/Core/MIPS/IR/IRCompBranch.cpp index 3b7ff163a5..ec686f7d5b 100644 --- a/Core/MIPS/IR/IRCompBranch.cpp +++ b/Core/MIPS/IR/IRCompBranch.cpp @@ -96,11 +96,11 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) { CompileDelaySlot(); int dcAmount = js.downcountAmount; - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; FlushAll(); - ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs, rhs); + ir.Write(ComparisonToExit(cc), 0, lhs, rhs, ResolveNotTakenTarget(branchInfo)); // This makes the block "impure" :( if (likely && !branchInfo.delaySlotIsBranch) CompileDelaySlot(); @@ -114,7 +114,7 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) { } FlushAll(); - ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr)); + ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr); // Account for the delay slot. js.compilerPC += 4; @@ -147,11 +147,11 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink, CompileDelaySlot(); int dcAmount = js.downcountAmount; - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; FlushAll(); - ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs); + ir.Write(ComparisonToExit(cc), 0, lhs, 0, ResolveNotTakenTarget(branchInfo)); if (likely && !branchInfo.delaySlotIsBranch) CompileDelaySlot(); if (branchInfo.delaySlotIsBranch) { @@ -165,7 +165,7 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink, // Taken FlushAll(); - ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr)); + ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr); // Account for the delay slot. js.compilerPC += 4; @@ -225,12 +225,12 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) { CompileDelaySlot(); int dcAmount = js.downcountAmount; - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; FlushAll(); // Not taken - ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0); + ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo)); // Taken if (likely && !branchInfo.delaySlotIsBranch) CompileDelaySlot(); @@ -244,7 +244,7 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) { } FlushAll(); - ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr)); + ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr); // Account for the delay slot. js.compilerPC += 4; @@ -284,14 +284,14 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) { CompileDelaySlot(); int dcAmount = js.downcountAmount; - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; int imm3 = (op >> 18) & 7; - ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, ir.AddConstant(1 << imm3)); + ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, 0, 1 << imm3); FlushAll(); - ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0); + ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo)); if (likely && !branchInfo.delaySlotIsBranch) CompileDelaySlot(); @@ -306,7 +306,7 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) { // Taken FlushAll(); - ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr)); + ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr); // Account for the delay slot. js.compilerPC += 4; @@ -355,11 +355,11 @@ void IRFrontend::Comp_Jump(MIPSOpcode op) { } int dcAmount = js.downcountAmount; - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; FlushAll(); - ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr)); + ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr); // Account for the delay slot. js.compilerPC += 4; @@ -420,7 +420,7 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) { } int dcAmount = js.downcountAmount; - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; ir.Write(IROp::ExitToReg, 0, destReg, 0); @@ -433,18 +433,25 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) { void IRFrontend::Comp_Syscall(MIPSOpcode op) { // Note: If we're in a delay slot, this is off by one compared to the interpreter. int dcAmount = js.downcountAmount + (js.inDelaySlot ? -1 : 0); - ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, dcAmount); js.downcountAmount = 0; // If not in a delay slot, we need to update PC. + // However we also need the PC for some diagnostics so let's just always do it. + // Not exactly a bottleneck. if (!js.inDelaySlot) { - ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC() + 4)); + ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC() + 4); } FlushAll(); RestoreRoundingMode(); - ir.Write(IROp::Syscall, 0, ir.AddConstant(op.encoding)); + const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC); + if (func) { + ir.Write(IROp::Syscall, 0, 0, 0, op.encoding); + } else { + ir.Write(IROp::SyscallUnresolved, 0, 0, 0, js.compilerPC); + } ApplyRoundingMode(); ir.Write(IROp::ExitToPC); @@ -452,7 +459,7 @@ void IRFrontend::Comp_Syscall(MIPSOpcode op) { } void IRFrontend::Comp_Break(MIPSOpcode op) { - ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC())); + ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC()); ir.Write(IROp::Break); js.compiling = false; } diff --git a/Core/MIPS/IR/IRCompFPU.cpp b/Core/MIPS/IR/IRCompFPU.cpp index cedf708745..b0743b3ddd 100644 --- a/Core/MIPS/IR/IRCompFPU.cpp +++ b/Core/MIPS/IR/IRCompFPU.cpp @@ -83,11 +83,11 @@ void IRFrontend::Comp_FPULS(MIPSOpcode op) { switch (op >> 26) { case 49: // lwc1 - ir.Write(IROp::LoadFloat, ft, rs, ir.AddConstant(offset)); + ir.Write(IROp::LoadFloat, ft, rs, 0, offset); break; case 57: // swc1 - ir.Write(IROp::StoreFloat, ft, rs, ir.AddConstant(offset)); + ir.Write(IROp::StoreFloat, ft, rs, 0, offset); break; default: @@ -207,10 +207,10 @@ void IRFrontend::Comp_mxc1(MIPSOpcode op) { // This needs to insert fpcond. ir.Write(IROp::FpCtrlToReg, rt); } else if (fs == 0) { - ir.Write(IROp::SetConst, rt, ir.AddConstant(MIPSState::FCR0_VALUE)); + ir.WriteSetConstant(rt, MIPSState::FCR0_VALUE); } else { // Unsupported regs are always 0. - ir.Write(IROp::SetConst, rt, ir.AddConstant(0)); + ir.WriteSetConstant(rt, 0); } return; diff --git a/Core/MIPS/IR/IRCompLoadStore.cpp b/Core/MIPS/IR/IRCompLoadStore.cpp index 1a4aac5069..49b0e740e2 100644 --- a/Core/MIPS/IR/IRCompLoadStore.cpp +++ b/Core/MIPS/IR/IRCompLoadStore.cpp @@ -61,42 +61,42 @@ namespace MIPSComp { switch (o) { // Load case 35: - ir.Write(IROp::Load32, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load32, rt, rs, 0, offset); break; case 37: - ir.Write(IROp::Load16, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load16, rt, rs, 0, offset); break; case 33: - ir.Write(IROp::Load16Ext, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load16Ext, rt, rs, 0, offset); break; case 36: - ir.Write(IROp::Load8, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load8, rt, rs, 0, offset); break; case 32: - ir.Write(IROp::Load8Ext, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load8Ext, rt, rs, 0, offset); break; // Store case 43: - ir.Write(IROp::Store32, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Store32, rt, rs, 0, offset); break; case 41: - ir.Write(IROp::Store16, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Store16, rt, rs, 0, offset); break; case 40: - ir.Write(IROp::Store8, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Store8, rt, rs, 0, offset); break; case 34: //lwl - ir.Write(IROp::Load32Left, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load32Left, rt, rs, 0, offset); break; case 38: //lwr - ir.Write(IROp::Load32Right, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load32Right, rt, rs, 0, offset); break; case 42: //swl - ir.Write(IROp::Store32Left, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Store32Left, rt, rs, 0, offset); break; case 46: //swr - ir.Write(IROp::Store32Right, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Store32Right, rt, rs, 0, offset); break; default: @@ -117,11 +117,11 @@ namespace MIPSComp { switch (op >> 26) { case 48: // ll - ir.Write(IROp::Load32Linked, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Load32Linked, rt, rs, 0, offset); break; case 56: // sc - ir.Write(IROp::Store32Conditional, rt, rs, ir.AddConstant(offset)); + ir.Write(IROp::Store32Conditional, rt, rs, 0, offset); break; default: diff --git a/Core/MIPS/IR/IRCompVFPU.cpp b/Core/MIPS/IR/IRCompVFPU.cpp index 5507d6f041..e552b91987 100644 --- a/Core/MIPS/IR/IRCompVFPU.cpp +++ b/Core/MIPS/IR/IRCompVFPU.cpp @@ -256,11 +256,11 @@ namespace MIPSComp { } // Nope, it has something else going on. - zeroedLanes = -1; + zeroedLanes = (u32)-1; break; } - if (zeroedLanes != -1) { + if (zeroedLanes != (u32)-1) { InitRegs(vregs, tempReg); ir.Write(IROp::Vec4Init, vregs[0], (int)Vec4Init::AllZERO); ir.Write(IROp::Vec4Blend, vregs[0], origV[0], vregs[0], zeroedLanes); @@ -284,8 +284,9 @@ namespace MIPSComp { if (!constants) { if (regnum >= n) { // Depends on the op, but often zero. - ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(0.0f)); + ir.WriteSetConstantFloat(vregs[i], 0.0f); } else if (abs) { + // Could have a FNAbs op, but probably not worth it. ir.Write(IROp::FAbs, vregs[i], origV[regnum]); if (negate) ir.Write(IROp::FNeg, vregs[i], vregs[i]); @@ -297,9 +298,9 @@ namespace MIPSComp { } } else { if (negate) { - ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(-constantArray[regnum + (abs << 2)])); + ir.WriteSetConstantFloat(vregs[i], -constantArray[regnum + (abs << 2)]); } else { - ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(constantArray[regnum + (abs << 2)])); + ir.WriteSetConstantFloat(vregs[i], constantArray[regnum + (abs << 2)]); } } } @@ -401,11 +402,11 @@ namespace MIPSComp { switch (op >> 26) { case 50: //lv.s - ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset)); + ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, 0, offset); break; case 58: //sv.s - ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset)); + ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, 0, offset); break; default: @@ -461,29 +462,29 @@ namespace MIPSComp { switch (optype) { case LSVType::LVQ: if (IsVec4(V_Quad, vregs)) { - ir.Write(IROp::LoadVec4, vregs[0], rs, ir.AddConstant(imm)); + ir.Write(IROp::LoadVec4, vregs[0], rs, 0, imm); } else { // Let's not even bother with "vertical" loads for now. if (!g_Config.bFastMemory) ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 0, (u32)imm); - ir.Write(IROp::LoadFloat, vregs[0], rs, ir.AddConstant(imm)); - ir.Write(IROp::LoadFloat, vregs[1], rs, ir.AddConstant(imm + 4)); - ir.Write(IROp::LoadFloat, vregs[2], rs, ir.AddConstant(imm + 8)); - ir.Write(IROp::LoadFloat, vregs[3], rs, ir.AddConstant(imm + 12)); + ir.Write(IROp::LoadFloat, vregs[0], rs, 0, imm); + ir.Write(IROp::LoadFloat, vregs[1], rs, 0, imm + 4); + ir.Write(IROp::LoadFloat, vregs[2], rs, 0, imm + 8); + ir.Write(IROp::LoadFloat, vregs[3], rs, 0, imm + 12); } break; case LSVType::SVQ: if (IsVec4(V_Quad, vregs)) { - ir.Write(IROp::StoreVec4, vregs[0], rs, ir.AddConstant(imm)); + ir.Write(IROp::StoreVec4, vregs[0], rs, 0, imm); } else { // Let's not even bother with "vertical" stores for now. if (!g_Config.bFastMemory) ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 1, (u32)imm); - ir.Write(IROp::StoreFloat, vregs[0], rs, ir.AddConstant(imm)); - ir.Write(IROp::StoreFloat, vregs[1], rs, ir.AddConstant(imm + 4)); - ir.Write(IROp::StoreFloat, vregs[2], rs, ir.AddConstant(imm + 8)); - ir.Write(IROp::StoreFloat, vregs[3], rs, ir.AddConstant(imm + 12)); + ir.Write(IROp::StoreFloat, vregs[0], rs, 0, imm); + ir.Write(IROp::StoreFloat, vregs[1], rs, 0, imm + 4); + ir.Write(IROp::StoreFloat, vregs[2], rs, 0, imm + 8); + ir.Write(IROp::StoreFloat, vregs[3], rs, 0, imm + 12); } break; @@ -521,7 +522,7 @@ namespace MIPSComp { ir.Write(IROp::Vec4Init, dregs[0], (int)(type == 6 ? Vec4Init::AllZERO : Vec4Init::AllONE)); } else { for (int i = 0; i < n; i++) { - ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(type == 6 ? 0.0f : 1.0f)); + ir.WriteSetConstantFloat(dregs[i], type == 6 ? 0.0f : 1.0f); } } ApplyPrefixD(dregs, sz, vd); @@ -549,14 +550,14 @@ namespace MIPSComp { } else { switch (sz) { case V_Pair: - ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 1) == 0 ? 1.0f : 0.0f)); - ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 1) == 1 ? 1.0f : 0.0f)); + ir.WriteSetConstantFloat(dregs[0], (vd & 1) == 0 ? 1.0f : 0.0f); + ir.WriteSetConstantFloat(dregs[1], (vd & 1) == 1 ? 1.0f : 0.0f); break; case V_Quad: - ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 3) == 0 ? 1.0f : 0.0f)); - ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 3) == 1 ? 1.0f : 0.0f)); - ir.Write(IROp::SetConstF, dregs[2], ir.AddConstantFloat((vd & 3) == 2 ? 1.0f : 0.0f)); - ir.Write(IROp::SetConstF, dregs[3], ir.AddConstantFloat((vd & 3) == 3 ? 1.0f : 0.0f)); + ir.WriteSetConstantFloat(dregs[0], (vd & 3) == 0 ? 1.0f : 0.0f); + ir.WriteSetConstantFloat(dregs[1], (vd & 3) == 1 ? 1.0f : 0.0f); + ir.WriteSetConstantFloat(dregs[2], (vd & 3) == 2 ? 1.0f : 0.0f); + ir.WriteSetConstantFloat(dregs[3], (vd & 3) == 3 ? 1.0f : 0.0f); break; default: INVALIDOP; @@ -594,19 +595,19 @@ namespace MIPSComp { switch ((op >> 16) & 0xF) { case 3: // vmidt if (x == 0 && y == 0) - ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f)); + ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f); else if (x == y) ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]); else - ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f)); + ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f); break; case 6: // vmzero // Likely to be fast. - ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f)); + ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f); break; case 7: // vmone if (x == 0 && y == 0) - ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f)); + ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f); else ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]); break; @@ -707,7 +708,7 @@ namespace MIPSComp { GetVectorRegsPrefixD(dregs, V_Single, _VD); // We have to start at +0.000 in case any values are -0.000. - ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(0.0f)); + ir.WriteSetConstantFloat(IRVTEMP_0, 0.0f); for (int i = 0; i < n; ++i) { ir.Write(IROp::FAdd, IRVTEMP_0, IRVTEMP_0, sregs[i]); } @@ -717,7 +718,7 @@ namespace MIPSComp { ir.Write(IROp::FMov, dregs[0], IRVTEMP_0); break; case 7: // vavg - ir.Write(IROp::SetConstF, IRVTEMP_0 + 1, ir.AddConstantFloat(vavg_table[n - 1])); + ir.WriteSetConstantFloat(IRVTEMP_0 + 1, vavg_table[n - 1]); ir.Write(IROp::FMul, dregs[0], IRVTEMP_0, IRVTEMP_0 + 1); break; } @@ -939,7 +940,7 @@ namespace MIPSComp { case VecDo3Op::VSGE: // vsge ir.Write(IROp::FCmp, (int)IRFpCompareMode::LessUnordered, sregs[i], tregs[i]); ir.Write(IROp::FpCondToReg, IRTEMP_1); - ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, ir.AddConstant(1)); + ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, 0, 1); ir.Write(IROp::FMovFromGPR, tempregs[i], IRTEMP_1); ir.Write(IROp::FCvtSW, tempregs[i], tempregs[i]); break; @@ -1266,7 +1267,7 @@ namespace MIPSComp { u32 mask; if (GetVFPUCtrlMask(imm - 128, &mask)) { if (mask != 0xFFFFFFFF) { - ir.Write(IROp::AndConst, IRTEMP_0, rt, ir.AddConstant(mask)); + ir.Write(IROp::AndConst, IRTEMP_0, rt, 0, mask); ir.Write(IROp::SetCtrlVFPUReg, imm - 128, IRTEMP_0); } else { ir.Write(IROp::SetCtrlVFPUReg, imm - 128, rt); @@ -1322,7 +1323,7 @@ namespace MIPSComp { if (GetVFPUCtrlMask(imm, &mask)) { if (mask != 0xFFFFFFFF) { ir.Write(IROp::FMovToGPR, IRTEMP_0, vfpuBase + voffset[imm]); - ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, ir.AddConstant(mask)); + ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, 0, mask); ir.Write(IROp::SetCtrlVFPUReg, imm, IRTEMP_0); } else { ir.Write(IROp::SetCtrlVFPUFReg, imm, vfpuBase + voffset[vs]); @@ -2146,7 +2147,7 @@ namespace MIPSComp { s32 imm = SignExtend16ToS32(op); u8 dreg; GetVectorRegsPrefixD(&dreg, V_Single, _VT); - ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat((float)imm)); + ir.WriteSetConstantFloat(dreg, (float)imm); ApplyPrefixD(&dreg, V_Single, _VT); } @@ -2164,7 +2165,7 @@ namespace MIPSComp { u8 dreg; GetVectorRegsPrefixD(&dreg, V_Single, _VT); - ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat(fval.f)); + ir.WriteSetConstantFloat(dreg, fval.f); ApplyPrefixD(&dreg, V_Single, _VT); } @@ -2186,17 +2187,17 @@ namespace MIPSComp { GetVectorRegsPrefixD(dregs, sz, vd); if (IsVec4(sz, dregs)) { - ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum])); + ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]); ir.Write(IROp::Vec4Shuffle, dregs[0], IRVTEMP_0, 0); } else if (IsVec3of4(sz, dregs) && opts.preferVec4) { - ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum])); + ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]); ir.Write(IROp::Vec4Shuffle, IRVTEMP_0, IRVTEMP_0, 0); ir.Write(IROp::Vec4Blend, dregs[0], dregs[0], IRVTEMP_0, 0x7); } else { for (int i = 0; i < n; i++) { // Most of the time, materializing a float is slower than copying from another float. if (i == 0) - ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(cst_constants[conNum])); + ir.WriteSetConstantFloat(dregs[i], cst_constants[conNum]); else ir.Write(IROp::FMov, dregs[i], dregs[0]); } @@ -2253,7 +2254,7 @@ namespace MIPSComp { for (int i = 0; i < n; i++) { switch (d[i]) { case '0': - ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(0.0f)); + ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f); break; case 's': if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) { @@ -2271,7 +2272,7 @@ namespace MIPSComp { else if (dregs[sineLane] == sreg[0]) ir.Write(IROp::FCos, dregs[i], IRVTEMP_0); else - ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(1.0f)); + ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f); break; } } diff --git a/Core/MIPS/IR/IRFrontend.cpp b/Core/MIPS/IR/IRFrontend.cpp index c25660a566..2e125320c9 100644 --- a/Core/MIPS/IR/IRFrontend.cpp +++ b/Core/MIPS/IR/IRFrontend.cpp @@ -76,17 +76,17 @@ void IRFrontend::FlushPrefixV() { } if ((js.prefixSFlag & JitState::PREFIX_DIRTY) != 0) { - ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, ir.AddConstant(js.prefixS)); + ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, 0, 0, js.prefixS); js.prefixSFlag = (JitState::PrefixState) (js.prefixSFlag & ~JitState::PREFIX_DIRTY); } if ((js.prefixTFlag & JitState::PREFIX_DIRTY) != 0) { - ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, ir.AddConstant(js.prefixT)); + ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, 0, 0, js.prefixT); js.prefixTFlag = (JitState::PrefixState) (js.prefixTFlag & ~JitState::PREFIX_DIRTY); } if ((js.prefixDFlag & JitState::PREFIX_DIRTY) != 0) { - ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, ir.AddConstant(js.prefixD)); + ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, 0, 0, js.prefixD); js.prefixDFlag = (JitState::PrefixState) (js.prefixDFlag & ~JitState::PREFIX_DIRTY); } @@ -164,8 +164,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) { } else if (entry->replaceFunc) { FlushAll(); RestoreRoundingMode(); - ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC())); - ir.Write(IROp::CallReplacement, IRTEMP_0, ir.AddConstant(index)); + ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC()); + ir.Write(IROp::CallReplacement, IRTEMP_0, 0, 0, index); if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) { // Compile the original instruction at this address. We ignore cycles for hooks. @@ -175,8 +175,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) { ApplyRoundingMode(); // If IRTEMP_0 was set to 1, it means the replacement needs to run again (sliced.) // This is necessary for replacements that take a lot of cycles. - ir.Write(IROp::Downcount, 0, ir.AddConstant(js.downcountAmount)); - ir.Write(IROp::ExitToConstIfNeq, ir.AddConstant(GetCompilerPC()), IRTEMP_0, MIPS_REG_ZERO); + ir.Write(IROp::Downcount, 0, 0, 0, js.downcountAmount); + ir.Write(IROp::ExitToConstIfNeq, 0, IRTEMP_0, MIPS_REG_ZERO, GetCompilerPC()); ir.Write(IROp::ExitToReg, 0, MIPS_REG_RA, 0); js.compiling = false; } @@ -187,7 +187,7 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) { void IRFrontend::Comp_Generic(MIPSOpcode op) { FlushAll(); - ir.Write(IROp::Interpret, 0, ir.AddConstant(op.encoding)); + ir.Write(IROp::Interpret, 0, 0, 0, op.encoding); const MIPSInfo info = MIPSGetInfo(op); if ((info & IS_VFPU) != 0 && (info & VFPU_NO_PREFIX) == 0) { // If it does eat them, it'll happen in MIPSCompileOp(). @@ -359,7 +359,7 @@ void IRFrontend::CheckBreakpoint(u32 addr) { FlushAll(); // Can't skip this even at the start of a block, might impact block linking. - ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC())); + ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC()); RestoreRoundingMode(); // At this point, downcount HAS the delay slot, but not the instruction itself. @@ -374,11 +374,12 @@ void IRFrontend::CheckBreakpoint(u32 addr) { } } int downcountAmount = js.downcountAmount + downcountOffset; - if (downcountAmount != 0) - ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount)); + if (downcountAmount != 0) { + ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount); + } // Note that this means downcount can't be metadata on the block. js.downcountAmount = -downcountOffset; - ir.Write(IROp::Breakpoint, 0, ir.AddConstant(addr)); + ir.Write(IROp::Breakpoint, 0, 0, 0, addr); ApplyRoundingMode(); js.hadBreakpoints = true; @@ -390,7 +391,7 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) { FlushAll(); // Can't skip this even at the start of a block, might impact block linking. - ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC())); + ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC()); RestoreRoundingMode(); // At this point, downcount HAS the delay slot, but not the instruction itself. @@ -407,10 +408,10 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) { } int downcountAmount = js.downcountAmount + downcountOffset; if (downcountAmount != 0) - ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount)); + ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount); // Note that this means downcount can't be metadata on the block. js.downcountAmount = -downcountOffset; - ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, ir.AddConstant(offset)); + ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, 0, offset); ApplyRoundingMode(); js.hadBreakpoints = true; diff --git a/Core/MIPS/IR/IRInst.cpp b/Core/MIPS/IR/IRInst.cpp index 62f6375751..53a00ef8c9 100644 --- a/Core/MIPS/IR/IRInst.cpp +++ b/Core/MIPS/IR/IRInst.cpp @@ -176,6 +176,7 @@ static const IRMeta irMeta[] = { { IROp::ExitToConstIfLtZ, "ExitIfLtZ", "CG", IRFLAG_EXIT }, { IROp::ExitToReg, "ExitToReg", "_G", IRFLAG_EXIT }, { IROp::Syscall, "Syscall", "_C", IRFLAG_EXIT }, + { IROp::SyscallUnresolved, "SyscallUnresolved", "C", IRFLAG_EXIT }, { IROp::Break, "Break", "", IRFLAG_EXIT }, { IROp::SetPC, "SetPC", "_G" }, { IROp::SetPCConst, "SetPC", "_C" }, @@ -206,31 +207,35 @@ void InitIR() { } } -void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2) { +void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2, u32 constant) { IRInst inst; inst.op = op; inst.dest = dst; inst.src1 = src1; inst.src2 = src2; - inst.constant = nextConst_; + inst.constant = constant; insts_.push_back(inst); +} - nextConst_ = 0; +void IRWriter::WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant) { + u32 constant; + memcpy(&constant, &fconstant, sizeof(u32)); + + IRInst inst; + inst.op = op; + inst.dest = dst; + inst.src1 = src1; + inst.src2 = src2; + inst.constant = constant; + insts_.push_back(inst); } void IRWriter::WriteSetConstant(u8 dst, u32 value) { - Write(IROp::SetConst, dst, AddConstant(value)); + Write(IROp::SetConst, dst, 0, 0, value); } -int IRWriter::AddConstant(u32 value) { - nextConst_ = value; - return 255; -} - -int IRWriter::AddConstantFloat(float value) { - u32 val; - memcpy(&val, &value, 4); - return AddConstant(val); +void IRWriter::WriteSetConstantFloat(u8 dst, float value) { + WriteFC(IROp::SetConstF, dst, 0, 0, value); } void IRWriter::ReplaceConstant(size_t instNumber, u32 newConstant) { diff --git a/Core/MIPS/IR/IRInst.h b/Core/MIPS/IR/IRInst.h index 4348d527e5..313f7aa525 100644 --- a/Core/MIPS/IR/IRInst.h +++ b/Core/MIPS/IR/IRInst.h @@ -222,7 +222,8 @@ enum class IROp : uint8_t { ExitToConstIfFpFalse, ExitToPC, // Used after a syscall to give us a way to do things before returning. - Syscall, + Syscall, // Needs the syscall instruction work in the constant. + SyscallUnresolved, // Used when the syscall is not resolved at compile time. PC in the constant. SetPC, // hack to make syscall returns work SetPCConst, // hack to make replacement know PC CallReplacement, @@ -384,18 +385,14 @@ public: return *this; } - void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0); - void Write(IROp op, IRReg dst, IRReg src1, IRReg src2, uint32_t c) { - AddConstant(c); - Write(op, dst, src1, src2); - } + void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0, u32 constant = 0); + void WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant); + + void WriteSetConstant(u8 dst, u32 value); + void WriteSetConstantFloat(u8 dst, float value); void Write(IRInst inst) { insts_.push_back(inst); } - void WriteSetConstant(u8 dst, u32 value); - - int AddConstant(u32 value); - int AddConstantFloat(float value); void Reserve(size_t s) { insts_.reserve(s); @@ -409,7 +406,6 @@ public: private: std::vector insts_; - u32 nextConst_ = 0; }; struct IROptions { diff --git a/Core/MIPS/IR/IRInterpreter.cpp b/Core/MIPS/IR/IRInterpreter.cpp index 8fc7c14294..27465a1616 100644 --- a/Core/MIPS/IR/IRInterpreter.cpp +++ b/Core/MIPS/IR/IRInterpreter.cpp @@ -1203,10 +1203,24 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) { case IROp::Syscall: // IROp::SetPC was (hopefully) executed before. { + // If we get here, the syscall is valid. MIPSOpcode op(inst->constant); CallSyscall(op); - if (coreState != CORE_RUNNING_CPU) + if (coreState != CORE_RUNNING_CPU) { CoreTiming::ForceCheck(mips); + } + break; + } + + case IROp::SyscallUnresolved: + { + // If we get here, the syscall is invalid. + u32 pc = inst->constant; + CallSyscallUnresolvedAtPC(pc); + if (coreState != CORE_RUNNING_CPU) { + // hm, what's this for? + CoreTiming::ForceCheck(mips); + } break; } @@ -1300,7 +1314,7 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) { } break; - case IROp::Nop: // TODO: This shouldn't crash, but for now we should not emit nops, so... + case IROp::Nop: // Unused, add a break if we start using it to avoid UNREACHABLE. case IROp::Bad: default: // Unimplemented IR op. Bad. We define it as unreachable so the compiler can optimize better (remove the range check). diff --git a/Core/MIPS/IR/IRNativeCommon.cpp b/Core/MIPS/IR/IRNativeCommon.cpp index 5dea3fedc1..20641a1c91 100644 --- a/Core/MIPS/IR/IRNativeCommon.cpp +++ b/Core/MIPS/IR/IRNativeCommon.cpp @@ -440,6 +440,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) { break; case IROp::Syscall: + case IROp::SyscallUnresolved: case IROp::CallReplacement: case IROp::Break: CompIR_System(inst); diff --git a/Core/MIPS/IR/IRPassSimplify.cpp b/Core/MIPS/IR/IRPassSimplify.cpp index 30b96a2b3d..373093904d 100644 --- a/Core/MIPS/IR/IRPassSimplify.cpp +++ b/Core/MIPS/IR/IRPassSimplify.cpp @@ -258,21 +258,21 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions if (opts.unalignedLoadStore) { // Write out one unaligned op. - out.Write(replaceOp, inst.dest, inst.src1, out.AddConstant(inst.constant + replaceOff)); + out.Write(replaceOp, inst.dest, inst.src1, 0, inst.constant + replaceOff); } else if (replaceOp == IROp::Load32) { // We can still combine to a simpler set of two loads. // We start by isolating the address and shift amount. // IRTEMP_LR_ADDR = rs + imm - out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant + replaceOff)); + out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant + replaceOff); // IRTEMP_LR_SHIFT = (addr & 3) * 8 - out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3)); + out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3); out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3); // IRTEMP_LR_ADDR = addr & 0xfffffffc - out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC)); + out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC); // IRTEMP_LR_VALUE = low_word, dest = high_word - out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, out.AddConstant(0)); - out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(4)); + out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, 0, 0); + out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 4); // Now we just need to adjust and combine dest and IRTEMP_LR_VALUE. // inst.dest >>= shift (putting its bits in the right spot.) @@ -281,7 +281,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions out.Write(IROp::ShlImm, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, 8); // IRTEMP_LR_SHIFT = 24 - shift out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24); // IRTEMP_LR_VALUE <<= (24 - shift) out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT); @@ -297,18 +297,18 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions auto addCommonProlog = [&]() { // IRTEMP_LR_ADDR = rs + imm - out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant)); + out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant); // IRTEMP_LR_SHIFT = (addr & 3) * 8 - out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3)); + out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3); out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3); // IRTEMP_LR_ADDR = addr & 0xfffffffc (for stores, later) - out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC)); + out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC); // IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR) - out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(0)); + out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 0); }; auto addCommonStore = [&](int off = 0) { // RAM(IRTEMP_LR_ADDR) = IRTEMP_LR_VALUE - out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(off)); + out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, off); }; switch (inst.op) { @@ -327,7 +327,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions out.Write(IROp::And, inst.dest, inst.dest, IRTEMP_LR_MASK); // IRTEMP_LR_SHIFT = 24 - shift out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24); // IRTEMP_LR_VALUE <<= (24 - shift) out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT); // dest |= IRTEMP_LR_VALUE @@ -336,7 +336,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions bool src1Dirty = inst.dest == inst.src1; while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) { // IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta) - out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant)); + out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, nextOp().constant - inst.constant); // dest &= IRTEMP_LR_MASK out.Write(IROp::And, nextOp().dest, nextOp().dest, IRTEMP_LR_MASK); @@ -362,7 +362,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions out.Write(IROp::Shr, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT); // IRTEMP_LR_SHIFT = 24 - shift out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24); // dest &= (0xffffff00 << (24 - shift)) // Alternatively, could shift to a wall and back (but would require two shifts each way.) out.WriteSetConstant(IRTEMP_LR_MASK, 0xffffff00); @@ -377,12 +377,12 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions bool src1Dirty = inst.dest == inst.src1; while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) { // IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta) - out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant)); + out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, (u32)(nextOp().constant - inst.constant)); if (shiftNeedsReverse) { // IRTEMP_LR_SHIFT = shift again out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24); shiftNeedsReverse = false; } // IRTEMP_LR_VALUE >>= IRTEMP_LR_SHIFT @@ -411,7 +411,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK); // IRTEMP_LR_SHIFT = 24 - shift out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24); // IRTEMP_LR_VALUE |= src3 >> (24 - shift) out.Write(IROp::Shr, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT); out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK); @@ -429,11 +429,11 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions // IRTEMP_LR_VALUE &= 0x00ffffff << (24 - shift) out.WriteSetConstant(IRTEMP_LR_MASK, 0x00ffffff); out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24); out.Write(IROp::Shr, IRTEMP_LR_MASK, IRTEMP_LR_MASK, IRTEMP_LR_SHIFT); out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK); out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT); - out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24)); + out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24); // IRTEMP_LR_VALUE |= src3 << shift out.Write(IROp::Shl, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT); out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK); @@ -508,7 +508,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts if (inst.dest != inst.src1) out.Write(IROp::Mov, inst.dest, inst.src1); } else { - out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, out.AddConstant(imm2)); + out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, 0, imm2); } } else if (symmetric && gpr.IsImm(inst.src1)) { const u32 imm1 = gpr.GetImm(inst.src1); @@ -518,7 +518,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts if (inst.dest != inst.src2) out.Write(IROp::Mov, inst.dest, inst.src2); } else { - out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, out.AddConstant(imm1)); + out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, 0, imm1); } } else { gpr.MapDirtyInIn(inst.dest, inst.src1, inst.src2); @@ -632,7 +632,8 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::FMovFromGPR: if (gpr.IsImm(inst.src1)) { - out.Write(IROp::SetConstF, inst.dest, out.AddConstant(gpr.GetImm(inst.src1))); + // NOTE: SetConstantFloat doesn't work here since we actually want the bits. + out.Write(IROp::SetConstF, inst.dest, 0, 0, gpr.GetImm(inst.src1)); } else { gpr.MapIn(inst.src1); goto doDefault; @@ -661,7 +662,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::Store32Conditional: if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) { gpr.MapIn(inst.dest); - out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant)); + out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant); } else { gpr.MapInIn(inst.dest, inst.src1); goto doDefault; @@ -670,7 +671,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::StoreFloat: case IROp::StoreVec4: if (gpr.IsImm(inst.src1)) { - out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant)); + out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant); } else { gpr.MapIn(inst.src1); goto doDefault; @@ -685,7 +686,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::Load32Linked: if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) { gpr.MapDirty(inst.dest); - out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant)); + out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant); } else { gpr.MapDirtyIn(inst.dest, inst.src1); goto doDefault; @@ -694,7 +695,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::LoadFloat: case IROp::LoadVec4: if (gpr.IsImm(inst.src1)) { - out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant)); + out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant); } else { gpr.MapIn(inst.src1); goto doDefault; @@ -704,7 +705,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::Load32Right: if (gpr.IsImm(inst.src1)) { gpr.MapIn(inst.dest); - out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant)); + out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant); } else { gpr.MapInIn(inst.dest, inst.src1); goto doDefault; @@ -716,7 +717,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::ValidateAddress32: case IROp::ValidateAddress128: if (gpr.IsImm(inst.src1)) { - out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant)); + out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant); } else { gpr.MapIn(inst.src1); goto doDefault; @@ -729,7 +730,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::SetPC: if (gpr.IsImm(inst.src1)) { - out.Write(IROp::SetPCConst, out.AddConstant(gpr.GetImm(inst.src1))); + out.Write(IROp::SetPCConst, 0, 0, 0, gpr.GetImm(inst.src1)); } else { gpr.MapIn(inst.src1); goto doDefault; @@ -772,7 +773,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::SetCtrlVFPUReg: if (gpr.IsImm(inst.src1)) { - out.Write(IROp::SetCtrlVFPU, inst.dest, out.AddConstant(gpr.GetImm(inst.src1))); + out.Write(IROp::SetCtrlVFPU, inst.dest, 0, 0, gpr.GetImm(inst.src1)); } else { gpr.MapDirtyIn(IRREG_VFPU_CTRL_BASE + inst.dest, inst.src1); out.Write(inst); @@ -872,7 +873,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts // Reduce bloat by skipping on fail, and const exit on pass. if (passed) { gpr.FlushAll(); - out.Write(IROp::ExitToConst, out.AddConstant(inst.constant)); + out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant); skipNextExitToConst = true; } break; @@ -896,7 +897,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts if (passed) { gpr.FlushAll(); - out.Write(IROp::ExitToConst, out.AddConstant(inst.constant)); + out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant); skipNextExitToConst = true; } break; @@ -918,7 +919,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts // Prefer ExitToConst to allow block linking. u32 dest = gpr.GetImm(inst.src1); gpr.FlushAll(); - out.Write(IROp::ExitToConst, out.AddConstant(dest)); + out.Write(IROp::ExitToConst, 0, 0, 0, dest); break; } gpr.FlushAll(); @@ -2040,10 +2041,11 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) { } else if (inst.constant == 0xBF800000) { out.Write(IROp::Vec4Init, temp, (int)Vec4Init::AllMinusONE); } else { - out.Write(IROp::SetConstF, temp, out.AddConstant(inst.constant)); + // NOTE: WriteSetConstantFloat doesn't work here since we actually want the bits. + out.Write(IROp::SetConstF, temp, 0, 0, inst.constant); out.Write(IROp::Vec4Shuffle, temp, temp, 0); } - out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask); + out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask); isVec4Dirty[inst.dest & ~3] = true; continue; } @@ -2054,7 +2056,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) { u8 blendMask = 1 << (inst.dest & 3); out.Write(IROp::FMovFromGPR, temp, inst.src1); out.Write(IROp::Vec4Shuffle, temp, temp, 0); - out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask); + out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask); isVec4Dirty[inst.dest & ~3] = true; continue; } @@ -2065,7 +2067,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) { u8 blendMask = 1 << (inst.dest & 3); out.Write(inst.op, temp, inst.src1, inst.src2, inst.constant); out.Write(IROp::Vec4Shuffle, temp, temp, 0); - out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask); + out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask); isVec4Dirty[inst.dest & ~3] = true; continue; } diff --git a/Core/MIPS/Interpreter.cpp b/Core/MIPS/Interpreter.cpp index e468dee158..2aa75a6f71 100644 --- a/Core/MIPS/Interpreter.cpp +++ b/Core/MIPS/Interpreter.cpp @@ -184,6 +184,7 @@ namespace MIPSInt { } void Int_Syscall(MIPSState *mips, MIPSOpcode op) { + const u32 syscallPC = mips->pc - 4; // Need to pre-move PC, as CallSyscall may result in a rescheduling! // To do this neater, we'll need a little generated kernel loop that syscall can jump to and then RFI from // but I don't see a need to bother. @@ -193,7 +194,7 @@ namespace MIPSInt { mips->pc += 4; } mips->inDelaySlot = false; - CallSyscall(op); + CallSyscallWithPC(op, syscallPC); } void Int_Sync(MIPSState *mips, MIPSOpcode op) { diff --git a/Core/MIPS/LoongArch64/LoongArch64CompSystem.cpp b/Core/MIPS/LoongArch64/LoongArch64CompSystem.cpp index c5721ec700..104acba387 100644 --- a/Core/MIPS/LoongArch64/LoongArch64CompSystem.cpp +++ b/Core/MIPS/LoongArch64/LoongArch64CompSystem.cpp @@ -185,22 +185,38 @@ void LoongArch64JitBackend::CompIR_System(IRInst inst) { // Skip the CallSyscall where possible. { MIPSOpcode op(inst.constant); - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) { - LI(R4, (uintptr_t)GetSyscallFuncPointer(op)); - QuickCallFunction((const u8 *)quickFunc, SCRATCH2); + const HLEFunction *func = GetSyscallFunctionData(op, 0); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + LI(R4, func); + QuickCallFunction((const u8 *)quickFunc, SCRATCH2); + } else { + LI(R4, (int32_t)inst.constant); + QuickCallFunction(&CallSyscall, SCRATCH2); + } } else { - LI(R4, (int32_t)inst.constant); - QuickCallFunction(&CallSyscall, SCRATCH2); + // Shouldn't get here. + LI(R4, 0); + QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2); } } #endif - WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT); LoadStaticRegisters(); // This is always followed by an ExitToPC, where we check coreState. break; + case IROp::SyscallUnresolved: + FlushAll(); + SaveStaticRegisters(); + WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL); + LI(R4, inst.constant); + QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2); + WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT); + LoadStaticRegisters(); + break; + case IROp::CallReplacement: FlushAll(); SaveStaticRegisters(); @@ -275,4 +291,4 @@ void LoongArch64JitBackend::CompIR_ValidateAddress(IRInst inst) { } } -} // namespace MIPSComp \ No newline at end of file +} // namespace MIPSComp diff --git a/Core/MIPS/LoongArch64/LoongArch64Jit.cpp b/Core/MIPS/LoongArch64/LoongArch64Jit.cpp index 7d7b028645..eb9e56570e 100644 --- a/Core/MIPS/LoongArch64/LoongArch64Jit.cpp +++ b/Core/MIPS/LoongArch64/LoongArch64Jit.cpp @@ -408,4 +408,4 @@ LoongArch64Reg LoongArch64JitBackend::NormalizeR(IRReg rs, IRReg rd, LoongArch64 } } -} // namespace MIPSComp \ No newline at end of file +} // namespace MIPSComp diff --git a/Core/MIPS/MIPSTables.cpp b/Core/MIPS/MIPSTables.cpp index e6ac3ce753..fec07b393a 100644 --- a/Core/MIPS/MIPSTables.cpp +++ b/Core/MIPS/MIPSTables.cpp @@ -1116,18 +1116,19 @@ static void RunUntilDowncountZeroFast(MIPSState *mips) { int cycleCount = 0; // Don't stop in a delay slot! do { - if (!Memory::IsValid4AlignedAddress(mips->pc)) { - Core_ExecException(mips->pc, mips->pc, ExecExceptionType::JUMP); + const u32 pc = mips->pc; + if (!Memory::IsValid4AlignedAddress(pc)) { + Core_ExecException(pc, pc, ExecExceptionType::JUMP); return; } - const MIPSOpcode op = MIPSOpcode(Memory::ReadUnchecked_U32(mips->pc)); + const MIPSOpcode op = MIPSOpcode(Memory::ReadUnchecked_U32(pc)); const bool wasInDelaySlot = mips->inDelaySlot; const int cycles = ExecInstruction(mips, op); if (cycles < 0) { // Not a recognized instruction (invalid encoding, or an unimplemented kernel-mode only instruction // with no interpreter implementation, e.g. tge/tlt/teq/). - Core_ExecException(mips->pc, mips->pc, ExecExceptionType::ILLEGAL); + Core_ExecException(pc, pc, ExecExceptionType::ILLEGAL); return; } cycleCount += cycles; diff --git a/Core/MIPS/RiscV/RiscVCompSystem.cpp b/Core/MIPS/RiscV/RiscVCompSystem.cpp index d3cbabf1d5..564cb5fea3 100644 --- a/Core/MIPS/RiscV/RiscVCompSystem.cpp +++ b/Core/MIPS/RiscV/RiscVCompSystem.cpp @@ -193,13 +193,20 @@ void RiscVJitBackend::CompIR_System(IRInst inst) { // Skip the CallSyscall where possible. { MIPSOpcode op(inst.constant); - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) { - LI(X10, (uintptr_t)GetSyscallFuncPointer(op)); - QuickCallFunction((const u8 *)quickFunc, SCRATCH2); + const HLEFunction *func = GetSyscallFunctionData(op, 0); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + LI(X10, (uintptr_t)func); + QuickCallFunction((const u8 *)quickFunc, SCRATCH2); + } else { + LI(X10, (int32_t)inst.constant); + QuickCallFunction(&CallSyscall, SCRATCH2); + } } else { - LI(X10, (int32_t)inst.constant); - QuickCallFunction(&CallSyscall, SCRATCH2); + // Shouldn't get here. + LI(X10, 0); + QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2); } } #endif diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 8db2c28883..dff306a95c 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -681,11 +681,17 @@ void Jit::Comp_Syscall(MIPSOpcode op) ABI_CallFunctionC(&CallSyscall, op.encoding); #else // Skip the CallSyscall where possible. - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) - ABI_CallFunctionP(quickFunc, (void *)GetSyscallFuncPointer(op)); - else - ABI_CallFunctionC(&CallSyscall, op.encoding); + const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + ABI_CallFunctionP(quickFunc, (void *)func); + } else { + ABI_CallFunctionC(&CallSyscall, op.encoding); + } + } else { + ABI_CallFunctionC(&CallSyscallUnresolvedAtPC, js.compilerPC); + } #endif ApplyRoundingMode(); diff --git a/Core/MIPS/x86/X64IRCompSystem.cpp b/Core/MIPS/x86/X64IRCompSystem.cpp index b137ee176d..0fe7bcb262 100644 --- a/Core/MIPS/x86/X64IRCompSystem.cpp +++ b/Core/MIPS/x86/X64IRCompSystem.cpp @@ -211,11 +211,18 @@ void X64JitBackend::CompIR_System(IRInst inst) { // Skip the CallSyscall where possible. { MIPSOpcode op(inst.constant); - void *quickFunc = GetQuickSyscallFunc(op); - if (quickFunc) { - ABI_CallFunctionP((const u8 *)quickFunc, (void *)GetSyscallFuncPointer(op)); + const HLEFunction *func = GetSyscallFunctionData(op, 0); + if (func) { + void *quickFunc = GetQuickSyscallFunc(func, op); + if (quickFunc) { + ABI_CallFunctionP((const u8 *)quickFunc, (void *)func); + } else { + ABI_CallFunctionC((const u8 *)&CallSyscall, inst.constant); + } } else { - ABI_CallFunctionC((const u8 *)&CallSyscall, inst.constant); + _dbg_assert_(false); + // Shouldn't get here, this should be resolved during ->IR compilation. + ABI_CallFunctionC((const u8 *)&CallSyscallUnresolvedAtPC, 0); } } #endif @@ -225,6 +232,15 @@ void X64JitBackend::CompIR_System(IRInst inst) { // This is always followed by an ExitToPC, where we check coreState. break; + case IROp::SyscallUnresolved: + FlushAll(); + SaveStaticRegisters(); + WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL); + ABI_CallFunctionC((const u8 *)&CallSyscallUnresolvedAtPC, inst.constant); + WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT); + LoadStaticRegisters(); + break; + case IROp::CallReplacement: FlushAll(); SaveStaticRegisters(); diff --git a/Core/MemMap.h b/Core/MemMap.h index cf554cc16f..2444f0c6e7 100644 --- a/Core/MemMap.h +++ b/Core/MemMap.h @@ -296,6 +296,10 @@ inline void MemcpyUnchecked(const u32 to_address, const u32 from_address, const MemcpyUnchecked(GetPointerWriteUnchecked(to_address), from_address, len); } +inline bool AddressesEqualAfterMask(const u32 address1, const u32 address2) { + return (address1 & 0x3FFFFFFF) == (address2 & 0x3FFFFFFF); +} + // Applies to user mode. // Without a length, IsValidAddress is generally semi-meaningless, unless it's about a single byte access. For larger accesses, use IsValid4AlignedAddress // etc when appropriate, or for longer sizes, use IsValidRange or IsValid4AlignedRange for example. Checking aligned-ness helps avoid the problem diff --git a/Core/System.cpp b/Core/System.cpp index fe22a31080..9416dc3a32 100644 --- a/Core/System.cpp +++ b/Core/System.cpp @@ -93,7 +93,7 @@ static FileLoader *g_loadedFile; static std::mutex loadingLock; static std::thread g_loadingThread; -bool coreCollectDebugStats = false; +bool g_coreCollectDebugStats = false; static int coreCollectDebugStatsCounter = 0; static volatile CPUThreadState cpuThreadState = CPU_THREAD_NOT_RUNNING; @@ -608,8 +608,8 @@ void UpdateLoadedFile(FileLoader *fileLoader) { void PSP_UpdateDebugStats(bool collectStats) { bool newState = collectStats || coreCollectDebugStatsCounter > 0; - if (coreCollectDebugStats != newState) { - coreCollectDebugStats = newState; + if (g_coreCollectDebugStats != newState) { + g_coreCollectDebugStats = newState; mipsr4k.ClearJitCache(); } diff --git a/Core/System.h b/Core/System.h index a5dd39489d..7019ebd8b3 100644 --- a/Core/System.h +++ b/Core/System.h @@ -98,7 +98,7 @@ Path GetSysDirectory(PSPDirectories directoryType); bool CreateSysDirectories(); -extern bool coreCollectDebugStats; +extern bool g_coreCollectDebugStats; inline CoreParameter &PSP_CoreParameter() { extern CoreParameter g_CoreParameter; diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 06ec795633..794eeae3c0 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -1155,7 +1155,7 @@ void DrawEngineCommon::DepthRasterSubmitRaw(GEPrimitiveType prim, const VertexDe return; } - TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, coreCollectDebugStats); + TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, g_coreCollectDebugStats); // Decode. int numDecoded = 0; @@ -1197,7 +1197,7 @@ void DrawEngineCommon::DepthRasterPredecoded(GEPrimitiveType prim, const void *i return; } - TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, coreCollectDebugStats); + TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, g_coreCollectDebugStats); // Make sure these have already been indexed away. _dbg_assert_(prim != GE_PRIM_TRIANGLE_STRIP && prim != GE_PRIM_TRIANGLE_FAN); @@ -1234,7 +1234,7 @@ void DrawEngineCommon::FlushQueuedDepth() { rasterTimeStart_ = 0.0; } - const bool collectStats = coreCollectDebugStats; + const bool collectStats = g_coreCollectDebugStats; const bool lowQ = g_Config.iDepthRasterMode == (int)DepthRasterMode::LOW_QUALITY; for (const auto &draw : depthDraws_) { int *tx = depthScreenVerts_; diff --git a/GPU/GPUCommon.cpp b/GPU/GPUCommon.cpp index be6157e334..81bf045b56 100644 --- a/GPU/GPUCommon.cpp +++ b/GPU/GPUCommon.cpp @@ -745,7 +745,7 @@ DLResult GPUCommon::ProcessDLQueue() { } } - TimeCollector collectStat(&gpuStats.perFrame.msProcessingDisplayLists, coreCollectDebugStats); + TimeCollector collectStat(&gpuStats.perFrame.msProcessingDisplayLists, g_coreCollectDebugStats); auto GetNextListIndex = [&]() -> int { if (dlQueue.empty()) @@ -873,7 +873,7 @@ DLResult GPUCommon::ProcessDLQueue() { currentList = nullptr; - if (coreCollectDebugStats) { + if (g_coreCollectDebugStats) { gpuStats.perFrame.otherGPUCycles += cyclesExecuted; } @@ -1394,7 +1394,7 @@ void GPUCommon::FastLoadBoneMatrix(u32 target) { cyclesExecuted += 2 * 14; // one to reset the counter, 12 to load the matrix, and a return. - if (coreCollectDebugStats) { + if (g_coreCollectDebugStats) { gpuStats.perFrame.otherGPUCycles += 2 * 14; } } diff --git a/GPU/Software/BinManager.cpp b/GPU/Software/BinManager.cpp index fddafdd28e..c1e4d12dc5 100644 --- a/GPU/Software/BinManager.cpp +++ b/GPU/Software/BinManager.cpp @@ -563,8 +563,10 @@ void BinManager::Flush(const char *reason) { return; double st = 0.0; - if (coreCollectDebugStats) + const bool collectDebugStats = g_coreCollectDebugStats; + if (collectDebugStats) { st = time_now_d(); + } Drain(true); waitable_->Wait(); taskRanges_.clear(); @@ -584,16 +586,17 @@ void BinManager::Flush(const char *reason) { queueRange_.x2 = 0; queueRange_.y2 = 0; - for (auto &pending : pendingWrites_) + for (BinDirtyRange &pending : pendingWrites_) { pending.base = 0; + } pendingOverlap_ = false; pendingReads_.clear(); // We'll need to set the pending writes and reads again, since we just flushed it. dirty_ |= SoftDirty::BINNER_RANGE | SoftDirty::BINNER_OVERLAP; - if (coreCollectDebugStats) { - double et = time_now_d(); + if (collectDebugStats) { + const double et = time_now_d(); flushReasonTimes_[reason] += et - st; if (et - st > slowestFlushTime_) { slowestFlushTime_ = et - st; @@ -611,7 +614,7 @@ void BinManager::OptimizePendingStates(uint16_t first, uint16_t last) { last--; } - int count = (QUEUED_STATES + last - first) % QUEUED_STATES + 1; + const int count = (QUEUED_STATES + last - first) % QUEUED_STATES + 1; for (int i = 0; i < count; ++i) { size_t pos = (first + i) % QUEUED_STATES; OptimizeRasterState(&states_[pos]); diff --git a/UI/DeveloperToolsScreen.cpp b/UI/DeveloperToolsScreen.cpp index 63321d4d9c..7e5eff89c2 100644 --- a/UI/DeveloperToolsScreen.cpp +++ b/UI/DeveloperToolsScreen.cpp @@ -154,7 +154,8 @@ void DeveloperToolsScreen::CreateGeneralTab(UI::LinearLayout *list) { core->HideChoice(1); core->HideChoice(3); } - // TODO: Enable "JIT using IR" on more architectures. + // TODO: Enable "JIT using IR" on more architectures. ARM32 needs more testing. + // Note also that Loongarch and RISC-V only have a jit-ir backend, and it's used for the JIT option, so the fourth option isn't shown. #if !PPSSPP_ARCH(X86) && !PPSSPP_ARCH(AMD64) && !PPSSPP_ARCH(ARM64) core->HideChoice(3); #endif diff --git a/android/src/org/ppsspp/ppsspp/PpssppActivity.java b/android/src/org/ppsspp/ppsspp/PpssppActivity.java index 428f318c3d..a3d1c92ccf 100644 --- a/android/src/org/ppsspp/ppsspp/PpssppActivity.java +++ b/android/src/org/ppsspp/ppsspp/PpssppActivity.java @@ -1603,8 +1603,6 @@ public class PpssppActivity extends AppCompatActivity implements SensorEventList return true; } else if (command.equals("showKeyboard") && surfView != null) { InputMethodManager inputMethodManager = (InputMethodManager) getSystemService(Context.INPUT_METHOD_SERVICE); - // No idea what the point of the ApplicationWindowToken is or if it - // matters where we get it from... inputMethodManager.showSoftInput(surfView, InputMethodManager.SHOW_IMPLICIT); return true; } else if (command.equals("hideKeyboard") && surfView != null) {