Merge pull request #22097 from hrydgard/unresolved-syscall

Log unresolved syscalls properly
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-08-15 21:57:51 +02:00
commit 45e11ee90f
39 files changed
+420 -294

No files matched your search

+2 -2
View File
@@ -98,8 +98,8 @@ struct LogChannel {
#endif
bool enabled = true;
bool IsEnabled(LogLevel level) const {
if (level > this->level || !this->enabled)
bool IsEnabled(LogLevel logLevel) const {
if (logLevel > level || !enabled)
return false;
return true;
}
+6 -6
View File
@@ -197,17 +197,17 @@ inline __m128i _mm_mullo_epi32_SSE2(const __m128i v0, const __m128i v1) {
inline __m128i _mm_max_epu16_SSE2(const __m128i v0, const __m128i v1) {
return _mm_xor_si128(
_mm_max_epi16(
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
_mm_set1_epi16((int16_t)0x8000));
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
}
inline __m128i _mm_min_epu16_SSE2(const __m128i v0, const __m128i v1) {
return _mm_xor_si128(
_mm_min_epi16(
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
_mm_set1_epi16((int16_t)0x8000));
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
}
// SSE2 replacement for half of a _mm_packus_epi32 but without the saturation.
+1 -1
View File
@@ -34,7 +34,7 @@ inline constexpr uint32_t RoundDownToMultipleOf(uint32_t v, uint32_t multiple) {
// TODO: this should just use a bitscan.
inline uint32_t log2i(uint32_t val) {
unsigned int ret = -1;
unsigned int ret = (unsigned int)-1;
while (val != 0) {
val >>= 1; ret++;
}
+51 -33
View File
@@ -78,7 +78,7 @@ static const HLEFunction *g_stack[MAX_SYSCALL_RECURSION];
u32 g_syscallPC;
int g_stackSize;
static int idleOp;
static int g_idleOp;
// Split syscall support. NOTE: This needs to be saved in DoState somehow!
static int splitSyscallEatCycles = 0;
@@ -266,7 +266,7 @@ void HLEInit() {
RegisterAllModules();
g_stackSize = 0;
delayedResultEvent = CoreTiming::RegisterEvent("HLEDelayedResult", hleDelayResultFinish);
idleOp = GetSyscallOp("FakeSysCalls", NID_IDLE);
g_idleOp = GetSyscallOp("FakeSysCalls", NID_IDLE);
}
void HLEDoState(PointerWrap &p) {
@@ -801,10 +801,10 @@ void hleFinishSyscallAfterGe() {
hleFinishSyscall(nullptr);
}
static void updateSyscallStats(int modulenum, int funcnum, double total) {
static void UpdateSyscallStats(int modulenum, int funcnum, double total) {
const char *name = moduleDB[modulenum].funcTable[funcnum].name;
// Ignore this one, especially for msInSyscalls (although that ignores CoreTiming events.)
if (0 == strcmp(name, "_sceKernelIdle"))
if (equals(name, "_sceKernelIdle"))
return;
if (total > kernelStats.slowestSyscallTime) {
@@ -822,7 +822,7 @@ static void updateSyscallStats(int modulenum, int funcnum, double total) {
kernelStats.summedSlowestSyscallName = name;
}
} else {
double newTotal = kernelStats.summedMsInSyscalls[statCall] += total;
const double newTotal = kernelStats.summedMsInSyscalls[statCall] += total;
if (newTotal > kernelStats.summedSlowestSyscallTime) {
kernelStats.summedSlowestSyscallTime = newTotal;
kernelStats.summedSlowestSyscallName = name;
@@ -895,66 +895,76 @@ static void CallSyscallWithoutFlags(const HLEFunction *info) {
g_stackSize = 0;
}
const HLEFunction *GetSyscallFuncPointer(MIPSOpcode op) {
void LogBadSyscallAtPC(u32 pc, bool compilePhase) {
const LogLevel level = compilePhase ? LogLevel::LWARNING : LogLevel::LERROR;
const char *phase = compilePhase ? "compile" : "run";
std::string importModuleName, importingModuleName;
u32 nid = 0;
if (pc && KernelFindImportByStubAddr(pc, &importModuleName, &nid, &importingModuleName)) {
const char *funcName = GetHLEFuncName(importModuleName, nid);
GENERIC_LOG(Log::HLE, level, "Unknown syscall (%s) at %08x: unresolved import %s/%08x (%s), called from '%s'", phase, pc, importModuleName.c_str(), nid, funcName ? funcName : "(unknown)", importingModuleName.c_str());
} else {
char buffer[256];
DescribeAddress(currentDebugMIPS, pc, buffer, sizeof(buffer));
GENERIC_LOG(Log::HLE, level, "Unknown syscall (%s) at %08x (%s): was unable to determine more information", phase, pc, buffer);
}
}
const HLEFunction *GetSyscallFunctionData(MIPSOpcode op, u32 pcForDiagnostics) {
u32 callno = (op >> 6) & 0xFFFFF; //20 bits
int funcnum = callno & 0xFFF;
int modulenum = (callno & 0xFF000) >> 12;
if (funcnum == 0xfff) {
std::string_view modName = modulenum >= (int)moduleDB.size() ? "(unknown)" : moduleDB[modulenum].name;
// This is what a still-unresolved import looks like once written as a syscall opcode -
// the original module name/NID aren't recoverable from the opcode itself (see
// WriteFuncMissingStub), but the calling address is a stub we may still be tracking.
std::string importModuleName, importingModuleName;
u32 nid = 0;
if (currentMIPS->pc >= 8 && KernelFindImportByStubAddr(currentMIPS->pc - 8, &importModuleName, &nid, &importingModuleName)) {
const char *funcName = GetHLEFuncName(importModuleName, nid);
ERROR_LOG(Log::HLE, "Unknown syscall: unresolved import %s/%08x (%s), called from '%s'", importModuleName.c_str(), nid, funcName ? funcName : "(unknown)", importingModuleName.c_str());
} else {
ERROR_LOG(Log::HLE, "Unknown syscall: Module: '%.*s' (module: %d func: %d)", (int)modName.size(), modName.data(), modulenum, funcnum);
}
return NULL;
LogBadSyscallAtPC(pcForDiagnostics, PSP_CoreParameter().cpuCore != CPUCore::INTERPRETER);
return nullptr;
}
if (modulenum >= (int)moduleDB.size()) {
ERROR_LOG(Log::HLE, "Syscall had bad module number %d - probably executing garbage", modulenum);
return NULL;
return nullptr;
}
if (funcnum >= moduleDB[modulenum].numFunctions) {
ERROR_LOG(Log::HLE, "Syscall had bad function number %d in module %d - probably executing garbage", funcnum, modulenum);
return NULL;
return nullptr;
}
return &moduleDB[modulenum].funcTable[funcnum];
}
void *GetQuickSyscallFunc(MIPSOpcode op) {
if (coreCollectDebugStats)
void *GetQuickSyscallFunc(const HLEFunction *info, MIPSOpcode op) {
if (g_coreCollectDebugStats)
return nullptr;
const HLEFunction *info = GetSyscallFuncPointer(op);
if (!info || !info->func)
return nullptr;
VERBOSE_LOG(Log::HLE, "Compiling syscall to '%s'", info->name);
// TODO: Do this with a flag?
if (op == idleOp)
if (op == g_idleOp) {
return (void *)info->func;
if (info->flags != 0)
} else if (info->flags != 0) {
return (void *)&CallSyscallWithFlags;
return (void *)&CallSyscallWithoutFlags;
} else {
return (void *)&CallSyscallWithoutFlags;
}
}
void hleSetFlipTime(double t) {
hleFlipTime = t;
}
void CallSyscall(MIPSOpcode op) {
void CallSyscallWithPC(MIPSOpcode op, u32 pc) {
PROFILE_THIS_SCOPE("syscall");
double start = 0.0; // need to initialize to fix the race condition where coreCollectDebugStats is enabled in the middle of this func.
if (coreCollectDebugStats) {
const bool collectStats = g_coreCollectDebugStats;
double start = 0.0;
if (collectStats) {
start = time_now_d();
}
const HLEFunction *info = GetSyscallFuncPointer(op);
const HLEFunction *info = GetSyscallFunctionData(op, pc);
if (!info) {
// We haven't incremented the stack yet.
RETURN(SCE_KERNEL_ERROR_LIBRARY_NOT_YET_LINKED);
@@ -962,7 +972,7 @@ void CallSyscall(MIPSOpcode op) {
}
if (info->func) {
if (op == idleOp)
if (op == g_idleOp)
info->func();
else if (info->flags != 0)
CallSyscallWithFlags(info);
@@ -974,19 +984,27 @@ void CallSyscall(MIPSOpcode op) {
ERROR_LOG_REPORT(Log::HLE, "Unimplemented HLE function %s", info->name ? info->name : "(\?\?\?)");
}
if (coreCollectDebugStats) {
u32 callno = (op >> 6) & 0xFFFFF; //20 bits
int funcnum = callno & 0xFFF;
int modulenum = (callno & 0xFF000) >> 12;
if (collectStats) {
const u32 callno = (op >> 6) & 0xFFFFF; // 20 bits
const int funcnum = callno & 0xFFF; // 12 bits
const int modulenum = (callno & 0xFF000) >> 12;
double total = time_now_d() - start;
if (total >= hleFlipTime)
total -= hleFlipTime;
_dbg_assert_msg_(total >= 0.0, "Time spent in syscall became negative");
hleFlipTime = 0.0;
updateSyscallStats(modulenum, funcnum, total);
UpdateSyscallStats(modulenum, funcnum, total);
}
}
void CallSyscall(MIPSOpcode op) {
CallSyscallWithPC(op, 0);
}
void CallSyscallUnresolvedAtPC(u32 pc) {
LogBadSyscallAtPC(pc, false);
}
void hlePushFuncDesc(std::string_view module, std::string_view funcName) {
const HLEModule *mod = GetHLEModuleByName(module);
_dbg_assert_(mod != nullptr);
+5 -3
View File
@@ -184,14 +184,16 @@ size_t HLEFormatLogArgs(const MIPSState *mips, char *message, size_t sz, const c
u32 GetSyscallOp(std::string_view module, u32 nib);
bool WriteHLESyscall(std::string_view module, u32 nib, u32 address);
void CallSyscall(MIPSOpcode op);
void CallSyscallWithPC(MIPSOpcode op, u32 pc); // better diagnostics
void CallSyscallUnresolvedAtPC(u32 pc);
void WriteFuncStub(u32 stubAddr, u32 symAddr);
void WriteFuncMissingStub(u32 stubAddr, u32 nid);
void HLEReturnFromMipsCall();
const HLEFunction *GetSyscallFuncPointer(MIPSOpcode op);
// For jit, takes arg: const HLEFunction *
void *GetQuickSyscallFunc(MIPSOpcode op);
const HLEFunction *GetSyscallFunctionData(MIPSOpcode op, u32 pcForDiagnostics);
// For jit, the returned function takes the arg: const HLEFunction *
void *GetQuickSyscallFunc(const HLEFunction *info, MIPSOpcode op);
void hleDoLogInternal(Log t, LogLevel level, u64 res, const char *file, int line, const char *reportTag, const char *reason, const char *formatted_reason);
+5 -5
View File
@@ -507,7 +507,7 @@ static void DoFrameIdleTiming() {
#endif
}
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
DisplayNotifySleep(time_now_d() - before);
}
}
@@ -720,7 +720,7 @@ void __DisplayFlip(int cyclesLate) {
CoreTiming::ScheduleEvent(0 - cyclesLate, afterFlipEvent, 0);
numVBlanksSinceFlip = 0;
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
// Track how long we sleep (whether vsync or sleep_ms.)
DisplayNotifySleep(time_now_d() - frameSleepStart, frameSleepPos);
}
@@ -786,7 +786,7 @@ void hleLagSync(u64 userdata, int cyclesLate) {
const int over = (int)((now - goal) * 1000000);
ScheduleLagSync(over - emuOver);
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
DisplayNotifySleep(now - before);
}
}
@@ -846,9 +846,9 @@ void __DisplaySetFramebuf(u32 topaddr, int linesize, int pixelFormat, int sync)
// Doing it in non-buffered though creates problems (black screen) on occasion though
// so let's not.
if (!flippedThisFrame && !g_Config.bSkipBufferEffects) {
double before_flip = time_now_d();
const double before_flip = time_now_d();
__DisplayFlip(0);
double after_flip = time_now_d();
const double after_flip = time_now_d();
// Ignore for debug stats.
hleSetFlipTime(after_flip - before_flip);
}
+2 -2
View File
@@ -780,7 +780,7 @@ void UnexportFuncSymbol(const FuncSymbolExport &func) {
}
}
// Used to add detail to the "Unknown syscall" log in HLE.cpp's GetSyscallFuncPointer - a call
// Used to add detail to the "Unknown syscall" log in HLE.cpp's GetSyscallFunctionData - a call
// through a still-unresolved import ends up as a generic "invalid syscall" opcode that no
// longer carries the original module name/NID, but the (fixed, unique) address of the syscall
// instruction itself does - it's exactly the stubAddr every pending FuncSymbolImport recorded
@@ -793,7 +793,7 @@ bool KernelFindImportByStubAddr(u32 stubAddr, std::string *importModuleName, u32
continue;
}
for (const auto &func : module->importedFuncs) {
if (func.stubAddr == stubAddr) {
if (Memory::AddressesEqualAfterMask(func.stubAddr, stubAddr)) {
*importModuleName = func.moduleName;
*nid = func.nid;
*importingModuleName = module->GetName();
+1 -1
View File
@@ -95,7 +95,7 @@ static void CalculateFPS() {
}
}
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
frameTimeHistory[frameTimeHistoryPos++] = (float)(now - lastFrameTimeHistory);
lastFrameTimeHistory = now;
frameTimeHistoryPos = frameTimeHistoryPos % frameTimeHistorySize;
+14 -11
View File
@@ -620,17 +620,20 @@ void ArmJit::Comp_Syscall(MIPSOpcode op)
QuickCallFunction(R1, (void *)&CallSyscall);
#else
// Skip the CallSyscall where possible.
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc)
{
gpr.SetRegImm(R0, (u32)(intptr_t)GetSyscallFuncPointer(op));
// Already flushed, so R1 is safe.
QuickCallFunction(R1, quickFunc);
}
else
{
gpr.SetRegImm(R0, op.encoding);
QuickCallFunction(R1, (void *)&CallSyscall);
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
gpr.SetRegImm(R0, (uintptr_t)func);
// Already flushed, so R1 is safe.
QuickCallFunction(R1, quickFunc);
} else {
gpr.SetRegImm(R0, op.encoding);
QuickCallFunction(R1, (void *)&CallSyscall);
}
} else {
gpr.SetRegImm(R0, js.compilerPC);
QuickCallFunction(R1, (void *)&CallSyscallUnresolvedAtPC);
}
#endif
ApplyRoundingMode();
+13 -8
View File
@@ -593,7 +593,6 @@ void Arm64Jit::Comp_JumpReg(MIPSOpcode op)
WriteExitDestInR(destReg);
js.compiling = false;
}
void Arm64Jit::Comp_Syscall(MIPSOpcode op)
{
@@ -636,14 +635,20 @@ void Arm64Jit::Comp_Syscall(MIPSOpcode op)
QuickCallFunction(X1, (void *)&CallSyscall);
#else
// Skip the CallSyscall where possible.
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc) {
MOVI2R(X0, (uintptr_t)GetSyscallFuncPointer(op));
// Already flushed, so X1 is safe.
QuickCallFunction(X1, quickFunc);
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
MOVI2R(X0, (uintptr_t)func);
// Already flushed, so X1 is safe.
QuickCallFunction(X1, quickFunc);
} else {
MOVI2R(W0, op.encoding);
QuickCallFunction(X1, (void *)&CallSyscall);
}
} else {
MOVI2R(W0, op.encoding);
QuickCallFunction(X1, (void *)&CallSyscall);
MOVI2R(W0, js.compilerPC);
QuickCallFunction(X1, (void *)&CallSyscallUnresolvedAtPC);
}
#endif
LoadStaticRegisters();
+23 -6
View File
@@ -219,13 +219,20 @@ void Arm64JitBackend::CompIR_System(IRInst inst) {
// Skip the CallSyscall where possible.
{
MIPSOpcode op(inst.constant);
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc) {
MOVP2R(X0, GetSyscallFuncPointer(op));
QuickCallFunction(SCRATCH2_64, (const u8 *)quickFunc);
const HLEFunction *func = GetSyscallFunctionData(op, 0);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
MOVP2R(X0, func);
QuickCallFunction(SCRATCH2_64, (const u8 *)quickFunc);
} else {
MOVI2R(W0, inst.constant);
QuickCallFunction(SCRATCH2_64, &CallSyscall);
}
} else {
MOVI2R(W0, inst.constant);
QuickCallFunction(SCRATCH2_64, &CallSyscall);
// Shouldn't get here.
MOVI2R(W0, 0);
QuickCallFunction(SCRATCH2_64, &CallSyscallUnresolvedAtPC);
}
}
#endif
@@ -235,6 +242,16 @@ void Arm64JitBackend::CompIR_System(IRInst inst) {
// This is always followed by an ExitToPC, where we check coreState.
break;
case IROp::SyscallUnresolved:
FlushAll();
SaveStaticRegisters();
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
MOVI2R(W0, inst.constant);
QuickCallFunction(SCRATCH2_64, &CallSyscallUnresolvedAtPC);
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
LoadStaticRegisters();
break;
case IROp::CallReplacement:
FlushAll();
SaveStaticRegisters();
+1
View File
@@ -65,6 +65,7 @@ static void NoBlockExits() {
_assert_msg_(false, "Never exited block, invalid IR?");
}
// TODO: Much of this function should be merged with the same function for the other backends.
bool Arm64JitBackend::CompileBlock(IRBlockCache *irBlockCache, int block_num) {
if (GetSpaceLeft() < 0x800)
return false;
+1 -1
View File
@@ -77,7 +77,7 @@ static int IRReadsFromList(const IRInstMeta &inst, IRReg regs[4], char type) {
if ((inst.m.flags & (IRFLAG_SRC3 | IRFLAG_SRC3DST)) != 0 && inst.m.types[0] == type)
regs[c++] = inst.src3;
if (inst.op == IROp::Interpret || inst.op == IROp::CallReplacement || inst.op == IROp::Syscall || inst.op == IROp::Break)
if (inst.op == IROp::Interpret || inst.op == IROp::CallReplacement || inst.op == IROp::Syscall || inst.op == IROp::SyscallUnresolved ||inst.op == IROp::Break)
return -1;
if (inst.op == IROp::Breakpoint || inst.op == IROp::MemoryCheck)
return -1;
+11 -11
View File
@@ -61,19 +61,19 @@ void IRFrontend::Comp_IType(MIPSOpcode op) {
switch (op >> 26) {
case 8: // same as addiu?
case 9: // R(rt) = R(rs) + simm; break; //addiu
ir.Write(IROp::AddConst, rt, rs, ir.AddConstant(simm));
ir.Write(IROp::AddConst, rt, rs, 0, (u32)simm);
break;
case 12: ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(uimm)); break;
case 13: ir.Write(IROp::OrConst, rt, rs, ir.AddConstant(uimm)); break;
case 14: ir.Write(IROp::XorConst, rt, rs, ir.AddConstant(uimm)); break;
case 12: ir.Write(IROp::AndConst, rt, rs, 0, uimm); break;
case 13: ir.Write(IROp::OrConst, rt, rs, 0, uimm); break;
case 14: ir.Write(IROp::XorConst, rt, rs, 0, uimm); break;
case 10: // R(rt) = (s32)R(rs) < simm; break; //slti
ir.Write(IROp::SltConst, rt, rs, ir.AddConstant(simm));
ir.Write(IROp::SltConst, rt, rs, 0, (u32)simm);
break;
case 11: // R(rt) = R(rs) < suimm; break; //sltiu
ir.Write(IROp::SltUConst, rt, rs, ir.AddConstant(suimm));
ir.Write(IROp::SltUConst, rt, rs, 0, suimm);
break;
case 15: // R(rt) = uimm << 16; //lui
@@ -197,7 +197,7 @@ void IRFrontend::CompShiftVar(MIPSOpcode op, IROp shiftOp) {
// The interpreter already masks where needed, don't need to generate extra ops.
ir.Write(shiftOp, rd, rt, rs);
} else {
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(31));
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, (u32)31);
ir.Write(shiftOp, rd, rt, IRTEMP_0);
}
}
@@ -244,9 +244,9 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
case 0x0: // ext
if (pos != 0) {
ir.Write(IROp::ShrImm, rt, rs, pos);
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(mask));
ir.Write(IROp::AndConst, rt, rt, 0, mask);
} else {
ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(mask));
ir.Write(IROp::AndConst, rt, rs, 0, mask);
}
break;
@@ -259,7 +259,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
if (size != 32) {
// Need to use the sourcemask.
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(sourcemask));
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, sourcemask);
if (pos != 0) {
ir.Write(IROp::ShlImm, IRTEMP_0, IRTEMP_0, pos);
}
@@ -271,7 +271,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
ir.Write(IROp::Mov, IRTEMP_0, rs);
}
}
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(destmask));
ir.Write(IROp::AndConst, rt, rt, 0, destmask);
ir.Write(IROp::Or, rt, rt, IRTEMP_0);
}
break;
+27 -20
View File
@@ -96,11 +96,11 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs, rhs);
ir.Write(ComparisonToExit(cc), 0, lhs, rhs, ResolveNotTakenTarget(branchInfo));
// This makes the block "impure" :(
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
@@ -114,7 +114,7 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
}
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -147,11 +147,11 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs);
ir.Write(ComparisonToExit(cc), 0, lhs, 0, ResolveNotTakenTarget(branchInfo));
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
if (branchInfo.delaySlotIsBranch) {
@@ -165,7 +165,7 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
// Taken
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -225,12 +225,12 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
// Not taken
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
// Taken
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
@@ -244,7 +244,7 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
}
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -284,14 +284,14 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
int imm3 = (op >> 18) & 7;
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, ir.AddConstant(1 << imm3));
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, 0, 1 << imm3);
FlushAll();
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
@@ -306,7 +306,7 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
// Taken
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -355,11 +355,11 @@ void IRFrontend::Comp_Jump(MIPSOpcode op) {
}
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -420,7 +420,7 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
}
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
ir.Write(IROp::ExitToReg, 0, destReg, 0);
@@ -433,18 +433,25 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
void IRFrontend::Comp_Syscall(MIPSOpcode op) {
// Note: If we're in a delay slot, this is off by one compared to the interpreter.
int dcAmount = js.downcountAmount + (js.inDelaySlot ? -1 : 0);
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
// If not in a delay slot, we need to update PC.
// However we also need the PC for some diagnostics so let's just always do it.
// Not exactly a bottleneck.
if (!js.inDelaySlot) {
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC() + 4));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC() + 4);
}
FlushAll();
RestoreRoundingMode();
ir.Write(IROp::Syscall, 0, ir.AddConstant(op.encoding));
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
if (func) {
ir.Write(IROp::Syscall, 0, 0, 0, op.encoding);
} else {
ir.Write(IROp::SyscallUnresolved, 0, 0, 0, js.compilerPC);
}
ApplyRoundingMode();
ir.Write(IROp::ExitToPC);
@@ -452,7 +459,7 @@ void IRFrontend::Comp_Syscall(MIPSOpcode op) {
}
void IRFrontend::Comp_Break(MIPSOpcode op) {
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
ir.Write(IROp::Break);
js.compiling = false;
}
+4 -4
View File
@@ -83,11 +83,11 @@ void IRFrontend::Comp_FPULS(MIPSOpcode op) {
switch (op >> 26) {
case 49: // lwc1
ir.Write(IROp::LoadFloat, ft, rs, ir.AddConstant(offset));
ir.Write(IROp::LoadFloat, ft, rs, 0, offset);
break;
case 57: // swc1
ir.Write(IROp::StoreFloat, ft, rs, ir.AddConstant(offset));
ir.Write(IROp::StoreFloat, ft, rs, 0, offset);
break;
default:
@@ -207,10 +207,10 @@ void IRFrontend::Comp_mxc1(MIPSOpcode op) {
// This needs to insert fpcond.
ir.Write(IROp::FpCtrlToReg, rt);
} else if (fs == 0) {
ir.Write(IROp::SetConst, rt, ir.AddConstant(MIPSState::FCR0_VALUE));
ir.WriteSetConstant(rt, MIPSState::FCR0_VALUE);
} else {
// Unsupported regs are always 0.
ir.Write(IROp::SetConst, rt, ir.AddConstant(0));
ir.WriteSetConstant(rt, 0);
}
return;
+14 -14
View File
@@ -61,42 +61,42 @@ namespace MIPSComp {
switch (o) {
// Load
case 35:
ir.Write(IROp::Load32, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32, rt, rs, 0, offset);
break;
case 37:
ir.Write(IROp::Load16, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load16, rt, rs, 0, offset);
break;
case 33:
ir.Write(IROp::Load16Ext, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load16Ext, rt, rs, 0, offset);
break;
case 36:
ir.Write(IROp::Load8, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load8, rt, rs, 0, offset);
break;
case 32:
ir.Write(IROp::Load8Ext, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load8Ext, rt, rs, 0, offset);
break;
// Store
case 43:
ir.Write(IROp::Store32, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32, rt, rs, 0, offset);
break;
case 41:
ir.Write(IROp::Store16, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store16, rt, rs, 0, offset);
break;
case 40:
ir.Write(IROp::Store8, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store8, rt, rs, 0, offset);
break;
case 34: //lwl
ir.Write(IROp::Load32Left, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32Left, rt, rs, 0, offset);
break;
case 38: //lwr
ir.Write(IROp::Load32Right, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32Right, rt, rs, 0, offset);
break;
case 42: //swl
ir.Write(IROp::Store32Left, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32Left, rt, rs, 0, offset);
break;
case 46: //swr
ir.Write(IROp::Store32Right, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32Right, rt, rs, 0, offset);
break;
default:
@@ -117,11 +117,11 @@ namespace MIPSComp {
switch (op >> 26) {
case 48: // ll
ir.Write(IROp::Load32Linked, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32Linked, rt, rs, 0, offset);
break;
case 56: // sc
ir.Write(IROp::Store32Conditional, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32Conditional, rt, rs, 0, offset);
break;
default:
+41 -40
View File
@@ -256,11 +256,11 @@ namespace MIPSComp {
}
// Nope, it has something else going on.
zeroedLanes = -1;
zeroedLanes = (u32)-1;
break;
}
if (zeroedLanes != -1) {
if (zeroedLanes != (u32)-1) {
InitRegs(vregs, tempReg);
ir.Write(IROp::Vec4Init, vregs[0], (int)Vec4Init::AllZERO);
ir.Write(IROp::Vec4Blend, vregs[0], origV[0], vregs[0], zeroedLanes);
@@ -284,8 +284,9 @@ namespace MIPSComp {
if (!constants) {
if (regnum >= n) {
// Depends on the op, but often zero.
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(vregs[i], 0.0f);
} else if (abs) {
// Could have a FNAbs op, but probably not worth it.
ir.Write(IROp::FAbs, vregs[i], origV[regnum]);
if (negate)
ir.Write(IROp::FNeg, vregs[i], vregs[i]);
@@ -297,9 +298,9 @@ namespace MIPSComp {
}
} else {
if (negate) {
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(-constantArray[regnum + (abs << 2)]));
ir.WriteSetConstantFloat(vregs[i], -constantArray[regnum + (abs << 2)]);
} else {
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(constantArray[regnum + (abs << 2)]));
ir.WriteSetConstantFloat(vregs[i], constantArray[regnum + (abs << 2)]);
}
}
}
@@ -401,11 +402,11 @@ namespace MIPSComp {
switch (op >> 26) {
case 50: //lv.s
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, 0, offset);
break;
case 58: //sv.s
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, 0, offset);
break;
default:
@@ -461,29 +462,29 @@ namespace MIPSComp {
switch (optype) {
case LSVType::LVQ:
if (IsVec4(V_Quad, vregs)) {
ir.Write(IROp::LoadVec4, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::LoadVec4, vregs[0], rs, 0, imm);
} else {
// Let's not even bother with "vertical" loads for now.
if (!g_Config.bFastMemory)
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 0, (u32)imm);
ir.Write(IROp::LoadFloat, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::LoadFloat, vregs[1], rs, ir.AddConstant(imm + 4));
ir.Write(IROp::LoadFloat, vregs[2], rs, ir.AddConstant(imm + 8));
ir.Write(IROp::LoadFloat, vregs[3], rs, ir.AddConstant(imm + 12));
ir.Write(IROp::LoadFloat, vregs[0], rs, 0, imm);
ir.Write(IROp::LoadFloat, vregs[1], rs, 0, imm + 4);
ir.Write(IROp::LoadFloat, vregs[2], rs, 0, imm + 8);
ir.Write(IROp::LoadFloat, vregs[3], rs, 0, imm + 12);
}
break;
case LSVType::SVQ:
if (IsVec4(V_Quad, vregs)) {
ir.Write(IROp::StoreVec4, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::StoreVec4, vregs[0], rs, 0, imm);
} else {
// Let's not even bother with "vertical" stores for now.
if (!g_Config.bFastMemory)
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 1, (u32)imm);
ir.Write(IROp::StoreFloat, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::StoreFloat, vregs[1], rs, ir.AddConstant(imm + 4));
ir.Write(IROp::StoreFloat, vregs[2], rs, ir.AddConstant(imm + 8));
ir.Write(IROp::StoreFloat, vregs[3], rs, ir.AddConstant(imm + 12));
ir.Write(IROp::StoreFloat, vregs[0], rs, 0, imm);
ir.Write(IROp::StoreFloat, vregs[1], rs, 0, imm + 4);
ir.Write(IROp::StoreFloat, vregs[2], rs, 0, imm + 8);
ir.Write(IROp::StoreFloat, vregs[3], rs, 0, imm + 12);
}
break;
@@ -521,7 +522,7 @@ namespace MIPSComp {
ir.Write(IROp::Vec4Init, dregs[0], (int)(type == 6 ? Vec4Init::AllZERO : Vec4Init::AllONE));
} else {
for (int i = 0; i < n; i++) {
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(type == 6 ? 0.0f : 1.0f));
ir.WriteSetConstantFloat(dregs[i], type == 6 ? 0.0f : 1.0f);
}
}
ApplyPrefixD(dregs, sz, vd);
@@ -549,14 +550,14 @@ namespace MIPSComp {
} else {
switch (sz) {
case V_Pair:
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 1) == 0 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 1) == 1 ? 1.0f : 0.0f));
ir.WriteSetConstantFloat(dregs[0], (vd & 1) == 0 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[1], (vd & 1) == 1 ? 1.0f : 0.0f);
break;
case V_Quad:
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 3) == 0 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 3) == 1 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[2], ir.AddConstantFloat((vd & 3) == 2 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[3], ir.AddConstantFloat((vd & 3) == 3 ? 1.0f : 0.0f));
ir.WriteSetConstantFloat(dregs[0], (vd & 3) == 0 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[1], (vd & 3) == 1 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[2], (vd & 3) == 2 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[3], (vd & 3) == 3 ? 1.0f : 0.0f);
break;
default:
INVALIDOP;
@@ -594,19 +595,19 @@ namespace MIPSComp {
switch ((op >> 16) & 0xF) {
case 3: // vmidt
if (x == 0 && y == 0)
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
else if (x == y)
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
else
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
break;
case 6: // vmzero
// Likely to be fast.
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
break;
case 7: // vmone
if (x == 0 && y == 0)
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
else
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
break;
@@ -707,7 +708,7 @@ namespace MIPSComp {
GetVectorRegsPrefixD(dregs, V_Single, _VD);
// We have to start at +0.000 in case any values are -0.000.
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(IRVTEMP_0, 0.0f);
for (int i = 0; i < n; ++i) {
ir.Write(IROp::FAdd, IRVTEMP_0, IRVTEMP_0, sregs[i]);
}
@@ -717,7 +718,7 @@ namespace MIPSComp {
ir.Write(IROp::FMov, dregs[0], IRVTEMP_0);
break;
case 7: // vavg
ir.Write(IROp::SetConstF, IRVTEMP_0 + 1, ir.AddConstantFloat(vavg_table[n - 1]));
ir.WriteSetConstantFloat(IRVTEMP_0 + 1, vavg_table[n - 1]);
ir.Write(IROp::FMul, dregs[0], IRVTEMP_0, IRVTEMP_0 + 1);
break;
}
@@ -939,7 +940,7 @@ namespace MIPSComp {
case VecDo3Op::VSGE: // vsge
ir.Write(IROp::FCmp, (int)IRFpCompareMode::LessUnordered, sregs[i], tregs[i]);
ir.Write(IROp::FpCondToReg, IRTEMP_1);
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, ir.AddConstant(1));
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, 0, 1);
ir.Write(IROp::FMovFromGPR, tempregs[i], IRTEMP_1);
ir.Write(IROp::FCvtSW, tempregs[i], tempregs[i]);
break;
@@ -1266,7 +1267,7 @@ namespace MIPSComp {
u32 mask;
if (GetVFPUCtrlMask(imm - 128, &mask)) {
if (mask != 0xFFFFFFFF) {
ir.Write(IROp::AndConst, IRTEMP_0, rt, ir.AddConstant(mask));
ir.Write(IROp::AndConst, IRTEMP_0, rt, 0, mask);
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, IRTEMP_0);
} else {
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, rt);
@@ -1322,7 +1323,7 @@ namespace MIPSComp {
if (GetVFPUCtrlMask(imm, &mask)) {
if (mask != 0xFFFFFFFF) {
ir.Write(IROp::FMovToGPR, IRTEMP_0, vfpuBase + voffset[imm]);
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, ir.AddConstant(mask));
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, 0, mask);
ir.Write(IROp::SetCtrlVFPUReg, imm, IRTEMP_0);
} else {
ir.Write(IROp::SetCtrlVFPUFReg, imm, vfpuBase + voffset[vs]);
@@ -2146,7 +2147,7 @@ namespace MIPSComp {
s32 imm = SignExtend16ToS32(op);
u8 dreg;
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat((float)imm));
ir.WriteSetConstantFloat(dreg, (float)imm);
ApplyPrefixD(&dreg, V_Single, _VT);
}
@@ -2164,7 +2165,7 @@ namespace MIPSComp {
u8 dreg;
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat(fval.f));
ir.WriteSetConstantFloat(dreg, fval.f);
ApplyPrefixD(&dreg, V_Single, _VT);
}
@@ -2186,17 +2187,17 @@ namespace MIPSComp {
GetVectorRegsPrefixD(dregs, sz, vd);
if (IsVec4(sz, dregs)) {
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
ir.Write(IROp::Vec4Shuffle, dregs[0], IRVTEMP_0, 0);
} else if (IsVec3of4(sz, dregs) && opts.preferVec4) {
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
ir.Write(IROp::Vec4Shuffle, IRVTEMP_0, IRVTEMP_0, 0);
ir.Write(IROp::Vec4Blend, dregs[0], dregs[0], IRVTEMP_0, 0x7);
} else {
for (int i = 0; i < n; i++) {
// Most of the time, materializing a float is slower than copying from another float.
if (i == 0)
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(cst_constants[conNum]));
ir.WriteSetConstantFloat(dregs[i], cst_constants[conNum]);
else
ir.Write(IROp::FMov, dregs[i], dregs[0]);
}
@@ -2253,7 +2254,7 @@ namespace MIPSComp {
for (int i = 0; i < n; i++) {
switch (d[i]) {
case '0':
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(0.0f));
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f);
break;
case 's':
if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) {
@@ -2271,7 +2272,7 @@ namespace MIPSComp {
else if (dregs[sineLane] == sreg[0])
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
else
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(1.0f));
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
break;
}
}
+16 -15
View File
@@ -76,17 +76,17 @@ void IRFrontend::FlushPrefixV() {
}
if ((js.prefixSFlag & JitState::PREFIX_DIRTY) != 0) {
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, ir.AddConstant(js.prefixS));
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, 0, 0, js.prefixS);
js.prefixSFlag = (JitState::PrefixState) (js.prefixSFlag & ~JitState::PREFIX_DIRTY);
}
if ((js.prefixTFlag & JitState::PREFIX_DIRTY) != 0) {
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, ir.AddConstant(js.prefixT));
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, 0, 0, js.prefixT);
js.prefixTFlag = (JitState::PrefixState) (js.prefixTFlag & ~JitState::PREFIX_DIRTY);
}
if ((js.prefixDFlag & JitState::PREFIX_DIRTY) != 0) {
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, ir.AddConstant(js.prefixD));
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, 0, 0, js.prefixD);
js.prefixDFlag = (JitState::PrefixState) (js.prefixDFlag & ~JitState::PREFIX_DIRTY);
}
@@ -164,8 +164,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
} else if (entry->replaceFunc) {
FlushAll();
RestoreRoundingMode();
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::CallReplacement, IRTEMP_0, ir.AddConstant(index));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
ir.Write(IROp::CallReplacement, IRTEMP_0, 0, 0, index);
if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) {
// Compile the original instruction at this address. We ignore cycles for hooks.
@@ -175,8 +175,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
ApplyRoundingMode();
// If IRTEMP_0 was set to 1, it means the replacement needs to run again (sliced.)
// This is necessary for replacements that take a lot of cycles.
ir.Write(IROp::Downcount, 0, ir.AddConstant(js.downcountAmount));
ir.Write(IROp::ExitToConstIfNeq, ir.AddConstant(GetCompilerPC()), IRTEMP_0, MIPS_REG_ZERO);
ir.Write(IROp::Downcount, 0, 0, 0, js.downcountAmount);
ir.Write(IROp::ExitToConstIfNeq, 0, IRTEMP_0, MIPS_REG_ZERO, GetCompilerPC());
ir.Write(IROp::ExitToReg, 0, MIPS_REG_RA, 0);
js.compiling = false;
}
@@ -187,7 +187,7 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
void IRFrontend::Comp_Generic(MIPSOpcode op) {
FlushAll();
ir.Write(IROp::Interpret, 0, ir.AddConstant(op.encoding));
ir.Write(IROp::Interpret, 0, 0, 0, op.encoding);
const MIPSInfo info = MIPSGetInfo(op);
if ((info & IS_VFPU) != 0 && (info & VFPU_NO_PREFIX) == 0) {
// If it does eat them, it'll happen in MIPSCompileOp().
@@ -359,7 +359,7 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
FlushAll();
// Can't skip this even at the start of a block, might impact block linking.
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
RestoreRoundingMode();
// At this point, downcount HAS the delay slot, but not the instruction itself.
@@ -374,11 +374,12 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
}
}
int downcountAmount = js.downcountAmount + downcountOffset;
if (downcountAmount != 0)
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
if (downcountAmount != 0) {
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
}
// Note that this means downcount can't be metadata on the block.
js.downcountAmount = -downcountOffset;
ir.Write(IROp::Breakpoint, 0, ir.AddConstant(addr));
ir.Write(IROp::Breakpoint, 0, 0, 0, addr);
ApplyRoundingMode();
js.hadBreakpoints = true;
@@ -390,7 +391,7 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
FlushAll();
// Can't skip this even at the start of a block, might impact block linking.
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
RestoreRoundingMode();
// At this point, downcount HAS the delay slot, but not the instruction itself.
@@ -407,10 +408,10 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
}
int downcountAmount = js.downcountAmount + downcountOffset;
if (downcountAmount != 0)
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
// Note that this means downcount can't be metadata on the block.
js.downcountAmount = -downcountOffset;
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, ir.AddConstant(offset));
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, 0, offset);
ApplyRoundingMode();
js.hadBreakpoints = true;
+18 -13
View File
@@ -176,6 +176,7 @@ static const IRMeta irMeta[] = {
{ IROp::ExitToConstIfLtZ, "ExitIfLtZ", "CG", IRFLAG_EXIT },
{ IROp::ExitToReg, "ExitToReg", "_G", IRFLAG_EXIT },
{ IROp::Syscall, "Syscall", "_C", IRFLAG_EXIT },
{ IROp::SyscallUnresolved, "SyscallUnresolved", "C", IRFLAG_EXIT },
{ IROp::Break, "Break", "", IRFLAG_EXIT },
{ IROp::SetPC, "SetPC", "_G" },
{ IROp::SetPCConst, "SetPC", "_C" },
@@ -206,31 +207,35 @@ void InitIR() {
}
}
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2) {
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2, u32 constant) {
IRInst inst;
inst.op = op;
inst.dest = dst;
inst.src1 = src1;
inst.src2 = src2;
inst.constant = nextConst_;
inst.constant = constant;
insts_.push_back(inst);
}
nextConst_ = 0;
void IRWriter::WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant) {
u32 constant;
memcpy(&constant, &fconstant, sizeof(u32));
IRInst inst;
inst.op = op;
inst.dest = dst;
inst.src1 = src1;
inst.src2 = src2;
inst.constant = constant;
insts_.push_back(inst);
}
void IRWriter::WriteSetConstant(u8 dst, u32 value) {
Write(IROp::SetConst, dst, AddConstant(value));
Write(IROp::SetConst, dst, 0, 0, value);
}
int IRWriter::AddConstant(u32 value) {
nextConst_ = value;
return 255;
}
int IRWriter::AddConstantFloat(float value) {
u32 val;
memcpy(&val, &value, 4);
return AddConstant(val);
void IRWriter::WriteSetConstantFloat(u8 dst, float value) {
WriteFC(IROp::SetConstF, dst, 0, 0, value);
}
void IRWriter::ReplaceConstant(size_t instNumber, u32 newConstant) {
+7 -11
View File
@@ -222,7 +222,8 @@ enum class IROp : uint8_t {
ExitToConstIfFpFalse,
ExitToPC, // Used after a syscall to give us a way to do things before returning.
Syscall,
Syscall, // Needs the syscall instruction work in the constant.
SyscallUnresolved, // Used when the syscall is not resolved at compile time. PC in the constant.
SetPC, // hack to make syscall returns work
SetPCConst, // hack to make replacement know PC
CallReplacement,
@@ -384,18 +385,14 @@ public:
return *this;
}
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0);
void Write(IROp op, IRReg dst, IRReg src1, IRReg src2, uint32_t c) {
AddConstant(c);
Write(op, dst, src1, src2);
}
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0, u32 constant = 0);
void WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant);
void WriteSetConstant(u8 dst, u32 value);
void WriteSetConstantFloat(u8 dst, float value);
void Write(IRInst inst) {
insts_.push_back(inst);
}
void WriteSetConstant(u8 dst, u32 value);
int AddConstant(u32 value);
int AddConstantFloat(float value);
void Reserve(size_t s) {
insts_.reserve(s);
@@ -409,7 +406,6 @@ public:
private:
std::vector<IRInst> insts_;
u32 nextConst_ = 0;
};
struct IROptions {
+16 -2
View File
@@ -1203,10 +1203,24 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
case IROp::Syscall:
// IROp::SetPC was (hopefully) executed before.
{
// If we get here, the syscall is valid.
MIPSOpcode op(inst->constant);
CallSyscall(op);
if (coreState != CORE_RUNNING_CPU)
if (coreState != CORE_RUNNING_CPU) {
CoreTiming::ForceCheck(mips);
}
break;
}
case IROp::SyscallUnresolved:
{
// If we get here, the syscall is invalid.
u32 pc = inst->constant;
CallSyscallUnresolvedAtPC(pc);
if (coreState != CORE_RUNNING_CPU) {
// hm, what's this for?
CoreTiming::ForceCheck(mips);
}
break;
}
@@ -1300,7 +1314,7 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
}
break;
case IROp::Nop: // TODO: This shouldn't crash, but for now we should not emit nops, so...
case IROp::Nop: // Unused, add a break if we start using it to avoid UNREACHABLE.
case IROp::Bad:
default:
// Unimplemented IR op. Bad. We define it as unreachable so the compiler can optimize better (remove the range check).
+1
View File
@@ -440,6 +440,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) {
break;
case IROp::Syscall:
case IROp::SyscallUnresolved:
case IROp::CallReplacement:
case IROp::Break:
CompIR_System(inst);
+40 -38
View File
@@ -258,21 +258,21 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
if (opts.unalignedLoadStore) {
// Write out one unaligned op.
out.Write(replaceOp, inst.dest, inst.src1, out.AddConstant(inst.constant + replaceOff));
out.Write(replaceOp, inst.dest, inst.src1, 0, inst.constant + replaceOff);
} else if (replaceOp == IROp::Load32) {
// We can still combine to a simpler set of two loads.
// We start by isolating the address and shift amount.
// IRTEMP_LR_ADDR = rs + imm
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant + replaceOff));
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant + replaceOff);
// IRTEMP_LR_SHIFT = (addr & 3) * 8
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
// IRTEMP_LR_ADDR = addr & 0xfffffffc
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
// IRTEMP_LR_VALUE = low_word, dest = high_word
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, out.AddConstant(0));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(4));
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, 0, 0);
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 4);
// Now we just need to adjust and combine dest and IRTEMP_LR_VALUE.
// inst.dest >>= shift (putting its bits in the right spot.)
@@ -281,7 +281,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::ShlImm, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, 8);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
// IRTEMP_LR_VALUE <<= (24 - shift)
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
@@ -297,18 +297,18 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
auto addCommonProlog = [&]() {
// IRTEMP_LR_ADDR = rs + imm
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant));
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant);
// IRTEMP_LR_SHIFT = (addr & 3) * 8
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
// IRTEMP_LR_ADDR = addr & 0xfffffffc (for stores, later)
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR)
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(0));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 0);
};
auto addCommonStore = [&](int off = 0) {
// RAM(IRTEMP_LR_ADDR) = IRTEMP_LR_VALUE
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(off));
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, off);
};
switch (inst.op) {
@@ -327,7 +327,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::And, inst.dest, inst.dest, IRTEMP_LR_MASK);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
// IRTEMP_LR_VALUE <<= (24 - shift)
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
// dest |= IRTEMP_LR_VALUE
@@ -336,7 +336,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
bool src1Dirty = inst.dest == inst.src1;
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, nextOp().constant - inst.constant);
// dest &= IRTEMP_LR_MASK
out.Write(IROp::And, nextOp().dest, nextOp().dest, IRTEMP_LR_MASK);
@@ -362,7 +362,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::Shr, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
// dest &= (0xffffff00 << (24 - shift))
// Alternatively, could shift to a wall and back (but would require two shifts each way.)
out.WriteSetConstant(IRTEMP_LR_MASK, 0xffffff00);
@@ -377,12 +377,12 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
bool src1Dirty = inst.dest == inst.src1;
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, (u32)(nextOp().constant - inst.constant));
if (shiftNeedsReverse) {
// IRTEMP_LR_SHIFT = shift again
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
shiftNeedsReverse = false;
}
// IRTEMP_LR_VALUE >>= IRTEMP_LR_SHIFT
@@ -411,7 +411,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
// IRTEMP_LR_VALUE |= src3 >> (24 - shift)
out.Write(IROp::Shr, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
@@ -429,11 +429,11 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
// IRTEMP_LR_VALUE &= 0x00ffffff << (24 - shift)
out.WriteSetConstant(IRTEMP_LR_MASK, 0x00ffffff);
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
out.Write(IROp::Shr, IRTEMP_LR_MASK, IRTEMP_LR_MASK, IRTEMP_LR_SHIFT);
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
// IRTEMP_LR_VALUE |= src3 << shift
out.Write(IROp::Shl, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
@@ -508,7 +508,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
if (inst.dest != inst.src1)
out.Write(IROp::Mov, inst.dest, inst.src1);
} else {
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, out.AddConstant(imm2));
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, 0, imm2);
}
} else if (symmetric && gpr.IsImm(inst.src1)) {
const u32 imm1 = gpr.GetImm(inst.src1);
@@ -518,7 +518,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
if (inst.dest != inst.src2)
out.Write(IROp::Mov, inst.dest, inst.src2);
} else {
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, out.AddConstant(imm1));
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, 0, imm1);
}
} else {
gpr.MapDirtyInIn(inst.dest, inst.src1, inst.src2);
@@ -632,7 +632,8 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::FMovFromGPR:
if (gpr.IsImm(inst.src1)) {
out.Write(IROp::SetConstF, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
// NOTE: SetConstantFloat doesn't work here since we actually want the bits.
out.Write(IROp::SetConstF, inst.dest, 0, 0, gpr.GetImm(inst.src1));
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -661,7 +662,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::Store32Conditional:
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
gpr.MapIn(inst.dest);
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapInIn(inst.dest, inst.src1);
goto doDefault;
@@ -670,7 +671,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::StoreFloat:
case IROp::StoreVec4:
if (gpr.IsImm(inst.src1)) {
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -685,7 +686,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::Load32Linked:
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
gpr.MapDirty(inst.dest);
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapDirtyIn(inst.dest, inst.src1);
goto doDefault;
@@ -694,7 +695,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::LoadFloat:
case IROp::LoadVec4:
if (gpr.IsImm(inst.src1)) {
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -704,7 +705,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::Load32Right:
if (gpr.IsImm(inst.src1)) {
gpr.MapIn(inst.dest);
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapInIn(inst.dest, inst.src1);
goto doDefault;
@@ -716,7 +717,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::ValidateAddress32:
case IROp::ValidateAddress128:
if (gpr.IsImm(inst.src1)) {
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -729,7 +730,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::SetPC:
if (gpr.IsImm(inst.src1)) {
out.Write(IROp::SetPCConst, out.AddConstant(gpr.GetImm(inst.src1)));
out.Write(IROp::SetPCConst, 0, 0, 0, gpr.GetImm(inst.src1));
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -772,7 +773,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::SetCtrlVFPUReg:
if (gpr.IsImm(inst.src1)) {
out.Write(IROp::SetCtrlVFPU, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
out.Write(IROp::SetCtrlVFPU, inst.dest, 0, 0, gpr.GetImm(inst.src1));
} else {
gpr.MapDirtyIn(IRREG_VFPU_CTRL_BASE + inst.dest, inst.src1);
out.Write(inst);
@@ -872,7 +873,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
// Reduce bloat by skipping on fail, and const exit on pass.
if (passed) {
gpr.FlushAll();
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
skipNextExitToConst = true;
}
break;
@@ -896,7 +897,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
if (passed) {
gpr.FlushAll();
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
skipNextExitToConst = true;
}
break;
@@ -918,7 +919,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
// Prefer ExitToConst to allow block linking.
u32 dest = gpr.GetImm(inst.src1);
gpr.FlushAll();
out.Write(IROp::ExitToConst, out.AddConstant(dest));
out.Write(IROp::ExitToConst, 0, 0, 0, dest);
break;
}
gpr.FlushAll();
@@ -2040,10 +2041,11 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
} else if (inst.constant == 0xBF800000) {
out.Write(IROp::Vec4Init, temp, (int)Vec4Init::AllMinusONE);
} else {
out.Write(IROp::SetConstF, temp, out.AddConstant(inst.constant));
// NOTE: WriteSetConstantFloat doesn't work here since we actually want the bits.
out.Write(IROp::SetConstF, temp, 0, 0, inst.constant);
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
}
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
isVec4Dirty[inst.dest & ~3] = true;
continue;
}
@@ -2054,7 +2056,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
u8 blendMask = 1 << (inst.dest & 3);
out.Write(IROp::FMovFromGPR, temp, inst.src1);
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
isVec4Dirty[inst.dest & ~3] = true;
continue;
}
@@ -2065,7 +2067,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
u8 blendMask = 1 << (inst.dest & 3);
out.Write(inst.op, temp, inst.src1, inst.src2, inst.constant);
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
isVec4Dirty[inst.dest & ~3] = true;
continue;
}
+2 -1
View File
@@ -184,6 +184,7 @@ namespace MIPSInt {
}
void Int_Syscall(MIPSState *mips, MIPSOpcode op) {
const u32 syscallPC = mips->pc - 4;
// Need to pre-move PC, as CallSyscall may result in a rescheduling!
// To do this neater, we'll need a little generated kernel loop that syscall can jump to and then RFI from
// but I don't see a need to bother.
@@ -193,7 +194,7 @@ namespace MIPSInt {
mips->pc += 4;
}
mips->inDelaySlot = false;
CallSyscall(op);
CallSyscallWithPC(op, syscallPC);
}
void Int_Sync(MIPSState *mips, MIPSOpcode op) {
@@ -185,22 +185,38 @@ void LoongArch64JitBackend::CompIR_System(IRInst inst) {
// Skip the CallSyscall where possible.
{
MIPSOpcode op(inst.constant);
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc) {
LI(R4, (uintptr_t)GetSyscallFuncPointer(op));
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
const HLEFunction *func = GetSyscallFunctionData(op, 0);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
LI(R4, func);
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
} else {
LI(R4, (int32_t)inst.constant);
QuickCallFunction(&CallSyscall, SCRATCH2);
}
} else {
LI(R4, (int32_t)inst.constant);
QuickCallFunction(&CallSyscall, SCRATCH2);
// Shouldn't get here.
LI(R4, 0);
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
}
}
#endif
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
LoadStaticRegisters();
// This is always followed by an ExitToPC, where we check coreState.
break;
case IROp::SyscallUnresolved:
FlushAll();
SaveStaticRegisters();
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
LI(R4, inst.constant);
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
LoadStaticRegisters();
break;
case IROp::CallReplacement:
FlushAll();
SaveStaticRegisters();
@@ -275,4 +291,4 @@ void LoongArch64JitBackend::CompIR_ValidateAddress(IRInst inst) {
}
}
} // namespace MIPSComp
} // namespace MIPSComp
+1 -1
View File
@@ -408,4 +408,4 @@ LoongArch64Reg LoongArch64JitBackend::NormalizeR(IRReg rs, IRReg rd, LoongArch64
}
}
} // namespace MIPSComp
} // namespace MIPSComp
+5 -4
View File
@@ -1116,18 +1116,19 @@ static void RunUntilDowncountZeroFast(MIPSState *mips) {
int cycleCount = 0;
// Don't stop in a delay slot!
do {
if (!Memory::IsValid4AlignedAddress(mips->pc)) {
Core_ExecException(mips->pc, mips->pc, ExecExceptionType::JUMP);
const u32 pc = mips->pc;
if (!Memory::IsValid4AlignedAddress(pc)) {
Core_ExecException(pc, pc, ExecExceptionType::JUMP);
return;
}
const MIPSOpcode op = MIPSOpcode(Memory::ReadUnchecked_U32(mips->pc));
const MIPSOpcode op = MIPSOpcode(Memory::ReadUnchecked_U32(pc));
const bool wasInDelaySlot = mips->inDelaySlot;
const int cycles = ExecInstruction(mips, op);
if (cycles < 0) {
// Not a recognized instruction (invalid encoding, or an unimplemented kernel-mode only instruction
// with no interpreter implementation, e.g. tge/tlt/teq/).
Core_ExecException(mips->pc, mips->pc, ExecExceptionType::ILLEGAL);
Core_ExecException(pc, pc, ExecExceptionType::ILLEGAL);
return;
}
cycleCount += cycles;
+13 -6
View File
@@ -193,13 +193,20 @@ void RiscVJitBackend::CompIR_System(IRInst inst) {
// Skip the CallSyscall where possible.
{
MIPSOpcode op(inst.constant);
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc) {
LI(X10, (uintptr_t)GetSyscallFuncPointer(op));
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
const HLEFunction *func = GetSyscallFunctionData(op, 0);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
LI(X10, (uintptr_t)func);
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
} else {
LI(X10, (int32_t)inst.constant);
QuickCallFunction(&CallSyscall, SCRATCH2);
}
} else {
LI(X10, (int32_t)inst.constant);
QuickCallFunction(&CallSyscall, SCRATCH2);
// Shouldn't get here.
LI(X10, 0);
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
}
}
#endif
+11 -5
View File
@@ -681,11 +681,17 @@ void Jit::Comp_Syscall(MIPSOpcode op)
ABI_CallFunctionC(&CallSyscall, op.encoding);
#else
// Skip the CallSyscall where possible.
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc)
ABI_CallFunctionP(quickFunc, (void *)GetSyscallFuncPointer(op));
else
ABI_CallFunctionC(&CallSyscall, op.encoding);
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
ABI_CallFunctionP(quickFunc, (void *)func);
} else {
ABI_CallFunctionC(&CallSyscall, op.encoding);
}
} else {
ABI_CallFunctionC(&CallSyscallUnresolvedAtPC, js.compilerPC);
}
#endif
ApplyRoundingMode();
+20 -4
View File
@@ -211,11 +211,18 @@ void X64JitBackend::CompIR_System(IRInst inst) {
// Skip the CallSyscall where possible.
{
MIPSOpcode op(inst.constant);
void *quickFunc = GetQuickSyscallFunc(op);
if (quickFunc) {
ABI_CallFunctionP((const u8 *)quickFunc, (void *)GetSyscallFuncPointer(op));
const HLEFunction *func = GetSyscallFunctionData(op, 0);
if (func) {
void *quickFunc = GetQuickSyscallFunc(func, op);
if (quickFunc) {
ABI_CallFunctionP((const u8 *)quickFunc, (void *)func);
} else {
ABI_CallFunctionC((const u8 *)&CallSyscall, inst.constant);
}
} else {
ABI_CallFunctionC((const u8 *)&CallSyscall, inst.constant);
_dbg_assert_(false);
// Shouldn't get here, this should be resolved during ->IR compilation.
ABI_CallFunctionC((const u8 *)&CallSyscallUnresolvedAtPC, 0);
}
}
#endif
@@ -225,6 +232,15 @@ void X64JitBackend::CompIR_System(IRInst inst) {
// This is always followed by an ExitToPC, where we check coreState.
break;
case IROp::SyscallUnresolved:
FlushAll();
SaveStaticRegisters();
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
ABI_CallFunctionC((const u8 *)&CallSyscallUnresolvedAtPC, inst.constant);
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
LoadStaticRegisters();
break;
case IROp::CallReplacement:
FlushAll();
SaveStaticRegisters();
+4
View File
@@ -296,6 +296,10 @@ inline void MemcpyUnchecked(const u32 to_address, const u32 from_address, const
MemcpyUnchecked(GetPointerWriteUnchecked(to_address), from_address, len);
}
inline bool AddressesEqualAfterMask(const u32 address1, const u32 address2) {
return (address1 & 0x3FFFFFFF) == (address2 & 0x3FFFFFFF);
}
// Applies to user mode.
// Without a length, IsValidAddress is generally semi-meaningless, unless it's about a single byte access. For larger accesses, use IsValid4AlignedAddress
// etc when appropriate, or for longer sizes, use IsValidRange or IsValid4AlignedRange for example. Checking aligned-ness helps avoid the problem
+3 -3
View File
@@ -93,7 +93,7 @@ static FileLoader *g_loadedFile;
static std::mutex loadingLock;
static std::thread g_loadingThread;
bool coreCollectDebugStats = false;
bool g_coreCollectDebugStats = false;
static int coreCollectDebugStatsCounter = 0;
static volatile CPUThreadState cpuThreadState = CPU_THREAD_NOT_RUNNING;
@@ -608,8 +608,8 @@ void UpdateLoadedFile(FileLoader *fileLoader) {
void PSP_UpdateDebugStats(bool collectStats) {
bool newState = collectStats || coreCollectDebugStatsCounter > 0;
if (coreCollectDebugStats != newState) {
coreCollectDebugStats = newState;
if (g_coreCollectDebugStats != newState) {
g_coreCollectDebugStats = newState;
mipsr4k.ClearJitCache();
}
+1 -1
View File
@@ -98,7 +98,7 @@ Path GetSysDirectory(PSPDirectories directoryType);
bool CreateSysDirectories();
extern bool coreCollectDebugStats;
extern bool g_coreCollectDebugStats;
inline CoreParameter &PSP_CoreParameter() {
extern CoreParameter g_CoreParameter;
+3 -3
View File
@@ -1155,7 +1155,7 @@ void DrawEngineCommon::DepthRasterSubmitRaw(GEPrimitiveType prim, const VertexDe
return;
}
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, coreCollectDebugStats);
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, g_coreCollectDebugStats);
// Decode.
int numDecoded = 0;
@@ -1197,7 +1197,7 @@ void DrawEngineCommon::DepthRasterPredecoded(GEPrimitiveType prim, const void *i
return;
}
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, coreCollectDebugStats);
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, g_coreCollectDebugStats);
// Make sure these have already been indexed away.
_dbg_assert_(prim != GE_PRIM_TRIANGLE_STRIP && prim != GE_PRIM_TRIANGLE_FAN);
@@ -1234,7 +1234,7 @@ void DrawEngineCommon::FlushQueuedDepth() {
rasterTimeStart_ = 0.0;
}
const bool collectStats = coreCollectDebugStats;
const bool collectStats = g_coreCollectDebugStats;
const bool lowQ = g_Config.iDepthRasterMode == (int)DepthRasterMode::LOW_QUALITY;
for (const auto &draw : depthDraws_) {
int *tx = depthScreenVerts_;
+3 -3
View File
@@ -745,7 +745,7 @@ DLResult GPUCommon::ProcessDLQueue() {
}
}
TimeCollector collectStat(&gpuStats.perFrame.msProcessingDisplayLists, coreCollectDebugStats);
TimeCollector collectStat(&gpuStats.perFrame.msProcessingDisplayLists, g_coreCollectDebugStats);
auto GetNextListIndex = [&]() -> int {
if (dlQueue.empty())
@@ -873,7 +873,7 @@ DLResult GPUCommon::ProcessDLQueue() {
currentList = nullptr;
if (coreCollectDebugStats) {
if (g_coreCollectDebugStats) {
gpuStats.perFrame.otherGPUCycles += cyclesExecuted;
}
@@ -1394,7 +1394,7 @@ void GPUCommon::FastLoadBoneMatrix(u32 target) {
cyclesExecuted += 2 * 14; // one to reset the counter, 12 to load the matrix, and a return.
if (coreCollectDebugStats) {
if (g_coreCollectDebugStats) {
gpuStats.perFrame.otherGPUCycles += 2 * 14;
}
}
+8 -5
View File
@@ -563,8 +563,10 @@ void BinManager::Flush(const char *reason) {
return;
double st = 0.0;
if (coreCollectDebugStats)
const bool collectDebugStats = g_coreCollectDebugStats;
if (collectDebugStats) {
st = time_now_d();
}
Drain(true);
waitable_->Wait();
taskRanges_.clear();
@@ -584,16 +586,17 @@ void BinManager::Flush(const char *reason) {
queueRange_.x2 = 0;
queueRange_.y2 = 0;
for (auto &pending : pendingWrites_)
for (BinDirtyRange &pending : pendingWrites_) {
pending.base = 0;
}
pendingOverlap_ = false;
pendingReads_.clear();
// We'll need to set the pending writes and reads again, since we just flushed it.
dirty_ |= SoftDirty::BINNER_RANGE | SoftDirty::BINNER_OVERLAP;
if (coreCollectDebugStats) {
double et = time_now_d();
if (collectDebugStats) {
const double et = time_now_d();
flushReasonTimes_[reason] += et - st;
if (et - st > slowestFlushTime_) {
slowestFlushTime_ = et - st;
@@ -611,7 +614,7 @@ void BinManager::OptimizePendingStates(uint16_t first, uint16_t last) {
last--;
}
int count = (QUEUED_STATES + last - first) % QUEUED_STATES + 1;
const int count = (QUEUED_STATES + last - first) % QUEUED_STATES + 1;
for (int i = 0; i < count; ++i) {
size_t pos = (first + i) % QUEUED_STATES;
OptimizeRasterState(&states_[pos]);
+2 -1
View File
@@ -154,7 +154,8 @@ void DeveloperToolsScreen::CreateGeneralTab(UI::LinearLayout *list) {
core->HideChoice(1);
core->HideChoice(3);
}
// TODO: Enable "JIT using IR" on more architectures.
// TODO: Enable "JIT using IR" on more architectures. ARM32 needs more testing.
// Note also that Loongarch and RISC-V only have a jit-ir backend, and it's used for the JIT option, so the fourth option isn't shown.
#if !PPSSPP_ARCH(X86) && !PPSSPP_ARCH(AMD64) && !PPSSPP_ARCH(ARM64)
core->HideChoice(3);
#endif
@@ -1603,8 +1603,6 @@ public class PpssppActivity extends AppCompatActivity implements SensorEventList
return true;
} else if (command.equals("showKeyboard") && surfView != null) {
InputMethodManager inputMethodManager = (InputMethodManager) getSystemService(Context.INPUT_METHOD_SERVICE);
// No idea what the point of the ApplicationWindowToken is or if it
// matters where we get it from...
inputMethodManager.showSoftInput(surfView, InputMethodManager.SHOW_IMPLICIT);
return true;
} else if (command.equals("hideKeyboard") && surfView != null) {