mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #22097 from hrydgard/unresolved-syscall
Log unresolved syscalls properly
This commit is contained in:
39 files changed
+420
-294
No files matched your search
+2
-2
@@ -98,8 +98,8 @@ struct LogChannel {
|
||||
#endif
|
||||
bool enabled = true;
|
||||
|
||||
bool IsEnabled(LogLevel level) const {
|
||||
if (level > this->level || !this->enabled)
|
||||
bool IsEnabled(LogLevel logLevel) const {
|
||||
if (logLevel > level || !enabled)
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -197,17 +197,17 @@ inline __m128i _mm_mullo_epi32_SSE2(const __m128i v0, const __m128i v1) {
|
||||
inline __m128i _mm_max_epu16_SSE2(const __m128i v0, const __m128i v1) {
|
||||
return _mm_xor_si128(
|
||||
_mm_max_epi16(
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)0x8000));
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
|
||||
}
|
||||
|
||||
inline __m128i _mm_min_epu16_SSE2(const __m128i v0, const __m128i v1) {
|
||||
return _mm_xor_si128(
|
||||
_mm_min_epi16(
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)0x8000));
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
|
||||
}
|
||||
|
||||
// SSE2 replacement for half of a _mm_packus_epi32 but without the saturation.
|
||||
|
||||
@@ -34,7 +34,7 @@ inline constexpr uint32_t RoundDownToMultipleOf(uint32_t v, uint32_t multiple) {
|
||||
|
||||
// TODO: this should just use a bitscan.
|
||||
inline uint32_t log2i(uint32_t val) {
|
||||
unsigned int ret = -1;
|
||||
unsigned int ret = (unsigned int)-1;
|
||||
while (val != 0) {
|
||||
val >>= 1; ret++;
|
||||
}
|
||||
|
||||
+51
-33
@@ -78,7 +78,7 @@ static const HLEFunction *g_stack[MAX_SYSCALL_RECURSION];
|
||||
u32 g_syscallPC;
|
||||
int g_stackSize;
|
||||
|
||||
static int idleOp;
|
||||
static int g_idleOp;
|
||||
|
||||
// Split syscall support. NOTE: This needs to be saved in DoState somehow!
|
||||
static int splitSyscallEatCycles = 0;
|
||||
@@ -266,7 +266,7 @@ void HLEInit() {
|
||||
RegisterAllModules();
|
||||
g_stackSize = 0;
|
||||
delayedResultEvent = CoreTiming::RegisterEvent("HLEDelayedResult", hleDelayResultFinish);
|
||||
idleOp = GetSyscallOp("FakeSysCalls", NID_IDLE);
|
||||
g_idleOp = GetSyscallOp("FakeSysCalls", NID_IDLE);
|
||||
}
|
||||
|
||||
void HLEDoState(PointerWrap &p) {
|
||||
@@ -801,10 +801,10 @@ void hleFinishSyscallAfterGe() {
|
||||
hleFinishSyscall(nullptr);
|
||||
}
|
||||
|
||||
static void updateSyscallStats(int modulenum, int funcnum, double total) {
|
||||
static void UpdateSyscallStats(int modulenum, int funcnum, double total) {
|
||||
const char *name = moduleDB[modulenum].funcTable[funcnum].name;
|
||||
// Ignore this one, especially for msInSyscalls (although that ignores CoreTiming events.)
|
||||
if (0 == strcmp(name, "_sceKernelIdle"))
|
||||
if (equals(name, "_sceKernelIdle"))
|
||||
return;
|
||||
|
||||
if (total > kernelStats.slowestSyscallTime) {
|
||||
@@ -822,7 +822,7 @@ static void updateSyscallStats(int modulenum, int funcnum, double total) {
|
||||
kernelStats.summedSlowestSyscallName = name;
|
||||
}
|
||||
} else {
|
||||
double newTotal = kernelStats.summedMsInSyscalls[statCall] += total;
|
||||
const double newTotal = kernelStats.summedMsInSyscalls[statCall] += total;
|
||||
if (newTotal > kernelStats.summedSlowestSyscallTime) {
|
||||
kernelStats.summedSlowestSyscallTime = newTotal;
|
||||
kernelStats.summedSlowestSyscallName = name;
|
||||
@@ -895,66 +895,76 @@ static void CallSyscallWithoutFlags(const HLEFunction *info) {
|
||||
g_stackSize = 0;
|
||||
}
|
||||
|
||||
const HLEFunction *GetSyscallFuncPointer(MIPSOpcode op) {
|
||||
void LogBadSyscallAtPC(u32 pc, bool compilePhase) {
|
||||
const LogLevel level = compilePhase ? LogLevel::LWARNING : LogLevel::LERROR;
|
||||
const char *phase = compilePhase ? "compile" : "run";
|
||||
std::string importModuleName, importingModuleName;
|
||||
u32 nid = 0;
|
||||
if (pc && KernelFindImportByStubAddr(pc, &importModuleName, &nid, &importingModuleName)) {
|
||||
const char *funcName = GetHLEFuncName(importModuleName, nid);
|
||||
GENERIC_LOG(Log::HLE, level, "Unknown syscall (%s) at %08x: unresolved import %s/%08x (%s), called from '%s'", phase, pc, importModuleName.c_str(), nid, funcName ? funcName : "(unknown)", importingModuleName.c_str());
|
||||
} else {
|
||||
char buffer[256];
|
||||
DescribeAddress(currentDebugMIPS, pc, buffer, sizeof(buffer));
|
||||
GENERIC_LOG(Log::HLE, level, "Unknown syscall (%s) at %08x (%s): was unable to determine more information", phase, pc, buffer);
|
||||
}
|
||||
}
|
||||
|
||||
const HLEFunction *GetSyscallFunctionData(MIPSOpcode op, u32 pcForDiagnostics) {
|
||||
u32 callno = (op >> 6) & 0xFFFFF; //20 bits
|
||||
int funcnum = callno & 0xFFF;
|
||||
int modulenum = (callno & 0xFF000) >> 12;
|
||||
if (funcnum == 0xfff) {
|
||||
std::string_view modName = modulenum >= (int)moduleDB.size() ? "(unknown)" : moduleDB[modulenum].name;
|
||||
// This is what a still-unresolved import looks like once written as a syscall opcode -
|
||||
// the original module name/NID aren't recoverable from the opcode itself (see
|
||||
// WriteFuncMissingStub), but the calling address is a stub we may still be tracking.
|
||||
std::string importModuleName, importingModuleName;
|
||||
u32 nid = 0;
|
||||
if (currentMIPS->pc >= 8 && KernelFindImportByStubAddr(currentMIPS->pc - 8, &importModuleName, &nid, &importingModuleName)) {
|
||||
const char *funcName = GetHLEFuncName(importModuleName, nid);
|
||||
ERROR_LOG(Log::HLE, "Unknown syscall: unresolved import %s/%08x (%s), called from '%s'", importModuleName.c_str(), nid, funcName ? funcName : "(unknown)", importingModuleName.c_str());
|
||||
} else {
|
||||
ERROR_LOG(Log::HLE, "Unknown syscall: Module: '%.*s' (module: %d func: %d)", (int)modName.size(), modName.data(), modulenum, funcnum);
|
||||
}
|
||||
return NULL;
|
||||
LogBadSyscallAtPC(pcForDiagnostics, PSP_CoreParameter().cpuCore != CPUCore::INTERPRETER);
|
||||
return nullptr;
|
||||
}
|
||||
if (modulenum >= (int)moduleDB.size()) {
|
||||
ERROR_LOG(Log::HLE, "Syscall had bad module number %d - probably executing garbage", modulenum);
|
||||
return NULL;
|
||||
return nullptr;
|
||||
}
|
||||
if (funcnum >= moduleDB[modulenum].numFunctions) {
|
||||
ERROR_LOG(Log::HLE, "Syscall had bad function number %d in module %d - probably executing garbage", funcnum, modulenum);
|
||||
return NULL;
|
||||
return nullptr;
|
||||
}
|
||||
return &moduleDB[modulenum].funcTable[funcnum];
|
||||
}
|
||||
|
||||
void *GetQuickSyscallFunc(MIPSOpcode op) {
|
||||
if (coreCollectDebugStats)
|
||||
void *GetQuickSyscallFunc(const HLEFunction *info, MIPSOpcode op) {
|
||||
if (g_coreCollectDebugStats)
|
||||
return nullptr;
|
||||
|
||||
const HLEFunction *info = GetSyscallFuncPointer(op);
|
||||
if (!info || !info->func)
|
||||
return nullptr;
|
||||
|
||||
VERBOSE_LOG(Log::HLE, "Compiling syscall to '%s'", info->name);
|
||||
|
||||
// TODO: Do this with a flag?
|
||||
if (op == idleOp)
|
||||
if (op == g_idleOp) {
|
||||
return (void *)info->func;
|
||||
if (info->flags != 0)
|
||||
} else if (info->flags != 0) {
|
||||
return (void *)&CallSyscallWithFlags;
|
||||
return (void *)&CallSyscallWithoutFlags;
|
||||
} else {
|
||||
return (void *)&CallSyscallWithoutFlags;
|
||||
}
|
||||
}
|
||||
|
||||
void hleSetFlipTime(double t) {
|
||||
hleFlipTime = t;
|
||||
}
|
||||
|
||||
void CallSyscall(MIPSOpcode op) {
|
||||
void CallSyscallWithPC(MIPSOpcode op, u32 pc) {
|
||||
PROFILE_THIS_SCOPE("syscall");
|
||||
double start = 0.0; // need to initialize to fix the race condition where coreCollectDebugStats is enabled in the middle of this func.
|
||||
if (coreCollectDebugStats) {
|
||||
const bool collectStats = g_coreCollectDebugStats;
|
||||
double start = 0.0;
|
||||
if (collectStats) {
|
||||
start = time_now_d();
|
||||
}
|
||||
|
||||
const HLEFunction *info = GetSyscallFuncPointer(op);
|
||||
const HLEFunction *info = GetSyscallFunctionData(op, pc);
|
||||
if (!info) {
|
||||
// We haven't incremented the stack yet.
|
||||
RETURN(SCE_KERNEL_ERROR_LIBRARY_NOT_YET_LINKED);
|
||||
@@ -962,7 +972,7 @@ void CallSyscall(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
if (info->func) {
|
||||
if (op == idleOp)
|
||||
if (op == g_idleOp)
|
||||
info->func();
|
||||
else if (info->flags != 0)
|
||||
CallSyscallWithFlags(info);
|
||||
@@ -974,19 +984,27 @@ void CallSyscall(MIPSOpcode op) {
|
||||
ERROR_LOG_REPORT(Log::HLE, "Unimplemented HLE function %s", info->name ? info->name : "(\?\?\?)");
|
||||
}
|
||||
|
||||
if (coreCollectDebugStats) {
|
||||
u32 callno = (op >> 6) & 0xFFFFF; //20 bits
|
||||
int funcnum = callno & 0xFFF;
|
||||
int modulenum = (callno & 0xFF000) >> 12;
|
||||
if (collectStats) {
|
||||
const u32 callno = (op >> 6) & 0xFFFFF; // 20 bits
|
||||
const int funcnum = callno & 0xFFF; // 12 bits
|
||||
const int modulenum = (callno & 0xFF000) >> 12;
|
||||
double total = time_now_d() - start;
|
||||
if (total >= hleFlipTime)
|
||||
total -= hleFlipTime;
|
||||
_dbg_assert_msg_(total >= 0.0, "Time spent in syscall became negative");
|
||||
hleFlipTime = 0.0;
|
||||
updateSyscallStats(modulenum, funcnum, total);
|
||||
UpdateSyscallStats(modulenum, funcnum, total);
|
||||
}
|
||||
}
|
||||
|
||||
void CallSyscall(MIPSOpcode op) {
|
||||
CallSyscallWithPC(op, 0);
|
||||
}
|
||||
|
||||
void CallSyscallUnresolvedAtPC(u32 pc) {
|
||||
LogBadSyscallAtPC(pc, false);
|
||||
}
|
||||
|
||||
void hlePushFuncDesc(std::string_view module, std::string_view funcName) {
|
||||
const HLEModule *mod = GetHLEModuleByName(module);
|
||||
_dbg_assert_(mod != nullptr);
|
||||
|
||||
+5
-3
@@ -184,14 +184,16 @@ size_t HLEFormatLogArgs(const MIPSState *mips, char *message, size_t sz, const c
|
||||
u32 GetSyscallOp(std::string_view module, u32 nib);
|
||||
bool WriteHLESyscall(std::string_view module, u32 nib, u32 address);
|
||||
void CallSyscall(MIPSOpcode op);
|
||||
void CallSyscallWithPC(MIPSOpcode op, u32 pc); // better diagnostics
|
||||
void CallSyscallUnresolvedAtPC(u32 pc);
|
||||
void WriteFuncStub(u32 stubAddr, u32 symAddr);
|
||||
void WriteFuncMissingStub(u32 stubAddr, u32 nid);
|
||||
|
||||
void HLEReturnFromMipsCall();
|
||||
|
||||
const HLEFunction *GetSyscallFuncPointer(MIPSOpcode op);
|
||||
// For jit, takes arg: const HLEFunction *
|
||||
void *GetQuickSyscallFunc(MIPSOpcode op);
|
||||
const HLEFunction *GetSyscallFunctionData(MIPSOpcode op, u32 pcForDiagnostics);
|
||||
// For jit, the returned function takes the arg: const HLEFunction *
|
||||
void *GetQuickSyscallFunc(const HLEFunction *info, MIPSOpcode op);
|
||||
|
||||
void hleDoLogInternal(Log t, LogLevel level, u64 res, const char *file, int line, const char *reportTag, const char *reason, const char *formatted_reason);
|
||||
|
||||
|
||||
@@ -507,7 +507,7 @@ static void DoFrameIdleTiming() {
|
||||
#endif
|
||||
}
|
||||
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
|
||||
DisplayNotifySleep(time_now_d() - before);
|
||||
}
|
||||
}
|
||||
@@ -720,7 +720,7 @@ void __DisplayFlip(int cyclesLate) {
|
||||
CoreTiming::ScheduleEvent(0 - cyclesLate, afterFlipEvent, 0);
|
||||
numVBlanksSinceFlip = 0;
|
||||
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
|
||||
// Track how long we sleep (whether vsync or sleep_ms.)
|
||||
DisplayNotifySleep(time_now_d() - frameSleepStart, frameSleepPos);
|
||||
}
|
||||
@@ -786,7 +786,7 @@ void hleLagSync(u64 userdata, int cyclesLate) {
|
||||
const int over = (int)((now - goal) * 1000000);
|
||||
ScheduleLagSync(over - emuOver);
|
||||
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
|
||||
DisplayNotifySleep(now - before);
|
||||
}
|
||||
}
|
||||
@@ -846,9 +846,9 @@ void __DisplaySetFramebuf(u32 topaddr, int linesize, int pixelFormat, int sync)
|
||||
// Doing it in non-buffered though creates problems (black screen) on occasion though
|
||||
// so let's not.
|
||||
if (!flippedThisFrame && !g_Config.bSkipBufferEffects) {
|
||||
double before_flip = time_now_d();
|
||||
const double before_flip = time_now_d();
|
||||
__DisplayFlip(0);
|
||||
double after_flip = time_now_d();
|
||||
const double after_flip = time_now_d();
|
||||
// Ignore for debug stats.
|
||||
hleSetFlipTime(after_flip - before_flip);
|
||||
}
|
||||
|
||||
@@ -780,7 +780,7 @@ void UnexportFuncSymbol(const FuncSymbolExport &func) {
|
||||
}
|
||||
}
|
||||
|
||||
// Used to add detail to the "Unknown syscall" log in HLE.cpp's GetSyscallFuncPointer - a call
|
||||
// Used to add detail to the "Unknown syscall" log in HLE.cpp's GetSyscallFunctionData - a call
|
||||
// through a still-unresolved import ends up as a generic "invalid syscall" opcode that no
|
||||
// longer carries the original module name/NID, but the (fixed, unique) address of the syscall
|
||||
// instruction itself does - it's exactly the stubAddr every pending FuncSymbolImport recorded
|
||||
@@ -793,7 +793,7 @@ bool KernelFindImportByStubAddr(u32 stubAddr, std::string *importModuleName, u32
|
||||
continue;
|
||||
}
|
||||
for (const auto &func : module->importedFuncs) {
|
||||
if (func.stubAddr == stubAddr) {
|
||||
if (Memory::AddressesEqualAfterMask(func.stubAddr, stubAddr)) {
|
||||
*importModuleName = func.moduleName;
|
||||
*nid = func.nid;
|
||||
*importingModuleName = module->GetName();
|
||||
|
||||
+1
-1
@@ -95,7 +95,7 @@ static void CalculateFPS() {
|
||||
}
|
||||
}
|
||||
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || coreCollectDebugStats) {
|
||||
if ((DebugOverlay)g_Config.iDebugOverlay == DebugOverlay::FRAME_GRAPH || g_coreCollectDebugStats) {
|
||||
frameTimeHistory[frameTimeHistoryPos++] = (float)(now - lastFrameTimeHistory);
|
||||
lastFrameTimeHistory = now;
|
||||
frameTimeHistoryPos = frameTimeHistoryPos % frameTimeHistorySize;
|
||||
|
||||
@@ -620,17 +620,20 @@ void ArmJit::Comp_Syscall(MIPSOpcode op)
|
||||
QuickCallFunction(R1, (void *)&CallSyscall);
|
||||
#else
|
||||
// Skip the CallSyscall where possible.
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc)
|
||||
{
|
||||
gpr.SetRegImm(R0, (u32)(intptr_t)GetSyscallFuncPointer(op));
|
||||
// Already flushed, so R1 is safe.
|
||||
QuickCallFunction(R1, quickFunc);
|
||||
}
|
||||
else
|
||||
{
|
||||
gpr.SetRegImm(R0, op.encoding);
|
||||
QuickCallFunction(R1, (void *)&CallSyscall);
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
gpr.SetRegImm(R0, (uintptr_t)func);
|
||||
// Already flushed, so R1 is safe.
|
||||
QuickCallFunction(R1, quickFunc);
|
||||
} else {
|
||||
gpr.SetRegImm(R0, op.encoding);
|
||||
QuickCallFunction(R1, (void *)&CallSyscall);
|
||||
}
|
||||
} else {
|
||||
gpr.SetRegImm(R0, js.compilerPC);
|
||||
QuickCallFunction(R1, (void *)&CallSyscallUnresolvedAtPC);
|
||||
}
|
||||
#endif
|
||||
ApplyRoundingMode();
|
||||
|
||||
@@ -593,7 +593,6 @@ void Arm64Jit::Comp_JumpReg(MIPSOpcode op)
|
||||
WriteExitDestInR(destReg);
|
||||
js.compiling = false;
|
||||
}
|
||||
|
||||
|
||||
void Arm64Jit::Comp_Syscall(MIPSOpcode op)
|
||||
{
|
||||
@@ -636,14 +635,20 @@ void Arm64Jit::Comp_Syscall(MIPSOpcode op)
|
||||
QuickCallFunction(X1, (void *)&CallSyscall);
|
||||
#else
|
||||
// Skip the CallSyscall where possible.
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc) {
|
||||
MOVI2R(X0, (uintptr_t)GetSyscallFuncPointer(op));
|
||||
// Already flushed, so X1 is safe.
|
||||
QuickCallFunction(X1, quickFunc);
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
MOVI2R(X0, (uintptr_t)func);
|
||||
// Already flushed, so X1 is safe.
|
||||
QuickCallFunction(X1, quickFunc);
|
||||
} else {
|
||||
MOVI2R(W0, op.encoding);
|
||||
QuickCallFunction(X1, (void *)&CallSyscall);
|
||||
}
|
||||
} else {
|
||||
MOVI2R(W0, op.encoding);
|
||||
QuickCallFunction(X1, (void *)&CallSyscall);
|
||||
MOVI2R(W0, js.compilerPC);
|
||||
QuickCallFunction(X1, (void *)&CallSyscallUnresolvedAtPC);
|
||||
}
|
||||
#endif
|
||||
LoadStaticRegisters();
|
||||
|
||||
@@ -219,13 +219,20 @@ void Arm64JitBackend::CompIR_System(IRInst inst) {
|
||||
// Skip the CallSyscall where possible.
|
||||
{
|
||||
MIPSOpcode op(inst.constant);
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc) {
|
||||
MOVP2R(X0, GetSyscallFuncPointer(op));
|
||||
QuickCallFunction(SCRATCH2_64, (const u8 *)quickFunc);
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, 0);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
MOVP2R(X0, func);
|
||||
QuickCallFunction(SCRATCH2_64, (const u8 *)quickFunc);
|
||||
} else {
|
||||
MOVI2R(W0, inst.constant);
|
||||
QuickCallFunction(SCRATCH2_64, &CallSyscall);
|
||||
}
|
||||
} else {
|
||||
MOVI2R(W0, inst.constant);
|
||||
QuickCallFunction(SCRATCH2_64, &CallSyscall);
|
||||
// Shouldn't get here.
|
||||
MOVI2R(W0, 0);
|
||||
QuickCallFunction(SCRATCH2_64, &CallSyscallUnresolvedAtPC);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
@@ -235,6 +242,16 @@ void Arm64JitBackend::CompIR_System(IRInst inst) {
|
||||
// This is always followed by an ExitToPC, where we check coreState.
|
||||
break;
|
||||
|
||||
case IROp::SyscallUnresolved:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
|
||||
MOVI2R(W0, inst.constant);
|
||||
QuickCallFunction(SCRATCH2_64, &CallSyscallUnresolvedAtPC);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
LoadStaticRegisters();
|
||||
break;
|
||||
|
||||
case IROp::CallReplacement:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
|
||||
@@ -65,6 +65,7 @@ static void NoBlockExits() {
|
||||
_assert_msg_(false, "Never exited block, invalid IR?");
|
||||
}
|
||||
|
||||
// TODO: Much of this function should be merged with the same function for the other backends.
|
||||
bool Arm64JitBackend::CompileBlock(IRBlockCache *irBlockCache, int block_num) {
|
||||
if (GetSpaceLeft() < 0x800)
|
||||
return false;
|
||||
|
||||
@@ -77,7 +77,7 @@ static int IRReadsFromList(const IRInstMeta &inst, IRReg regs[4], char type) {
|
||||
if ((inst.m.flags & (IRFLAG_SRC3 | IRFLAG_SRC3DST)) != 0 && inst.m.types[0] == type)
|
||||
regs[c++] = inst.src3;
|
||||
|
||||
if (inst.op == IROp::Interpret || inst.op == IROp::CallReplacement || inst.op == IROp::Syscall || inst.op == IROp::Break)
|
||||
if (inst.op == IROp::Interpret || inst.op == IROp::CallReplacement || inst.op == IROp::Syscall || inst.op == IROp::SyscallUnresolved ||inst.op == IROp::Break)
|
||||
return -1;
|
||||
if (inst.op == IROp::Breakpoint || inst.op == IROp::MemoryCheck)
|
||||
return -1;
|
||||
|
||||
+11
-11
@@ -61,19 +61,19 @@ void IRFrontend::Comp_IType(MIPSOpcode op) {
|
||||
switch (op >> 26) {
|
||||
case 8: // same as addiu?
|
||||
case 9: // R(rt) = R(rs) + simm; break; //addiu
|
||||
ir.Write(IROp::AddConst, rt, rs, ir.AddConstant(simm));
|
||||
ir.Write(IROp::AddConst, rt, rs, 0, (u32)simm);
|
||||
break;
|
||||
|
||||
case 12: ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(uimm)); break;
|
||||
case 13: ir.Write(IROp::OrConst, rt, rs, ir.AddConstant(uimm)); break;
|
||||
case 14: ir.Write(IROp::XorConst, rt, rs, ir.AddConstant(uimm)); break;
|
||||
case 12: ir.Write(IROp::AndConst, rt, rs, 0, uimm); break;
|
||||
case 13: ir.Write(IROp::OrConst, rt, rs, 0, uimm); break;
|
||||
case 14: ir.Write(IROp::XorConst, rt, rs, 0, uimm); break;
|
||||
|
||||
case 10: // R(rt) = (s32)R(rs) < simm; break; //slti
|
||||
ir.Write(IROp::SltConst, rt, rs, ir.AddConstant(simm));
|
||||
ir.Write(IROp::SltConst, rt, rs, 0, (u32)simm);
|
||||
break;
|
||||
|
||||
case 11: // R(rt) = R(rs) < suimm; break; //sltiu
|
||||
ir.Write(IROp::SltUConst, rt, rs, ir.AddConstant(suimm));
|
||||
ir.Write(IROp::SltUConst, rt, rs, 0, suimm);
|
||||
break;
|
||||
|
||||
case 15: // R(rt) = uimm << 16; //lui
|
||||
@@ -197,7 +197,7 @@ void IRFrontend::CompShiftVar(MIPSOpcode op, IROp shiftOp) {
|
||||
// The interpreter already masks where needed, don't need to generate extra ops.
|
||||
ir.Write(shiftOp, rd, rt, rs);
|
||||
} else {
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(31));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, (u32)31);
|
||||
ir.Write(shiftOp, rd, rt, IRTEMP_0);
|
||||
}
|
||||
}
|
||||
@@ -244,9 +244,9 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
|
||||
case 0x0: // ext
|
||||
if (pos != 0) {
|
||||
ir.Write(IROp::ShrImm, rt, rs, pos);
|
||||
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, rt, rt, 0, mask);
|
||||
} else {
|
||||
ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, rt, rs, 0, mask);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -259,7 +259,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
|
||||
|
||||
if (size != 32) {
|
||||
// Need to use the sourcemask.
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(sourcemask));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, sourcemask);
|
||||
if (pos != 0) {
|
||||
ir.Write(IROp::ShlImm, IRTEMP_0, IRTEMP_0, pos);
|
||||
}
|
||||
@@ -271,7 +271,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
|
||||
ir.Write(IROp::Mov, IRTEMP_0, rs);
|
||||
}
|
||||
}
|
||||
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(destmask));
|
||||
ir.Write(IROp::AndConst, rt, rt, 0, destmask);
|
||||
ir.Write(IROp::Or, rt, rt, IRTEMP_0);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -96,11 +96,11 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs, rhs);
|
||||
ir.Write(ComparisonToExit(cc), 0, lhs, rhs, ResolveNotTakenTarget(branchInfo));
|
||||
// This makes the block "impure" :(
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
@@ -114,7 +114,7 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
}
|
||||
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -147,11 +147,11 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs);
|
||||
ir.Write(ComparisonToExit(cc), 0, lhs, 0, ResolveNotTakenTarget(branchInfo));
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
if (branchInfo.delaySlotIsBranch) {
|
||||
@@ -165,7 +165,7 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
|
||||
|
||||
// Taken
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -225,12 +225,12 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
// Not taken
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
|
||||
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
|
||||
// Taken
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
@@ -244,7 +244,7 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
}
|
||||
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -284,14 +284,14 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
int imm3 = (op >> 18) & 7;
|
||||
|
||||
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, ir.AddConstant(1 << imm3));
|
||||
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, 0, 1 << imm3);
|
||||
FlushAll();
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
|
||||
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
|
||||
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
@@ -306,7 +306,7 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
|
||||
// Taken
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -355,11 +355,11 @@ void IRFrontend::Comp_Jump(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -420,7 +420,7 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
ir.Write(IROp::ExitToReg, 0, destReg, 0);
|
||||
@@ -433,18 +433,25 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
|
||||
void IRFrontend::Comp_Syscall(MIPSOpcode op) {
|
||||
// Note: If we're in a delay slot, this is off by one compared to the interpreter.
|
||||
int dcAmount = js.downcountAmount + (js.inDelaySlot ? -1 : 0);
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
// If not in a delay slot, we need to update PC.
|
||||
// However we also need the PC for some diagnostics so let's just always do it.
|
||||
// Not exactly a bottleneck.
|
||||
if (!js.inDelaySlot) {
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC() + 4));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC() + 4);
|
||||
}
|
||||
|
||||
FlushAll();
|
||||
|
||||
RestoreRoundingMode();
|
||||
ir.Write(IROp::Syscall, 0, ir.AddConstant(op.encoding));
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
|
||||
if (func) {
|
||||
ir.Write(IROp::Syscall, 0, 0, 0, op.encoding);
|
||||
} else {
|
||||
ir.Write(IROp::SyscallUnresolved, 0, 0, 0, js.compilerPC);
|
||||
}
|
||||
ApplyRoundingMode();
|
||||
ir.Write(IROp::ExitToPC);
|
||||
|
||||
@@ -452,7 +459,7 @@ void IRFrontend::Comp_Syscall(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
void IRFrontend::Comp_Break(MIPSOpcode op) {
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
ir.Write(IROp::Break);
|
||||
js.compiling = false;
|
||||
}
|
||||
|
||||
@@ -83,11 +83,11 @@ void IRFrontend::Comp_FPULS(MIPSOpcode op) {
|
||||
|
||||
switch (op >> 26) {
|
||||
case 49: // lwc1
|
||||
ir.Write(IROp::LoadFloat, ft, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::LoadFloat, ft, rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 57: // swc1
|
||||
ir.Write(IROp::StoreFloat, ft, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::StoreFloat, ft, rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -207,10 +207,10 @@ void IRFrontend::Comp_mxc1(MIPSOpcode op) {
|
||||
// This needs to insert fpcond.
|
||||
ir.Write(IROp::FpCtrlToReg, rt);
|
||||
} else if (fs == 0) {
|
||||
ir.Write(IROp::SetConst, rt, ir.AddConstant(MIPSState::FCR0_VALUE));
|
||||
ir.WriteSetConstant(rt, MIPSState::FCR0_VALUE);
|
||||
} else {
|
||||
// Unsupported regs are always 0.
|
||||
ir.Write(IROp::SetConst, rt, ir.AddConstant(0));
|
||||
ir.WriteSetConstant(rt, 0);
|
||||
}
|
||||
return;
|
||||
|
||||
|
||||
@@ -61,42 +61,42 @@ namespace MIPSComp {
|
||||
switch (o) {
|
||||
// Load
|
||||
case 35:
|
||||
ir.Write(IROp::Load32, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32, rt, rs, 0, offset);
|
||||
break;
|
||||
case 37:
|
||||
ir.Write(IROp::Load16, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load16, rt, rs, 0, offset);
|
||||
break;
|
||||
case 33:
|
||||
ir.Write(IROp::Load16Ext, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load16Ext, rt, rs, 0, offset);
|
||||
break;
|
||||
case 36:
|
||||
ir.Write(IROp::Load8, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load8, rt, rs, 0, offset);
|
||||
break;
|
||||
case 32:
|
||||
ir.Write(IROp::Load8Ext, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load8Ext, rt, rs, 0, offset);
|
||||
break;
|
||||
// Store
|
||||
case 43:
|
||||
ir.Write(IROp::Store32, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32, rt, rs, 0, offset);
|
||||
break;
|
||||
case 41:
|
||||
ir.Write(IROp::Store16, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store16, rt, rs, 0, offset);
|
||||
break;
|
||||
case 40:
|
||||
ir.Write(IROp::Store8, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store8, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 34: //lwl
|
||||
ir.Write(IROp::Load32Left, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32Left, rt, rs, 0, offset);
|
||||
break;
|
||||
case 38: //lwr
|
||||
ir.Write(IROp::Load32Right, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32Right, rt, rs, 0, offset);
|
||||
break;
|
||||
case 42: //swl
|
||||
ir.Write(IROp::Store32Left, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32Left, rt, rs, 0, offset);
|
||||
break;
|
||||
case 46: //swr
|
||||
ir.Write(IROp::Store32Right, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32Right, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -117,11 +117,11 @@ namespace MIPSComp {
|
||||
|
||||
switch (op >> 26) {
|
||||
case 48: // ll
|
||||
ir.Write(IROp::Load32Linked, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32Linked, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 56: // sc
|
||||
ir.Write(IROp::Store32Conditional, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32Conditional, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
|
||||
+41
-40
@@ -256,11 +256,11 @@ namespace MIPSComp {
|
||||
}
|
||||
|
||||
// Nope, it has something else going on.
|
||||
zeroedLanes = -1;
|
||||
zeroedLanes = (u32)-1;
|
||||
break;
|
||||
}
|
||||
|
||||
if (zeroedLanes != -1) {
|
||||
if (zeroedLanes != (u32)-1) {
|
||||
InitRegs(vregs, tempReg);
|
||||
ir.Write(IROp::Vec4Init, vregs[0], (int)Vec4Init::AllZERO);
|
||||
ir.Write(IROp::Vec4Blend, vregs[0], origV[0], vregs[0], zeroedLanes);
|
||||
@@ -284,8 +284,9 @@ namespace MIPSComp {
|
||||
if (!constants) {
|
||||
if (regnum >= n) {
|
||||
// Depends on the op, but often zero.
|
||||
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(vregs[i], 0.0f);
|
||||
} else if (abs) {
|
||||
// Could have a FNAbs op, but probably not worth it.
|
||||
ir.Write(IROp::FAbs, vregs[i], origV[regnum]);
|
||||
if (negate)
|
||||
ir.Write(IROp::FNeg, vregs[i], vregs[i]);
|
||||
@@ -297,9 +298,9 @@ namespace MIPSComp {
|
||||
}
|
||||
} else {
|
||||
if (negate) {
|
||||
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(-constantArray[regnum + (abs << 2)]));
|
||||
ir.WriteSetConstantFloat(vregs[i], -constantArray[regnum + (abs << 2)]);
|
||||
} else {
|
||||
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(constantArray[regnum + (abs << 2)]));
|
||||
ir.WriteSetConstantFloat(vregs[i], constantArray[regnum + (abs << 2)]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -401,11 +402,11 @@ namespace MIPSComp {
|
||||
|
||||
switch (op >> 26) {
|
||||
case 50: //lv.s
|
||||
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 58: //sv.s
|
||||
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -461,29 +462,29 @@ namespace MIPSComp {
|
||||
switch (optype) {
|
||||
case LSVType::LVQ:
|
||||
if (IsVec4(V_Quad, vregs)) {
|
||||
ir.Write(IROp::LoadVec4, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::LoadVec4, vregs[0], rs, 0, imm);
|
||||
} else {
|
||||
// Let's not even bother with "vertical" loads for now.
|
||||
if (!g_Config.bFastMemory)
|
||||
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 0, (u32)imm);
|
||||
ir.Write(IROp::LoadFloat, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::LoadFloat, vregs[1], rs, ir.AddConstant(imm + 4));
|
||||
ir.Write(IROp::LoadFloat, vregs[2], rs, ir.AddConstant(imm + 8));
|
||||
ir.Write(IROp::LoadFloat, vregs[3], rs, ir.AddConstant(imm + 12));
|
||||
ir.Write(IROp::LoadFloat, vregs[0], rs, 0, imm);
|
||||
ir.Write(IROp::LoadFloat, vregs[1], rs, 0, imm + 4);
|
||||
ir.Write(IROp::LoadFloat, vregs[2], rs, 0, imm + 8);
|
||||
ir.Write(IROp::LoadFloat, vregs[3], rs, 0, imm + 12);
|
||||
}
|
||||
break;
|
||||
|
||||
case LSVType::SVQ:
|
||||
if (IsVec4(V_Quad, vregs)) {
|
||||
ir.Write(IROp::StoreVec4, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::StoreVec4, vregs[0], rs, 0, imm);
|
||||
} else {
|
||||
// Let's not even bother with "vertical" stores for now.
|
||||
if (!g_Config.bFastMemory)
|
||||
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 1, (u32)imm);
|
||||
ir.Write(IROp::StoreFloat, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::StoreFloat, vregs[1], rs, ir.AddConstant(imm + 4));
|
||||
ir.Write(IROp::StoreFloat, vregs[2], rs, ir.AddConstant(imm + 8));
|
||||
ir.Write(IROp::StoreFloat, vregs[3], rs, ir.AddConstant(imm + 12));
|
||||
ir.Write(IROp::StoreFloat, vregs[0], rs, 0, imm);
|
||||
ir.Write(IROp::StoreFloat, vregs[1], rs, 0, imm + 4);
|
||||
ir.Write(IROp::StoreFloat, vregs[2], rs, 0, imm + 8);
|
||||
ir.Write(IROp::StoreFloat, vregs[3], rs, 0, imm + 12);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -521,7 +522,7 @@ namespace MIPSComp {
|
||||
ir.Write(IROp::Vec4Init, dregs[0], (int)(type == 6 ? Vec4Init::AllZERO : Vec4Init::AllONE));
|
||||
} else {
|
||||
for (int i = 0; i < n; i++) {
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(type == 6 ? 0.0f : 1.0f));
|
||||
ir.WriteSetConstantFloat(dregs[i], type == 6 ? 0.0f : 1.0f);
|
||||
}
|
||||
}
|
||||
ApplyPrefixD(dregs, sz, vd);
|
||||
@@ -549,14 +550,14 @@ namespace MIPSComp {
|
||||
} else {
|
||||
switch (sz) {
|
||||
case V_Pair:
|
||||
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 1) == 0 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 1) == 1 ? 1.0f : 0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[0], (vd & 1) == 0 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[1], (vd & 1) == 1 ? 1.0f : 0.0f);
|
||||
break;
|
||||
case V_Quad:
|
||||
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 3) == 0 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 3) == 1 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[2], ir.AddConstantFloat((vd & 3) == 2 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[3], ir.AddConstantFloat((vd & 3) == 3 ? 1.0f : 0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[0], (vd & 3) == 0 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[1], (vd & 3) == 1 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[2], (vd & 3) == 2 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[3], (vd & 3) == 3 ? 1.0f : 0.0f);
|
||||
break;
|
||||
default:
|
||||
INVALIDOP;
|
||||
@@ -594,19 +595,19 @@ namespace MIPSComp {
|
||||
switch ((op >> 16) & 0xF) {
|
||||
case 3: // vmidt
|
||||
if (x == 0 && y == 0)
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
|
||||
else if (x == y)
|
||||
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
|
||||
else
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
|
||||
break;
|
||||
case 6: // vmzero
|
||||
// Likely to be fast.
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
|
||||
break;
|
||||
case 7: // vmone
|
||||
if (x == 0 && y == 0)
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
|
||||
else
|
||||
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
|
||||
break;
|
||||
@@ -707,7 +708,7 @@ namespace MIPSComp {
|
||||
GetVectorRegsPrefixD(dregs, V_Single, _VD);
|
||||
|
||||
// We have to start at +0.000 in case any values are -0.000.
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0, 0.0f);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
ir.Write(IROp::FAdd, IRVTEMP_0, IRVTEMP_0, sregs[i]);
|
||||
}
|
||||
@@ -717,7 +718,7 @@ namespace MIPSComp {
|
||||
ir.Write(IROp::FMov, dregs[0], IRVTEMP_0);
|
||||
break;
|
||||
case 7: // vavg
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0 + 1, ir.AddConstantFloat(vavg_table[n - 1]));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0 + 1, vavg_table[n - 1]);
|
||||
ir.Write(IROp::FMul, dregs[0], IRVTEMP_0, IRVTEMP_0 + 1);
|
||||
break;
|
||||
}
|
||||
@@ -939,7 +940,7 @@ namespace MIPSComp {
|
||||
case VecDo3Op::VSGE: // vsge
|
||||
ir.Write(IROp::FCmp, (int)IRFpCompareMode::LessUnordered, sregs[i], tregs[i]);
|
||||
ir.Write(IROp::FpCondToReg, IRTEMP_1);
|
||||
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, ir.AddConstant(1));
|
||||
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, 0, 1);
|
||||
ir.Write(IROp::FMovFromGPR, tempregs[i], IRTEMP_1);
|
||||
ir.Write(IROp::FCvtSW, tempregs[i], tempregs[i]);
|
||||
break;
|
||||
@@ -1266,7 +1267,7 @@ namespace MIPSComp {
|
||||
u32 mask;
|
||||
if (GetVFPUCtrlMask(imm - 128, &mask)) {
|
||||
if (mask != 0xFFFFFFFF) {
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rt, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rt, 0, mask);
|
||||
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, IRTEMP_0);
|
||||
} else {
|
||||
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, rt);
|
||||
@@ -1322,7 +1323,7 @@ namespace MIPSComp {
|
||||
if (GetVFPUCtrlMask(imm, &mask)) {
|
||||
if (mask != 0xFFFFFFFF) {
|
||||
ir.Write(IROp::FMovToGPR, IRTEMP_0, vfpuBase + voffset[imm]);
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, 0, mask);
|
||||
ir.Write(IROp::SetCtrlVFPUReg, imm, IRTEMP_0);
|
||||
} else {
|
||||
ir.Write(IROp::SetCtrlVFPUFReg, imm, vfpuBase + voffset[vs]);
|
||||
@@ -2146,7 +2147,7 @@ namespace MIPSComp {
|
||||
s32 imm = SignExtend16ToS32(op);
|
||||
u8 dreg;
|
||||
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
|
||||
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat((float)imm));
|
||||
ir.WriteSetConstantFloat(dreg, (float)imm);
|
||||
ApplyPrefixD(&dreg, V_Single, _VT);
|
||||
}
|
||||
|
||||
@@ -2164,7 +2165,7 @@ namespace MIPSComp {
|
||||
|
||||
u8 dreg;
|
||||
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
|
||||
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat(fval.f));
|
||||
ir.WriteSetConstantFloat(dreg, fval.f);
|
||||
ApplyPrefixD(&dreg, V_Single, _VT);
|
||||
}
|
||||
|
||||
@@ -2186,17 +2187,17 @@ namespace MIPSComp {
|
||||
GetVectorRegsPrefixD(dregs, sz, vd);
|
||||
|
||||
if (IsVec4(sz, dregs)) {
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
|
||||
ir.Write(IROp::Vec4Shuffle, dregs[0], IRVTEMP_0, 0);
|
||||
} else if (IsVec3of4(sz, dregs) && opts.preferVec4) {
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
|
||||
ir.Write(IROp::Vec4Shuffle, IRVTEMP_0, IRVTEMP_0, 0);
|
||||
ir.Write(IROp::Vec4Blend, dregs[0], dregs[0], IRVTEMP_0, 0x7);
|
||||
} else {
|
||||
for (int i = 0; i < n; i++) {
|
||||
// Most of the time, materializing a float is slower than copying from another float.
|
||||
if (i == 0)
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(cst_constants[conNum]));
|
||||
ir.WriteSetConstantFloat(dregs[i], cst_constants[conNum]);
|
||||
else
|
||||
ir.Write(IROp::FMov, dregs[i], dregs[0]);
|
||||
}
|
||||
@@ -2253,7 +2254,7 @@ namespace MIPSComp {
|
||||
for (int i = 0; i < n; i++) {
|
||||
switch (d[i]) {
|
||||
case '0':
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f);
|
||||
break;
|
||||
case 's':
|
||||
if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) {
|
||||
@@ -2271,7 +2272,7 @@ namespace MIPSComp {
|
||||
else if (dregs[sineLane] == sreg[0])
|
||||
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
|
||||
else
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(1.0f));
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
+16
-15
@@ -76,17 +76,17 @@ void IRFrontend::FlushPrefixV() {
|
||||
}
|
||||
|
||||
if ((js.prefixSFlag & JitState::PREFIX_DIRTY) != 0) {
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, ir.AddConstant(js.prefixS));
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, 0, 0, js.prefixS);
|
||||
js.prefixSFlag = (JitState::PrefixState) (js.prefixSFlag & ~JitState::PREFIX_DIRTY);
|
||||
}
|
||||
|
||||
if ((js.prefixTFlag & JitState::PREFIX_DIRTY) != 0) {
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, ir.AddConstant(js.prefixT));
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, 0, 0, js.prefixT);
|
||||
js.prefixTFlag = (JitState::PrefixState) (js.prefixTFlag & ~JitState::PREFIX_DIRTY);
|
||||
}
|
||||
|
||||
if ((js.prefixDFlag & JitState::PREFIX_DIRTY) != 0) {
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, ir.AddConstant(js.prefixD));
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, 0, 0, js.prefixD);
|
||||
js.prefixDFlag = (JitState::PrefixState) (js.prefixDFlag & ~JitState::PREFIX_DIRTY);
|
||||
}
|
||||
|
||||
@@ -164,8 +164,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
|
||||
} else if (entry->replaceFunc) {
|
||||
FlushAll();
|
||||
RestoreRoundingMode();
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::CallReplacement, IRTEMP_0, ir.AddConstant(index));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
ir.Write(IROp::CallReplacement, IRTEMP_0, 0, 0, index);
|
||||
|
||||
if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) {
|
||||
// Compile the original instruction at this address. We ignore cycles for hooks.
|
||||
@@ -175,8 +175,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
|
||||
ApplyRoundingMode();
|
||||
// If IRTEMP_0 was set to 1, it means the replacement needs to run again (sliced.)
|
||||
// This is necessary for replacements that take a lot of cycles.
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(js.downcountAmount));
|
||||
ir.Write(IROp::ExitToConstIfNeq, ir.AddConstant(GetCompilerPC()), IRTEMP_0, MIPS_REG_ZERO);
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, js.downcountAmount);
|
||||
ir.Write(IROp::ExitToConstIfNeq, 0, IRTEMP_0, MIPS_REG_ZERO, GetCompilerPC());
|
||||
ir.Write(IROp::ExitToReg, 0, MIPS_REG_RA, 0);
|
||||
js.compiling = false;
|
||||
}
|
||||
@@ -187,7 +187,7 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
|
||||
|
||||
void IRFrontend::Comp_Generic(MIPSOpcode op) {
|
||||
FlushAll();
|
||||
ir.Write(IROp::Interpret, 0, ir.AddConstant(op.encoding));
|
||||
ir.Write(IROp::Interpret, 0, 0, 0, op.encoding);
|
||||
const MIPSInfo info = MIPSGetInfo(op);
|
||||
if ((info & IS_VFPU) != 0 && (info & VFPU_NO_PREFIX) == 0) {
|
||||
// If it does eat them, it'll happen in MIPSCompileOp().
|
||||
@@ -359,7 +359,7 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
|
||||
FlushAll();
|
||||
|
||||
// Can't skip this even at the start of a block, might impact block linking.
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
|
||||
RestoreRoundingMode();
|
||||
// At this point, downcount HAS the delay slot, but not the instruction itself.
|
||||
@@ -374,11 +374,12 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
|
||||
}
|
||||
}
|
||||
int downcountAmount = js.downcountAmount + downcountOffset;
|
||||
if (downcountAmount != 0)
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
|
||||
if (downcountAmount != 0) {
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
|
||||
}
|
||||
// Note that this means downcount can't be metadata on the block.
|
||||
js.downcountAmount = -downcountOffset;
|
||||
ir.Write(IROp::Breakpoint, 0, ir.AddConstant(addr));
|
||||
ir.Write(IROp::Breakpoint, 0, 0, 0, addr);
|
||||
ApplyRoundingMode();
|
||||
|
||||
js.hadBreakpoints = true;
|
||||
@@ -390,7 +391,7 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
|
||||
FlushAll();
|
||||
|
||||
// Can't skip this even at the start of a block, might impact block linking.
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
|
||||
RestoreRoundingMode();
|
||||
// At this point, downcount HAS the delay slot, but not the instruction itself.
|
||||
@@ -407,10 +408,10 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
|
||||
}
|
||||
int downcountAmount = js.downcountAmount + downcountOffset;
|
||||
if (downcountAmount != 0)
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
|
||||
// Note that this means downcount can't be metadata on the block.
|
||||
js.downcountAmount = -downcountOffset;
|
||||
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, 0, offset);
|
||||
ApplyRoundingMode();
|
||||
|
||||
js.hadBreakpoints = true;
|
||||
|
||||
+18
-13
@@ -176,6 +176,7 @@ static const IRMeta irMeta[] = {
|
||||
{ IROp::ExitToConstIfLtZ, "ExitIfLtZ", "CG", IRFLAG_EXIT },
|
||||
{ IROp::ExitToReg, "ExitToReg", "_G", IRFLAG_EXIT },
|
||||
{ IROp::Syscall, "Syscall", "_C", IRFLAG_EXIT },
|
||||
{ IROp::SyscallUnresolved, "SyscallUnresolved", "C", IRFLAG_EXIT },
|
||||
{ IROp::Break, "Break", "", IRFLAG_EXIT },
|
||||
{ IROp::SetPC, "SetPC", "_G" },
|
||||
{ IROp::SetPCConst, "SetPC", "_C" },
|
||||
@@ -206,31 +207,35 @@ void InitIR() {
|
||||
}
|
||||
}
|
||||
|
||||
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2) {
|
||||
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2, u32 constant) {
|
||||
IRInst inst;
|
||||
inst.op = op;
|
||||
inst.dest = dst;
|
||||
inst.src1 = src1;
|
||||
inst.src2 = src2;
|
||||
inst.constant = nextConst_;
|
||||
inst.constant = constant;
|
||||
insts_.push_back(inst);
|
||||
}
|
||||
|
||||
nextConst_ = 0;
|
||||
void IRWriter::WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant) {
|
||||
u32 constant;
|
||||
memcpy(&constant, &fconstant, sizeof(u32));
|
||||
|
||||
IRInst inst;
|
||||
inst.op = op;
|
||||
inst.dest = dst;
|
||||
inst.src1 = src1;
|
||||
inst.src2 = src2;
|
||||
inst.constant = constant;
|
||||
insts_.push_back(inst);
|
||||
}
|
||||
|
||||
void IRWriter::WriteSetConstant(u8 dst, u32 value) {
|
||||
Write(IROp::SetConst, dst, AddConstant(value));
|
||||
Write(IROp::SetConst, dst, 0, 0, value);
|
||||
}
|
||||
|
||||
int IRWriter::AddConstant(u32 value) {
|
||||
nextConst_ = value;
|
||||
return 255;
|
||||
}
|
||||
|
||||
int IRWriter::AddConstantFloat(float value) {
|
||||
u32 val;
|
||||
memcpy(&val, &value, 4);
|
||||
return AddConstant(val);
|
||||
void IRWriter::WriteSetConstantFloat(u8 dst, float value) {
|
||||
WriteFC(IROp::SetConstF, dst, 0, 0, value);
|
||||
}
|
||||
|
||||
void IRWriter::ReplaceConstant(size_t instNumber, u32 newConstant) {
|
||||
|
||||
+7
-11
@@ -222,7 +222,8 @@ enum class IROp : uint8_t {
|
||||
ExitToConstIfFpFalse,
|
||||
ExitToPC, // Used after a syscall to give us a way to do things before returning.
|
||||
|
||||
Syscall,
|
||||
Syscall, // Needs the syscall instruction work in the constant.
|
||||
SyscallUnresolved, // Used when the syscall is not resolved at compile time. PC in the constant.
|
||||
SetPC, // hack to make syscall returns work
|
||||
SetPCConst, // hack to make replacement know PC
|
||||
CallReplacement,
|
||||
@@ -384,18 +385,14 @@ public:
|
||||
return *this;
|
||||
}
|
||||
|
||||
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0);
|
||||
void Write(IROp op, IRReg dst, IRReg src1, IRReg src2, uint32_t c) {
|
||||
AddConstant(c);
|
||||
Write(op, dst, src1, src2);
|
||||
}
|
||||
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0, u32 constant = 0);
|
||||
void WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant);
|
||||
|
||||
void WriteSetConstant(u8 dst, u32 value);
|
||||
void WriteSetConstantFloat(u8 dst, float value);
|
||||
void Write(IRInst inst) {
|
||||
insts_.push_back(inst);
|
||||
}
|
||||
void WriteSetConstant(u8 dst, u32 value);
|
||||
|
||||
int AddConstant(u32 value);
|
||||
int AddConstantFloat(float value);
|
||||
|
||||
void Reserve(size_t s) {
|
||||
insts_.reserve(s);
|
||||
@@ -409,7 +406,6 @@ public:
|
||||
|
||||
private:
|
||||
std::vector<IRInst> insts_;
|
||||
u32 nextConst_ = 0;
|
||||
};
|
||||
|
||||
struct IROptions {
|
||||
|
||||
@@ -1203,10 +1203,24 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
|
||||
case IROp::Syscall:
|
||||
// IROp::SetPC was (hopefully) executed before.
|
||||
{
|
||||
// If we get here, the syscall is valid.
|
||||
MIPSOpcode op(inst->constant);
|
||||
CallSyscall(op);
|
||||
if (coreState != CORE_RUNNING_CPU)
|
||||
if (coreState != CORE_RUNNING_CPU) {
|
||||
CoreTiming::ForceCheck(mips);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case IROp::SyscallUnresolved:
|
||||
{
|
||||
// If we get here, the syscall is invalid.
|
||||
u32 pc = inst->constant;
|
||||
CallSyscallUnresolvedAtPC(pc);
|
||||
if (coreState != CORE_RUNNING_CPU) {
|
||||
// hm, what's this for?
|
||||
CoreTiming::ForceCheck(mips);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1300,7 +1314,7 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
|
||||
}
|
||||
break;
|
||||
|
||||
case IROp::Nop: // TODO: This shouldn't crash, but for now we should not emit nops, so...
|
||||
case IROp::Nop: // Unused, add a break if we start using it to avoid UNREACHABLE.
|
||||
case IROp::Bad:
|
||||
default:
|
||||
// Unimplemented IR op. Bad. We define it as unreachable so the compiler can optimize better (remove the range check).
|
||||
|
||||
@@ -440,6 +440,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) {
|
||||
break;
|
||||
|
||||
case IROp::Syscall:
|
||||
case IROp::SyscallUnresolved:
|
||||
case IROp::CallReplacement:
|
||||
case IROp::Break:
|
||||
CompIR_System(inst);
|
||||
|
||||
@@ -258,21 +258,21 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
|
||||
if (opts.unalignedLoadStore) {
|
||||
// Write out one unaligned op.
|
||||
out.Write(replaceOp, inst.dest, inst.src1, out.AddConstant(inst.constant + replaceOff));
|
||||
out.Write(replaceOp, inst.dest, inst.src1, 0, inst.constant + replaceOff);
|
||||
} else if (replaceOp == IROp::Load32) {
|
||||
// We can still combine to a simpler set of two loads.
|
||||
// We start by isolating the address and shift amount.
|
||||
|
||||
// IRTEMP_LR_ADDR = rs + imm
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant + replaceOff));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant + replaceOff);
|
||||
// IRTEMP_LR_SHIFT = (addr & 3) * 8
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
|
||||
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
|
||||
// IRTEMP_LR_ADDR = addr & 0xfffffffc
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
|
||||
// IRTEMP_LR_VALUE = low_word, dest = high_word
|
||||
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, out.AddConstant(0));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(4));
|
||||
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, 0, 0);
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 4);
|
||||
|
||||
// Now we just need to adjust and combine dest and IRTEMP_LR_VALUE.
|
||||
// inst.dest >>= shift (putting its bits in the right spot.)
|
||||
@@ -281,7 +281,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::ShlImm, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, 8);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
|
||||
// IRTEMP_LR_VALUE <<= (24 - shift)
|
||||
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
|
||||
|
||||
@@ -297,18 +297,18 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
|
||||
auto addCommonProlog = [&]() {
|
||||
// IRTEMP_LR_ADDR = rs + imm
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant);
|
||||
// IRTEMP_LR_SHIFT = (addr & 3) * 8
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
|
||||
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
|
||||
// IRTEMP_LR_ADDR = addr & 0xfffffffc (for stores, later)
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
|
||||
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR)
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(0));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 0);
|
||||
};
|
||||
auto addCommonStore = [&](int off = 0) {
|
||||
// RAM(IRTEMP_LR_ADDR) = IRTEMP_LR_VALUE
|
||||
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(off));
|
||||
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, off);
|
||||
};
|
||||
|
||||
switch (inst.op) {
|
||||
@@ -327,7 +327,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::And, inst.dest, inst.dest, IRTEMP_LR_MASK);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
|
||||
// IRTEMP_LR_VALUE <<= (24 - shift)
|
||||
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
|
||||
// dest |= IRTEMP_LR_VALUE
|
||||
@@ -336,7 +336,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
bool src1Dirty = inst.dest == inst.src1;
|
||||
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
|
||||
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, nextOp().constant - inst.constant);
|
||||
|
||||
// dest &= IRTEMP_LR_MASK
|
||||
out.Write(IROp::And, nextOp().dest, nextOp().dest, IRTEMP_LR_MASK);
|
||||
@@ -362,7 +362,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::Shr, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
// dest &= (0xffffff00 << (24 - shift))
|
||||
// Alternatively, could shift to a wall and back (but would require two shifts each way.)
|
||||
out.WriteSetConstant(IRTEMP_LR_MASK, 0xffffff00);
|
||||
@@ -377,12 +377,12 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
bool src1Dirty = inst.dest == inst.src1;
|
||||
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
|
||||
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, (u32)(nextOp().constant - inst.constant));
|
||||
|
||||
if (shiftNeedsReverse) {
|
||||
// IRTEMP_LR_SHIFT = shift again
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
shiftNeedsReverse = false;
|
||||
}
|
||||
// IRTEMP_LR_VALUE >>= IRTEMP_LR_SHIFT
|
||||
@@ -411,7 +411,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
// IRTEMP_LR_VALUE |= src3 >> (24 - shift)
|
||||
out.Write(IROp::Shr, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
@@ -429,11 +429,11 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
// IRTEMP_LR_VALUE &= 0x00ffffff << (24 - shift)
|
||||
out.WriteSetConstant(IRTEMP_LR_MASK, 0x00ffffff);
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
out.Write(IROp::Shr, IRTEMP_LR_MASK, IRTEMP_LR_MASK, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
// IRTEMP_LR_VALUE |= src3 << shift
|
||||
out.Write(IROp::Shl, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
@@ -508,7 +508,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
if (inst.dest != inst.src1)
|
||||
out.Write(IROp::Mov, inst.dest, inst.src1);
|
||||
} else {
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, out.AddConstant(imm2));
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, 0, imm2);
|
||||
}
|
||||
} else if (symmetric && gpr.IsImm(inst.src1)) {
|
||||
const u32 imm1 = gpr.GetImm(inst.src1);
|
||||
@@ -518,7 +518,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
if (inst.dest != inst.src2)
|
||||
out.Write(IROp::Mov, inst.dest, inst.src2);
|
||||
} else {
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, out.AddConstant(imm1));
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, 0, imm1);
|
||||
}
|
||||
} else {
|
||||
gpr.MapDirtyInIn(inst.dest, inst.src1, inst.src2);
|
||||
@@ -632,7 +632,8 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
case IROp::FMovFromGPR:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(IROp::SetConstF, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
|
||||
// NOTE: SetConstantFloat doesn't work here since we actually want the bits.
|
||||
out.Write(IROp::SetConstF, inst.dest, 0, 0, gpr.GetImm(inst.src1));
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -661,7 +662,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::Store32Conditional:
|
||||
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
|
||||
gpr.MapIn(inst.dest);
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapInIn(inst.dest, inst.src1);
|
||||
goto doDefault;
|
||||
@@ -670,7 +671,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::StoreFloat:
|
||||
case IROp::StoreVec4:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -685,7 +686,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::Load32Linked:
|
||||
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
|
||||
gpr.MapDirty(inst.dest);
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapDirtyIn(inst.dest, inst.src1);
|
||||
goto doDefault;
|
||||
@@ -694,7 +695,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::LoadFloat:
|
||||
case IROp::LoadVec4:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -704,7 +705,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::Load32Right:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
gpr.MapIn(inst.dest);
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapInIn(inst.dest, inst.src1);
|
||||
goto doDefault;
|
||||
@@ -716,7 +717,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::ValidateAddress32:
|
||||
case IROp::ValidateAddress128:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -729,7 +730,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
case IROp::SetPC:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(IROp::SetPCConst, out.AddConstant(gpr.GetImm(inst.src1)));
|
||||
out.Write(IROp::SetPCConst, 0, 0, 0, gpr.GetImm(inst.src1));
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -772,7 +773,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
case IROp::SetCtrlVFPUReg:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(IROp::SetCtrlVFPU, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
|
||||
out.Write(IROp::SetCtrlVFPU, inst.dest, 0, 0, gpr.GetImm(inst.src1));
|
||||
} else {
|
||||
gpr.MapDirtyIn(IRREG_VFPU_CTRL_BASE + inst.dest, inst.src1);
|
||||
out.Write(inst);
|
||||
@@ -872,7 +873,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
// Reduce bloat by skipping on fail, and const exit on pass.
|
||||
if (passed) {
|
||||
gpr.FlushAll();
|
||||
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
|
||||
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
|
||||
skipNextExitToConst = true;
|
||||
}
|
||||
break;
|
||||
@@ -896,7 +897,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
if (passed) {
|
||||
gpr.FlushAll();
|
||||
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
|
||||
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
|
||||
skipNextExitToConst = true;
|
||||
}
|
||||
break;
|
||||
@@ -918,7 +919,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
// Prefer ExitToConst to allow block linking.
|
||||
u32 dest = gpr.GetImm(inst.src1);
|
||||
gpr.FlushAll();
|
||||
out.Write(IROp::ExitToConst, out.AddConstant(dest));
|
||||
out.Write(IROp::ExitToConst, 0, 0, 0, dest);
|
||||
break;
|
||||
}
|
||||
gpr.FlushAll();
|
||||
@@ -2040,10 +2041,11 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
|
||||
} else if (inst.constant == 0xBF800000) {
|
||||
out.Write(IROp::Vec4Init, temp, (int)Vec4Init::AllMinusONE);
|
||||
} else {
|
||||
out.Write(IROp::SetConstF, temp, out.AddConstant(inst.constant));
|
||||
// NOTE: WriteSetConstantFloat doesn't work here since we actually want the bits.
|
||||
out.Write(IROp::SetConstF, temp, 0, 0, inst.constant);
|
||||
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
|
||||
}
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
|
||||
isVec4Dirty[inst.dest & ~3] = true;
|
||||
continue;
|
||||
}
|
||||
@@ -2054,7 +2056,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
|
||||
u8 blendMask = 1 << (inst.dest & 3);
|
||||
out.Write(IROp::FMovFromGPR, temp, inst.src1);
|
||||
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
|
||||
isVec4Dirty[inst.dest & ~3] = true;
|
||||
continue;
|
||||
}
|
||||
@@ -2065,7 +2067,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
|
||||
u8 blendMask = 1 << (inst.dest & 3);
|
||||
out.Write(inst.op, temp, inst.src1, inst.src2, inst.constant);
|
||||
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
|
||||
isVec4Dirty[inst.dest & ~3] = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -184,6 +184,7 @@ namespace MIPSInt {
|
||||
}
|
||||
|
||||
void Int_Syscall(MIPSState *mips, MIPSOpcode op) {
|
||||
const u32 syscallPC = mips->pc - 4;
|
||||
// Need to pre-move PC, as CallSyscall may result in a rescheduling!
|
||||
// To do this neater, we'll need a little generated kernel loop that syscall can jump to and then RFI from
|
||||
// but I don't see a need to bother.
|
||||
@@ -193,7 +194,7 @@ namespace MIPSInt {
|
||||
mips->pc += 4;
|
||||
}
|
||||
mips->inDelaySlot = false;
|
||||
CallSyscall(op);
|
||||
CallSyscallWithPC(op, syscallPC);
|
||||
}
|
||||
|
||||
void Int_Sync(MIPSState *mips, MIPSOpcode op) {
|
||||
|
||||
@@ -185,22 +185,38 @@ void LoongArch64JitBackend::CompIR_System(IRInst inst) {
|
||||
// Skip the CallSyscall where possible.
|
||||
{
|
||||
MIPSOpcode op(inst.constant);
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc) {
|
||||
LI(R4, (uintptr_t)GetSyscallFuncPointer(op));
|
||||
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, 0);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
LI(R4, func);
|
||||
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
|
||||
} else {
|
||||
LI(R4, (int32_t)inst.constant);
|
||||
QuickCallFunction(&CallSyscall, SCRATCH2);
|
||||
}
|
||||
} else {
|
||||
LI(R4, (int32_t)inst.constant);
|
||||
QuickCallFunction(&CallSyscall, SCRATCH2);
|
||||
// Shouldn't get here.
|
||||
LI(R4, 0);
|
||||
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
LoadStaticRegisters();
|
||||
// This is always followed by an ExitToPC, where we check coreState.
|
||||
break;
|
||||
|
||||
case IROp::SyscallUnresolved:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
|
||||
LI(R4, inst.constant);
|
||||
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
LoadStaticRegisters();
|
||||
break;
|
||||
|
||||
case IROp::CallReplacement:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
@@ -275,4 +291,4 @@ void LoongArch64JitBackend::CompIR_ValidateAddress(IRInst inst) {
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace MIPSComp
|
||||
} // namespace MIPSComp
|
||||
@@ -408,4 +408,4 @@ LoongArch64Reg LoongArch64JitBackend::NormalizeR(IRReg rs, IRReg rd, LoongArch64
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace MIPSComp
|
||||
} // namespace MIPSComp
|
||||
@@ -1116,18 +1116,19 @@ static void RunUntilDowncountZeroFast(MIPSState *mips) {
|
||||
int cycleCount = 0;
|
||||
// Don't stop in a delay slot!
|
||||
do {
|
||||
if (!Memory::IsValid4AlignedAddress(mips->pc)) {
|
||||
Core_ExecException(mips->pc, mips->pc, ExecExceptionType::JUMP);
|
||||
const u32 pc = mips->pc;
|
||||
if (!Memory::IsValid4AlignedAddress(pc)) {
|
||||
Core_ExecException(pc, pc, ExecExceptionType::JUMP);
|
||||
return;
|
||||
}
|
||||
const MIPSOpcode op = MIPSOpcode(Memory::ReadUnchecked_U32(mips->pc));
|
||||
const MIPSOpcode op = MIPSOpcode(Memory::ReadUnchecked_U32(pc));
|
||||
|
||||
const bool wasInDelaySlot = mips->inDelaySlot;
|
||||
const int cycles = ExecInstruction(mips, op);
|
||||
if (cycles < 0) {
|
||||
// Not a recognized instruction (invalid encoding, or an unimplemented kernel-mode only instruction
|
||||
// with no interpreter implementation, e.g. tge/tlt/teq/).
|
||||
Core_ExecException(mips->pc, mips->pc, ExecExceptionType::ILLEGAL);
|
||||
Core_ExecException(pc, pc, ExecExceptionType::ILLEGAL);
|
||||
return;
|
||||
}
|
||||
cycleCount += cycles;
|
||||
|
||||
@@ -193,13 +193,20 @@ void RiscVJitBackend::CompIR_System(IRInst inst) {
|
||||
// Skip the CallSyscall where possible.
|
||||
{
|
||||
MIPSOpcode op(inst.constant);
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc) {
|
||||
LI(X10, (uintptr_t)GetSyscallFuncPointer(op));
|
||||
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, 0);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
LI(X10, (uintptr_t)func);
|
||||
QuickCallFunction((const u8 *)quickFunc, SCRATCH2);
|
||||
} else {
|
||||
LI(X10, (int32_t)inst.constant);
|
||||
QuickCallFunction(&CallSyscall, SCRATCH2);
|
||||
}
|
||||
} else {
|
||||
LI(X10, (int32_t)inst.constant);
|
||||
QuickCallFunction(&CallSyscall, SCRATCH2);
|
||||
// Shouldn't get here.
|
||||
LI(X10, 0);
|
||||
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -681,11 +681,17 @@ void Jit::Comp_Syscall(MIPSOpcode op)
|
||||
ABI_CallFunctionC(&CallSyscall, op.encoding);
|
||||
#else
|
||||
// Skip the CallSyscall where possible.
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc)
|
||||
ABI_CallFunctionP(quickFunc, (void *)GetSyscallFuncPointer(op));
|
||||
else
|
||||
ABI_CallFunctionC(&CallSyscall, op.encoding);
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, js.compilerPC);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
ABI_CallFunctionP(quickFunc, (void *)func);
|
||||
} else {
|
||||
ABI_CallFunctionC(&CallSyscall, op.encoding);
|
||||
}
|
||||
} else {
|
||||
ABI_CallFunctionC(&CallSyscallUnresolvedAtPC, js.compilerPC);
|
||||
}
|
||||
#endif
|
||||
|
||||
ApplyRoundingMode();
|
||||
|
||||
@@ -211,11 +211,18 @@ void X64JitBackend::CompIR_System(IRInst inst) {
|
||||
// Skip the CallSyscall where possible.
|
||||
{
|
||||
MIPSOpcode op(inst.constant);
|
||||
void *quickFunc = GetQuickSyscallFunc(op);
|
||||
if (quickFunc) {
|
||||
ABI_CallFunctionP((const u8 *)quickFunc, (void *)GetSyscallFuncPointer(op));
|
||||
const HLEFunction *func = GetSyscallFunctionData(op, 0);
|
||||
if (func) {
|
||||
void *quickFunc = GetQuickSyscallFunc(func, op);
|
||||
if (quickFunc) {
|
||||
ABI_CallFunctionP((const u8 *)quickFunc, (void *)func);
|
||||
} else {
|
||||
ABI_CallFunctionC((const u8 *)&CallSyscall, inst.constant);
|
||||
}
|
||||
} else {
|
||||
ABI_CallFunctionC((const u8 *)&CallSyscall, inst.constant);
|
||||
_dbg_assert_(false);
|
||||
// Shouldn't get here, this should be resolved during ->IR compilation.
|
||||
ABI_CallFunctionC((const u8 *)&CallSyscallUnresolvedAtPC, 0);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
@@ -225,6 +232,15 @@ void X64JitBackend::CompIR_System(IRInst inst) {
|
||||
// This is always followed by an ExitToPC, where we check coreState.
|
||||
break;
|
||||
|
||||
case IROp::SyscallUnresolved:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
|
||||
ABI_CallFunctionC((const u8 *)&CallSyscallUnresolvedAtPC, inst.constant);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
LoadStaticRegisters();
|
||||
break;
|
||||
|
||||
case IROp::CallReplacement:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
|
||||
@@ -296,6 +296,10 @@ inline void MemcpyUnchecked(const u32 to_address, const u32 from_address, const
|
||||
MemcpyUnchecked(GetPointerWriteUnchecked(to_address), from_address, len);
|
||||
}
|
||||
|
||||
inline bool AddressesEqualAfterMask(const u32 address1, const u32 address2) {
|
||||
return (address1 & 0x3FFFFFFF) == (address2 & 0x3FFFFFFF);
|
||||
}
|
||||
|
||||
// Applies to user mode.
|
||||
// Without a length, IsValidAddress is generally semi-meaningless, unless it's about a single byte access. For larger accesses, use IsValid4AlignedAddress
|
||||
// etc when appropriate, or for longer sizes, use IsValidRange or IsValid4AlignedRange for example. Checking aligned-ness helps avoid the problem
|
||||
|
||||
+3
-3
@@ -93,7 +93,7 @@ static FileLoader *g_loadedFile;
|
||||
static std::mutex loadingLock;
|
||||
static std::thread g_loadingThread;
|
||||
|
||||
bool coreCollectDebugStats = false;
|
||||
bool g_coreCollectDebugStats = false;
|
||||
static int coreCollectDebugStatsCounter = 0;
|
||||
|
||||
static volatile CPUThreadState cpuThreadState = CPU_THREAD_NOT_RUNNING;
|
||||
@@ -608,8 +608,8 @@ void UpdateLoadedFile(FileLoader *fileLoader) {
|
||||
|
||||
void PSP_UpdateDebugStats(bool collectStats) {
|
||||
bool newState = collectStats || coreCollectDebugStatsCounter > 0;
|
||||
if (coreCollectDebugStats != newState) {
|
||||
coreCollectDebugStats = newState;
|
||||
if (g_coreCollectDebugStats != newState) {
|
||||
g_coreCollectDebugStats = newState;
|
||||
mipsr4k.ClearJitCache();
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -98,7 +98,7 @@ Path GetSysDirectory(PSPDirectories directoryType);
|
||||
|
||||
bool CreateSysDirectories();
|
||||
|
||||
extern bool coreCollectDebugStats;
|
||||
extern bool g_coreCollectDebugStats;
|
||||
|
||||
inline CoreParameter &PSP_CoreParameter() {
|
||||
extern CoreParameter g_CoreParameter;
|
||||
|
||||
@@ -1155,7 +1155,7 @@ void DrawEngineCommon::DepthRasterSubmitRaw(GEPrimitiveType prim, const VertexDe
|
||||
return;
|
||||
}
|
||||
|
||||
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, coreCollectDebugStats);
|
||||
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, g_coreCollectDebugStats);
|
||||
|
||||
// Decode.
|
||||
int numDecoded = 0;
|
||||
@@ -1197,7 +1197,7 @@ void DrawEngineCommon::DepthRasterPredecoded(GEPrimitiveType prim, const void *i
|
||||
return;
|
||||
}
|
||||
|
||||
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, coreCollectDebugStats);
|
||||
TimeCollector collectStat(&gpuStats.perFrame.msPrepareDepth, g_coreCollectDebugStats);
|
||||
|
||||
// Make sure these have already been indexed away.
|
||||
_dbg_assert_(prim != GE_PRIM_TRIANGLE_STRIP && prim != GE_PRIM_TRIANGLE_FAN);
|
||||
@@ -1234,7 +1234,7 @@ void DrawEngineCommon::FlushQueuedDepth() {
|
||||
rasterTimeStart_ = 0.0;
|
||||
}
|
||||
|
||||
const bool collectStats = coreCollectDebugStats;
|
||||
const bool collectStats = g_coreCollectDebugStats;
|
||||
const bool lowQ = g_Config.iDepthRasterMode == (int)DepthRasterMode::LOW_QUALITY;
|
||||
for (const auto &draw : depthDraws_) {
|
||||
int *tx = depthScreenVerts_;
|
||||
|
||||
+3
-3
@@ -745,7 +745,7 @@ DLResult GPUCommon::ProcessDLQueue() {
|
||||
}
|
||||
}
|
||||
|
||||
TimeCollector collectStat(&gpuStats.perFrame.msProcessingDisplayLists, coreCollectDebugStats);
|
||||
TimeCollector collectStat(&gpuStats.perFrame.msProcessingDisplayLists, g_coreCollectDebugStats);
|
||||
|
||||
auto GetNextListIndex = [&]() -> int {
|
||||
if (dlQueue.empty())
|
||||
@@ -873,7 +873,7 @@ DLResult GPUCommon::ProcessDLQueue() {
|
||||
|
||||
currentList = nullptr;
|
||||
|
||||
if (coreCollectDebugStats) {
|
||||
if (g_coreCollectDebugStats) {
|
||||
gpuStats.perFrame.otherGPUCycles += cyclesExecuted;
|
||||
}
|
||||
|
||||
@@ -1394,7 +1394,7 @@ void GPUCommon::FastLoadBoneMatrix(u32 target) {
|
||||
|
||||
cyclesExecuted += 2 * 14; // one to reset the counter, 12 to load the matrix, and a return.
|
||||
|
||||
if (coreCollectDebugStats) {
|
||||
if (g_coreCollectDebugStats) {
|
||||
gpuStats.perFrame.otherGPUCycles += 2 * 14;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -563,8 +563,10 @@ void BinManager::Flush(const char *reason) {
|
||||
return;
|
||||
|
||||
double st = 0.0;
|
||||
if (coreCollectDebugStats)
|
||||
const bool collectDebugStats = g_coreCollectDebugStats;
|
||||
if (collectDebugStats) {
|
||||
st = time_now_d();
|
||||
}
|
||||
Drain(true);
|
||||
waitable_->Wait();
|
||||
taskRanges_.clear();
|
||||
@@ -584,16 +586,17 @@ void BinManager::Flush(const char *reason) {
|
||||
queueRange_.x2 = 0;
|
||||
queueRange_.y2 = 0;
|
||||
|
||||
for (auto &pending : pendingWrites_)
|
||||
for (BinDirtyRange &pending : pendingWrites_) {
|
||||
pending.base = 0;
|
||||
}
|
||||
pendingOverlap_ = false;
|
||||
pendingReads_.clear();
|
||||
|
||||
// We'll need to set the pending writes and reads again, since we just flushed it.
|
||||
dirty_ |= SoftDirty::BINNER_RANGE | SoftDirty::BINNER_OVERLAP;
|
||||
|
||||
if (coreCollectDebugStats) {
|
||||
double et = time_now_d();
|
||||
if (collectDebugStats) {
|
||||
const double et = time_now_d();
|
||||
flushReasonTimes_[reason] += et - st;
|
||||
if (et - st > slowestFlushTime_) {
|
||||
slowestFlushTime_ = et - st;
|
||||
@@ -611,7 +614,7 @@ void BinManager::OptimizePendingStates(uint16_t first, uint16_t last) {
|
||||
last--;
|
||||
}
|
||||
|
||||
int count = (QUEUED_STATES + last - first) % QUEUED_STATES + 1;
|
||||
const int count = (QUEUED_STATES + last - first) % QUEUED_STATES + 1;
|
||||
for (int i = 0; i < count; ++i) {
|
||||
size_t pos = (first + i) % QUEUED_STATES;
|
||||
OptimizeRasterState(&states_[pos]);
|
||||
|
||||
@@ -154,7 +154,8 @@ void DeveloperToolsScreen::CreateGeneralTab(UI::LinearLayout *list) {
|
||||
core->HideChoice(1);
|
||||
core->HideChoice(3);
|
||||
}
|
||||
// TODO: Enable "JIT using IR" on more architectures.
|
||||
// TODO: Enable "JIT using IR" on more architectures. ARM32 needs more testing.
|
||||
// Note also that Loongarch and RISC-V only have a jit-ir backend, and it's used for the JIT option, so the fourth option isn't shown.
|
||||
#if !PPSSPP_ARCH(X86) && !PPSSPP_ARCH(AMD64) && !PPSSPP_ARCH(ARM64)
|
||||
core->HideChoice(3);
|
||||
#endif
|
||||
|
||||
@@ -1603,8 +1603,6 @@ public class PpssppActivity extends AppCompatActivity implements SensorEventList
|
||||
return true;
|
||||
} else if (command.equals("showKeyboard") && surfView != null) {
|
||||
InputMethodManager inputMethodManager = (InputMethodManager) getSystemService(Context.INPUT_METHOD_SERVICE);
|
||||
// No idea what the point of the ApplicationWindowToken is or if it
|
||||
// matters where we get it from...
|
||||
inputMethodManager.showSoftInput(surfView, InputMethodManager.SHOW_IMPLICIT);
|
||||
return true;
|
||||
} else if (command.equals("hideKeyboard") && surfView != null) {
|
||||
|
||||
Reference in new issue
Block a user