mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
RISC-V, LoongArch64: fix caller-saved FPRs, div by zero, and more
RiscVRegCache::FlushBeforeCall claimed the caller-saved X and F registers "match between X0 and F0". They don't: X0-X4 are zero/ra/sp/gp/tp and are never allocated, but F0-F4 (ft0-ft4) are caller-saved and are in the FPR allocation order. A block that spilled into them and then called out (for vsin or vcos, say) had the callee destroy registers the cache still thought were live and dirty. LoongArch and arm64 get this right. The LoongArch signed div-by-zero sequence applied its ADDI_D(-1) after the jump target, so it ran on both paths: a negative numerator ended up with a quotient of 0 instead of 1, and a non-negative one borrowed into the remainder in the high half, giving hi = num - 1. Insert the -1 into the low word rather than subtracting, and branch around the other case. FCvtScaledWS loaded 0x7FFFFFFF into SCRATCHF1 and then immediately overwrote it with the multiplier, while the FSELs picked SCRATCHF2 - a register the function never writes. A NAN input produced whatever was left there. Also: - RISC-V had no SyscallUnresolved case, falling through to INVALIDOP, which is a live assert in release builds. Added it, matching arm64. - applyRoundingMode_ fed fcr31 straight to FSRM, which only takes three bits - the flags at 2-6 could turn a mode of 0 into 4. Mask with 3 first, as MIPS.cpp and LoongArch64Asm.cpp do. - OverwriteExit lacked arm64's assert that the exit fits the 8-byte hole, which matters more here since QuickJ is 4, 8 or 12 bytes. - Both backends allocated the FMin/FMax temp GPR inside the conditional NAN path, where a spill store would only run on one side while the regcache assumed it always ran. Hoisted above the branch. Both backends build now, and pspautotests and the unit tests pass under qemu on each - but nothing in the suite reaches the paths above, so the two worth checking were checked by running the emitted sequences directly. The div-by-zero one is wrong for all eight numerators tried and right for all eight after. The rounding one turns a guest mode of 0 into RMM whenever fcr31 has its inexact flag set, so 0.5 converts to 1 rather than 0. The rest are still by inspection: they need register pressure, an unresolved import, or a debug build to reach. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
This commit is contained in:
1 parent
ab6ccbd08c
commit
28b5d0ee5c
7 files changed
+52
-15
No files matched your search
@@ -598,16 +598,20 @@ void LoongArch64JitBackend::CompIR_Div(IRInst inst) {
|
||||
{
|
||||
// Start with divide by zero, the quotient and remainder are arbitrary numbers.
|
||||
FixupBranch skipNonZero = BNEZ(denomReg);
|
||||
// Clear the arbitrary number
|
||||
XOR(regs_.R(IRREG_LO), regs_.R(IRREG_LO), regs_.R(IRREG_LO));
|
||||
// Replace remainder to numReg
|
||||
BSTRINS_D(regs_.R(IRREG_LO), numReg, 63, 32);
|
||||
// Clear the arbitrary number
|
||||
XOR(regs_.R(IRREG_LO), regs_.R(IRREG_LO), regs_.R(IRREG_LO));
|
||||
// Replace remainder to numReg
|
||||
BSTRINS_D(regs_.R(IRREG_LO), numReg, 63, 32);
|
||||
FixupBranch keepNegOne = BGE(numReg, R_ZERO);
|
||||
// Replace quotient with 1.
|
||||
ADDI_D(regs_.R(IRREG_LO), regs_.R(IRREG_LO), 1);
|
||||
// Replace quotient with 1. The low half is zero, so setting the bit is enough.
|
||||
ORI(regs_.R(IRREG_LO), regs_.R(IRREG_LO), 1);
|
||||
FixupBranch skipNegOne = B();
|
||||
SetJumpTarget(keepNegOne);
|
||||
// Replace quotient with -1.
|
||||
ADDI_D(regs_.R(IRREG_LO), regs_.R(IRREG_LO), -1);
|
||||
// Replace quotient with -1. Insert it - subtracting one would borrow into the
|
||||
// remainder sitting in the high half.
|
||||
LI(R_RA, -1);
|
||||
BSTRINS_D(regs_.R(IRREG_LO), R_RA, 31, 0);
|
||||
SetJumpTarget(skipNegOne);
|
||||
SetJumpTarget(skipNonZero);
|
||||
|
||||
// For overflow, LoongArch sets LO right, but remainder to zero.
|
||||
|
||||
@@ -80,6 +80,11 @@ void LoongArch64JitBackend::CompIR_FCondAssign(IRInst inst) {
|
||||
CONDITIONAL_DISABLE;
|
||||
|
||||
regs_.Map(inst);
|
||||
|
||||
// Allocate this before the branch below. Allocating can spill a register, and that store
|
||||
// would then sit on the unordered-only path while the regcache believes it always ran.
|
||||
LoongArch64Reg isSrc1LowerReg = regs_.GetAndLockTempGPR();
|
||||
|
||||
FCMP_COND_S(FCC0, regs_.F(inst.src1), regs_.F(inst.src2), LoongArch64Fcond::CUN);
|
||||
MOVCF2GR(SCRATCH1, FCC0);
|
||||
FixupBranch unordered = BNEZ(SCRATCH1);
|
||||
@@ -109,7 +114,6 @@ void LoongArch64JitBackend::CompIR_FCondAssign(IRInst inst) {
|
||||
AND(R_RA, SCRATCH1, SCRATCH2);
|
||||
SRLI_W(R_RA, R_RA, 31);
|
||||
|
||||
LoongArch64Reg isSrc1LowerReg = regs_.GetAndLockTempGPR();
|
||||
SLT(isSrc1LowerReg, SCRATCH1, SCRATCH2);
|
||||
// Flip the flag (to reverse the min/max) based on if both were negative.
|
||||
XOR(isSrc1LowerReg, isSrc1LowerReg, R_RA);
|
||||
@@ -231,8 +235,8 @@ void LoongArch64JitBackend::CompIR_FCvt(IRInst inst) {
|
||||
|
||||
case IROp::FCvtScaledWS:
|
||||
regs_.Map(inst);
|
||||
// Prepare for the NAN result
|
||||
QuickFLI(32, SCRATCHF1, (uint32_t)(0x7FFFFFFF), SCRATCH1);
|
||||
// Prepare for the NAN result, which the FSELs below pick out of SCRATCHF2.
|
||||
QuickFLI(32, SCRATCHF2, (uint32_t)(0x7FFFFFFF), SCRATCH1);
|
||||
// Prepare the multiplier.
|
||||
QuickFLI(32, SCRATCHF1, (float)(1UL << (inst.src2 & 0x1F)), SCRATCH1);
|
||||
|
||||
|
||||
@@ -78,6 +78,9 @@ void RiscVJitBackend::GenerateFixedCode(MIPSState *mipsState) {
|
||||
{
|
||||
// Not sure if RISC-V has any flush to zero capability? Leaving it off for now...
|
||||
LWU(SCRATCH2, CTXREG, offsetof(MIPSState, fcr31));
|
||||
// FSRM only takes the low three bits, and fcr31 has plenty of other bits set (the flags
|
||||
// at 2-6, for one) - without this we'd hand it a reserved rounding mode.
|
||||
ANDI(SCRATCH2, SCRATCH2, 3);
|
||||
|
||||
// We can skip if the rounding mode is nearest (0) and flush is not set.
|
||||
// (as restoreRoundingMode cleared it out anyway)
|
||||
|
||||
@@ -83,6 +83,13 @@ void RiscVJitBackend::CompIR_FCondAssign(IRInst inst) {
|
||||
|
||||
// FMin and FMax are used by VFPU and handle NAN/INF as just a larger exponent.
|
||||
regs_.Map(inst);
|
||||
|
||||
// Allocate this before the branch below. Allocating can spill a register, and that store
|
||||
// would then sit on the NAN-only path while the regcache believes it always ran.
|
||||
RiscVReg isSrc1LowerReg = INVALID_REG;
|
||||
if (!cpu_info.RiscV_Zbb)
|
||||
isSrc1LowerReg = regs_.GetAndLockTempGPR();
|
||||
|
||||
FCLASS(32, SCRATCH1, regs_.F(inst.src1));
|
||||
FCLASS(32, SCRATCH2, regs_.F(inst.src2));
|
||||
|
||||
@@ -115,7 +122,6 @@ void RiscVJitBackend::CompIR_FCondAssign(IRInst inst) {
|
||||
MAX(SCRATCH1, SCRATCH1, SCRATCH2);
|
||||
SetJumpTarget(skipSwapCompare);
|
||||
} else {
|
||||
RiscVReg isSrc1LowerReg = regs_.GetAndLockTempGPR();
|
||||
SLT(isSrc1LowerReg, SCRATCH1, SCRATCH2);
|
||||
// Flip the flag (to reverse the min/max) based on if both were negative.
|
||||
XOR(isSrc1LowerReg, isSrc1LowerReg, R_RA);
|
||||
|
||||
@@ -216,6 +216,16 @@ void RiscVJitBackend::CompIR_System(IRInst inst) {
|
||||
// This is always followed by an ExitToPC, where we check coreState.
|
||||
break;
|
||||
|
||||
case IROp::SyscallUnresolved:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL);
|
||||
LI(X10, (int32_t)inst.constant);
|
||||
QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
LoadStaticRegisters();
|
||||
break;
|
||||
|
||||
case IROp::CallReplacement:
|
||||
FlushAll();
|
||||
SaveStaticRegisters();
|
||||
|
||||
@@ -211,6 +211,8 @@ void RiscVJitBackend::OverwriteExit(int srcOffset, int len, int block_num) {
|
||||
RiscVEmitter emitter(GetBasePtr() + srcOffset, writable);
|
||||
emitter.QuickJ(SCRATCH1, GetBasePtr() + nativeBlock->checkedOffset);
|
||||
int bytesWritten = (int)(emitter.GetWritableCodePtr() - writable);
|
||||
// QuickJ is 4, 8 or 12 bytes depending on the distance, and the hole is only 8.
|
||||
_dbg_assert_(bytesWritten <= MIN_BLOCK_EXIT_LEN);
|
||||
if (bytesWritten < len)
|
||||
emitter.ReserveCodeSpace(len - bytesWritten);
|
||||
emitter.FlushIcache();
|
||||
|
||||
@@ -134,17 +134,25 @@ void RiscVRegCache::EmitSaveStaticRegisters() {
|
||||
|
||||
void RiscVRegCache::FlushBeforeCall() {
|
||||
// These registers are not preserved by function calls.
|
||||
// They match between X0 and F0, conveniently.
|
||||
// X0-X4 are zero/ra/sp/gp/tp, which we never allocate, but F0-F4 (ft0-ft4) are caller-saved
|
||||
// and we do allocate them - so the two don't line up and need separate loops.
|
||||
for (int i = 5; i <= 7; ++i) {
|
||||
FlushNativeReg(X0 + i);
|
||||
FlushNativeReg(F0 + i);
|
||||
}
|
||||
for (int i = 10; i <= 17; ++i) {
|
||||
FlushNativeReg(X0 + i);
|
||||
FlushNativeReg(F0 + i);
|
||||
}
|
||||
for (int i = 28; i <= 31; ++i) {
|
||||
FlushNativeReg(X0 + i);
|
||||
}
|
||||
|
||||
for (int i = 0; i <= 7; ++i) {
|
||||
FlushNativeReg(F0 + i);
|
||||
}
|
||||
for (int i = 10; i <= 17; ++i) {
|
||||
FlushNativeReg(F0 + i);
|
||||
}
|
||||
for (int i = 28; i <= 31; ++i) {
|
||||
FlushNativeReg(F0 + i);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user