diff --git a/Core/MIPS/LoongArch64/LoongArch64CompALU.cpp b/Core/MIPS/LoongArch64/LoongArch64CompALU.cpp index 9ced74ecc3..9064c8f68f 100644 --- a/Core/MIPS/LoongArch64/LoongArch64CompALU.cpp +++ b/Core/MIPS/LoongArch64/LoongArch64CompALU.cpp @@ -598,16 +598,20 @@ void LoongArch64JitBackend::CompIR_Div(IRInst inst) { { // Start with divide by zero, the quotient and remainder are arbitrary numbers. FixupBranch skipNonZero = BNEZ(denomReg); - // Clear the arbitrary number - XOR(regs_.R(IRREG_LO), regs_.R(IRREG_LO), regs_.R(IRREG_LO)); - // Replace remainder to numReg - BSTRINS_D(regs_.R(IRREG_LO), numReg, 63, 32); + // Clear the arbitrary number + XOR(regs_.R(IRREG_LO), regs_.R(IRREG_LO), regs_.R(IRREG_LO)); + // Replace remainder to numReg + BSTRINS_D(regs_.R(IRREG_LO), numReg, 63, 32); FixupBranch keepNegOne = BGE(numReg, R_ZERO); - // Replace quotient with 1. - ADDI_D(regs_.R(IRREG_LO), regs_.R(IRREG_LO), 1); + // Replace quotient with 1. The low half is zero, so setting the bit is enough. + ORI(regs_.R(IRREG_LO), regs_.R(IRREG_LO), 1); + FixupBranch skipNegOne = B(); SetJumpTarget(keepNegOne); - // Replace quotient with -1. - ADDI_D(regs_.R(IRREG_LO), regs_.R(IRREG_LO), -1); + // Replace quotient with -1. Insert it - subtracting one would borrow into the + // remainder sitting in the high half. + LI(R_RA, -1); + BSTRINS_D(regs_.R(IRREG_LO), R_RA, 31, 0); + SetJumpTarget(skipNegOne); SetJumpTarget(skipNonZero); // For overflow, LoongArch sets LO right, but remainder to zero. diff --git a/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp b/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp index 44e27f114b..7fae145e9e 100644 --- a/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp +++ b/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp @@ -80,6 +80,11 @@ void LoongArch64JitBackend::CompIR_FCondAssign(IRInst inst) { CONDITIONAL_DISABLE; regs_.Map(inst); + + // Allocate this before the branch below. Allocating can spill a register, and that store + // would then sit on the unordered-only path while the regcache believes it always ran. + LoongArch64Reg isSrc1LowerReg = regs_.GetAndLockTempGPR(); + FCMP_COND_S(FCC0, regs_.F(inst.src1), regs_.F(inst.src2), LoongArch64Fcond::CUN); MOVCF2GR(SCRATCH1, FCC0); FixupBranch unordered = BNEZ(SCRATCH1); @@ -109,7 +114,6 @@ void LoongArch64JitBackend::CompIR_FCondAssign(IRInst inst) { AND(R_RA, SCRATCH1, SCRATCH2); SRLI_W(R_RA, R_RA, 31); - LoongArch64Reg isSrc1LowerReg = regs_.GetAndLockTempGPR(); SLT(isSrc1LowerReg, SCRATCH1, SCRATCH2); // Flip the flag (to reverse the min/max) based on if both were negative. XOR(isSrc1LowerReg, isSrc1LowerReg, R_RA); @@ -231,8 +235,8 @@ void LoongArch64JitBackend::CompIR_FCvt(IRInst inst) { case IROp::FCvtScaledWS: regs_.Map(inst); - // Prepare for the NAN result - QuickFLI(32, SCRATCHF1, (uint32_t)(0x7FFFFFFF), SCRATCH1); + // Prepare for the NAN result, which the FSELs below pick out of SCRATCHF2. + QuickFLI(32, SCRATCHF2, (uint32_t)(0x7FFFFFFF), SCRATCH1); // Prepare the multiplier. QuickFLI(32, SCRATCHF1, (float)(1UL << (inst.src2 & 0x1F)), SCRATCH1); diff --git a/Core/MIPS/RiscV/RiscVAsm.cpp b/Core/MIPS/RiscV/RiscVAsm.cpp index 396c45ec20..560b30b66c 100644 --- a/Core/MIPS/RiscV/RiscVAsm.cpp +++ b/Core/MIPS/RiscV/RiscVAsm.cpp @@ -78,6 +78,9 @@ void RiscVJitBackend::GenerateFixedCode(MIPSState *mipsState) { { // Not sure if RISC-V has any flush to zero capability? Leaving it off for now... LWU(SCRATCH2, CTXREG, offsetof(MIPSState, fcr31)); + // FSRM only takes the low three bits, and fcr31 has plenty of other bits set (the flags + // at 2-6, for one) - without this we'd hand it a reserved rounding mode. + ANDI(SCRATCH2, SCRATCH2, 3); // We can skip if the rounding mode is nearest (0) and flush is not set. // (as restoreRoundingMode cleared it out anyway) diff --git a/Core/MIPS/RiscV/RiscVCompFPU.cpp b/Core/MIPS/RiscV/RiscVCompFPU.cpp index eb26e5caed..01f58e1374 100644 --- a/Core/MIPS/RiscV/RiscVCompFPU.cpp +++ b/Core/MIPS/RiscV/RiscVCompFPU.cpp @@ -83,6 +83,13 @@ void RiscVJitBackend::CompIR_FCondAssign(IRInst inst) { // FMin and FMax are used by VFPU and handle NAN/INF as just a larger exponent. regs_.Map(inst); + + // Allocate this before the branch below. Allocating can spill a register, and that store + // would then sit on the NAN-only path while the regcache believes it always ran. + RiscVReg isSrc1LowerReg = INVALID_REG; + if (!cpu_info.RiscV_Zbb) + isSrc1LowerReg = regs_.GetAndLockTempGPR(); + FCLASS(32, SCRATCH1, regs_.F(inst.src1)); FCLASS(32, SCRATCH2, regs_.F(inst.src2)); @@ -115,7 +122,6 @@ void RiscVJitBackend::CompIR_FCondAssign(IRInst inst) { MAX(SCRATCH1, SCRATCH1, SCRATCH2); SetJumpTarget(skipSwapCompare); } else { - RiscVReg isSrc1LowerReg = regs_.GetAndLockTempGPR(); SLT(isSrc1LowerReg, SCRATCH1, SCRATCH2); // Flip the flag (to reverse the min/max) based on if both were negative. XOR(isSrc1LowerReg, isSrc1LowerReg, R_RA); diff --git a/Core/MIPS/RiscV/RiscVCompSystem.cpp b/Core/MIPS/RiscV/RiscVCompSystem.cpp index 564cb5fea3..ec7b3c01d0 100644 --- a/Core/MIPS/RiscV/RiscVCompSystem.cpp +++ b/Core/MIPS/RiscV/RiscVCompSystem.cpp @@ -216,6 +216,16 @@ void RiscVJitBackend::CompIR_System(IRInst inst) { // This is always followed by an ExitToPC, where we check coreState. break; + case IROp::SyscallUnresolved: + FlushAll(); + SaveStaticRegisters(); + WriteDebugProfilerStatus(IRProfilerStatus::SYSCALL); + LI(X10, (int32_t)inst.constant); + QuickCallFunction(&CallSyscallUnresolvedAtPC, SCRATCH2); + WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT); + LoadStaticRegisters(); + break; + case IROp::CallReplacement: FlushAll(); SaveStaticRegisters(); diff --git a/Core/MIPS/RiscV/RiscVJit.cpp b/Core/MIPS/RiscV/RiscVJit.cpp index 87181c989f..7d9d9fe7a1 100644 --- a/Core/MIPS/RiscV/RiscVJit.cpp +++ b/Core/MIPS/RiscV/RiscVJit.cpp @@ -211,6 +211,8 @@ void RiscVJitBackend::OverwriteExit(int srcOffset, int len, int block_num) { RiscVEmitter emitter(GetBasePtr() + srcOffset, writable); emitter.QuickJ(SCRATCH1, GetBasePtr() + nativeBlock->checkedOffset); int bytesWritten = (int)(emitter.GetWritableCodePtr() - writable); + // QuickJ is 4, 8 or 12 bytes depending on the distance, and the hole is only 8. + _dbg_assert_(bytesWritten <= MIN_BLOCK_EXIT_LEN); if (bytesWritten < len) emitter.ReserveCodeSpace(len - bytesWritten); emitter.FlushIcache(); diff --git a/Core/MIPS/RiscV/RiscVRegCache.cpp b/Core/MIPS/RiscV/RiscVRegCache.cpp index 0e5965bf36..9c91ae7883 100644 --- a/Core/MIPS/RiscV/RiscVRegCache.cpp +++ b/Core/MIPS/RiscV/RiscVRegCache.cpp @@ -134,17 +134,25 @@ void RiscVRegCache::EmitSaveStaticRegisters() { void RiscVRegCache::FlushBeforeCall() { // These registers are not preserved by function calls. - // They match between X0 and F0, conveniently. + // X0-X4 are zero/ra/sp/gp/tp, which we never allocate, but F0-F4 (ft0-ft4) are caller-saved + // and we do allocate them - so the two don't line up and need separate loops. for (int i = 5; i <= 7; ++i) { FlushNativeReg(X0 + i); - FlushNativeReg(F0 + i); } for (int i = 10; i <= 17; ++i) { FlushNativeReg(X0 + i); - FlushNativeReg(F0 + i); } for (int i = 28; i <= 31; ++i) { FlushNativeReg(X0 + i); + } + + for (int i = 0; i <= 7; ++i) { + FlushNativeReg(F0 + i); + } + for (int i = 10; i <= 17; ++i) { + FlushNativeReg(F0 + i); + } + for (int i = 28; i <= 31; ++i) { FlushNativeReg(F0 + i); } }