Merge pull request #22351 from hrydgard/jit-call-optimizations

JIT function-call optimizations
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-09-24 16:59:39 -06:00
commit a50fb6071f
27 files changed
+384 -66

No files matched your search

+23 -31
View File
@@ -30,6 +30,18 @@ alignas(32) static u8 saved_gpr_state[16 * 8];
static u16 saved_mxcsr;
#endif
#if PPSSPP_ARCH(AMD64)
// The registers a call may clobber, apart from RAX and XMM0-1, which the JIT uses as scratch.
// The callee preserves the rest (RBX, RBP, R12-R15, and XMM6-15 on Windows), so there's no need to.
#ifdef _WIN32
const int THUNK_LAST_XMM = 5;
const Gen::X64Reg thunkGPRs[] = { Gen::RCX, Gen::RDX, Gen::R8, Gen::R9, Gen::R10, Gen::R11 };
#else
const int THUNK_LAST_XMM = 15;
const Gen::X64Reg thunkGPRs[] = { Gen::RCX, Gen::RDX, Gen::R8, Gen::R9, Gen::R10, Gen::R11, Gen::RSI, Gen::RDI };
#endif
#endif
} // namespace
using namespace Gen;
@@ -46,21 +58,12 @@ void ThunkManager::Init()
BeginWrite(512);
save_regs = GetCodePtr();
#if PPSSPP_ARCH(AMD64)
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
for (int i = 2; i <= THUNK_LAST_XMM; i++)
MOVAPS(MDisp(RSP, stackOffset + (i - 2) * 16), (X64Reg)(XMM0 + i));
stackPosition = (ABI_GetNumXMMRegs() - 2) * 2;
stackPosition = (THUNK_LAST_XMM - 1) * 2;
STMXCSR(MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RCX));
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RDX));
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R8) );
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R9) );
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R10));
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R11));
#ifndef _WIN32
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RSI));
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RDI));
#endif
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RBX));
for (X64Reg reg : thunkGPRs)
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(reg));
#else
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
MOVAPS(M(saved_fp_state + i * 16), (X64Reg)(XMM0 + i));
@@ -72,21 +75,12 @@ void ThunkManager::Init()
load_regs = GetCodePtr();
#if PPSSPP_ARCH(AMD64)
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
for (int i = 2; i <= THUNK_LAST_XMM; i++)
MOVAPS((X64Reg)(XMM0 + i), MDisp(RSP, stackOffset + (i - 2) * 16));
stackPosition = (ABI_GetNumXMMRegs() - 2) * 2;
stackPosition = (THUNK_LAST_XMM - 1) * 2;
LDMXCSR(MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(RCX), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(RDX), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(R8) , MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(R9) , MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(R10), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(R11), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
#ifndef _WIN32
MOV(64, R(RSI), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
MOV(64, R(RDI), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
#endif
MOV(64, R(RBX), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
for (X64Reg reg : thunkGPRs)
MOV(64, R(reg), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
#else
LDMXCSR(M(&saved_mxcsr));
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
@@ -112,15 +106,13 @@ void ThunkManager::Shutdown()
int ThunkManager::ThunkBytesNeeded()
{
int space = (ABI_GetNumXMMRegs() - 2) * 16;
#if PPSSPP_ARCH(AMD64)
int space = (THUNK_LAST_XMM - 1) * 16;
// MXCSR
space += 8;
space += 7 * 8;
#ifndef _WIN32
space += 2 * 8;
#endif
space += (int)ARRAY_SIZE(thunkGPRs) * 8;
#else
int space = (ABI_GetNumXMMRegs() - 2) * 16;
// MXCSR
space += 4;
space += 2 * 4;
+10
View File
@@ -2250,6 +2250,16 @@ namespace MIPSComp
if (vd2 >= 0)
GetVectorRegs(dregs2, sz, vd2);
GetVectorRegs(&sreg, V_Single, vs);
// With the angle in a destination lane, the cosine is taken of what was written there.
// The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot.
for (int i = 0; i < n; i++) {
if (dregs[i] == sreg) {
DISABLE;
}
if (vd2 >= 0 && dregs2[i] == sreg) {
vd2 = -1;
}
}
int imm = (op >> 16) & 0x1f;
+62 -18
View File
@@ -950,6 +950,16 @@ namespace MIPSComp {
// The VFPU special functions call the exact C versions. The lanes stay in S8-S11 across the
// calls (callee-saved), and the results are stored to the destinations' homes.
// After fpr.FlushBeforeCall(), a VFPU register is either still mapped (in S8-S15) or its value is
// in memory. Loads it into dest either way.
void Arm64Jit::LoadVAfterCallFlush(ARM64Reg dest, u8 vreg) {
if (fpr.IsMappedV(vreg)) {
fp.FMOV(dest, fpr.V(vreg));
} else {
fp.LDR(32, INDEX_UNSIGNED, dest, CTXREG, fpr.GetMipsRegOffsetV(vreg));
}
}
void Arm64Jit::CompVV2OpCall(MIPSOpcode op) {
if (js.HasSPrefix()) {
DISABLE;
@@ -978,23 +988,39 @@ namespace MIPSComp {
GetVectorRegs(sregs, sz, _VS);
GetVectorRegs(dregs, sz, _VD);
// Values in S8-S15 survive the calls, so only the rest is flushed.
gpr.FlushBeforeCall();
fpr.FlushAll();
fpr.FlushBeforeCall();
for (int i = 0; i < n; i++) {
fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
}
for (int i = 0; i < n; i++) {
fp.FMOV(S0, (ARM64Reg)(S8 + i));
if (n == 1) {
LoadVAfterCallFlush(S0, sregs[0]);
QuickCallFunction(SCRATCH2_64, func);
if (negate) {
fp.FNEG((ARM64Reg)(S8 + i), S0);
} else {
} else {
// The lanes wait in S8 and up between the calls, so free those too.
for (int i = 0; i < n; i++) {
fpr.FlushArmReg((ARM64Reg)(S8 + i));
}
for (int i = 0; i < n; i++) {
LoadVAfterCallFlush((ARM64Reg)(S8 + i), sregs[i]);
}
for (int i = 0; i < n; i++) {
fp.FMOV(S0, (ARM64Reg)(S8 + i));
QuickCallFunction(SCRATCH2_64, func);
fp.FMOV((ARM64Reg)(S8 + i), S0);
}
// Into S0-S3, which the cache never allocates, before mapping the destinations.
for (int i = 0; i < n; i++) {
fp.FMOV((ARM64Reg)(S0 + i), (ARM64Reg)(S8 + i));
}
}
for (int i = 0; i < n; i++) {
fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
fpr.MapRegV(dregs[i], MAP_DIRTY | MAP_NOINIT);
if (negate) {
fp.FNEG(fpr.V(dregs[i]), (ARM64Reg)(S0 + i));
} else {
fp.FMOV(fpr.V(dregs[i]), (ARM64Reg)(S0 + i));
}
}
ApplyPrefixD(dregs, sz);
@@ -1065,20 +1091,28 @@ namespace MIPSComp {
GetVectorRegs(sregs, sz, _VS);
GetVectorRegs(dregs, outSz, _VD);
// The inputs wait in S8-S9 and the results in S10-S13 between the calls (callee-saved), so
// those are flushed along with the registers a call clobbers.
gpr.FlushBeforeCall();
fpr.FlushAll();
// The inputs stay in S8-S9 and the results in S10-S13 across the calls (callee-saved).
fpr.FlushBeforeCall();
for (int i = 0; i < 6; i++) {
fpr.FlushArmReg((ARM64Reg)(S8 + i));
}
for (int i = 0; i < nIn; i++) {
fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
LoadVAfterCallFlush((ARM64Reg)(S8 + i), sregs[i]);
}
for (int i = 0; i < nOut; i++) {
fp.FMOV(S0, (ARM64Reg)(S8 + i / 2));
QuickCallFunction(SCRATCH2_64, (i & 1) ? &vfpu_h2f_upper : &vfpu_h2f_lower);
fp.FMOV((ARM64Reg)(S10 + i), S0);
}
// Into S0-S3, which the cache never allocates, before mapping the destinations.
for (int i = 0; i < nOut; i++) {
fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S10 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
fp.FMOV((ARM64Reg)(S0 + i), (ARM64Reg)(S10 + i));
}
for (int i = 0; i < nOut; i++) {
fpr.MapRegV(dregs[i], MAP_DIRTY | MAP_NOINIT);
fp.FMOV(fpr.V(dregs[i]), (ARM64Reg)(S0 + i));
}
ApplyPrefixD(dregs, outSz);
@@ -2197,18 +2231,28 @@ namespace MIPSComp {
if (vd2 >= 0)
GetVectorRegs(dregs2, sz, vd2);
GetVectorRegs(&sreg, V_Single, vs);
// With the angle in a destination lane, the cosine is taken of what was written there.
// The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot.
for (int i = 0; i < n; i++) {
if (dregs[i] == sreg) {
DISABLE;
}
if (vd2 >= 0 && dregs2[i] == sreg) {
vd2 = -1;
}
}
int imm = (op >> 16) & 0x1f;
// Values in S8-S15 survive the call, so only the rest is flushed.
gpr.FlushBeforeCall();
fpr.FlushAll();
fpr.FlushBeforeCall();
// Don't need to SaveStaticRegs here as long as they are all in callee-save regs - this callee won't read them.
bool negSin1 = (imm & 0x10) ? true : false;
fpr.MapRegV(sreg);
fp.FMOV(S0, fpr.V(sreg));
LoadVAfterCallFlush(S0, sreg);
QuickCallFunction(SCRATCH2_64, negSin1 ? (void *)&SinCosNegSin : (void *)&SinCos);
// Here, sin and cos are stored together in Q0.d. On ARM32 we could use it directly
// but with ARM64's register organization, we need to split it up.
+19
View File
@@ -575,6 +575,25 @@ void Arm64JitBackend::CompIR_FSpecial(IRInst inst) {
callFuncF_F(inst.src2 ? &vfpu_h2f_upper : &vfpu_h2f_lower);
break;
case IROp::FSinCos:
// The helper returns the sine and cosine packed into D0, which the cache never allocates.
regs_.FlushBeforeCall();
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
if (regs_.IsFPRMapped(inst.src1)) {
int lane = regs_.GetFPRLane(inst.src1);
if (lane == 0)
fp_.FMOV(S0, regs_.F(inst.src1));
else
fp_.DUP(32, Q0, regs_.F(inst.src1), lane);
} else {
fp_.LDR(32, INDEX_UNSIGNED, S0, CTXREG, offsetof(MIPSState, f) + inst.src1 * 4);
}
QuickCallFunction(SCRATCH2_64, &vfpu_sincos_packed);
regs_.MapVec2(inst.dest, MIPSMap::NOINIT);
fp_.FMOV(regs_.FD(inst.dest), D0);
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
break;
default:
INVALIDOP;
break;
+1
View File
@@ -237,6 +237,7 @@ private:
void CompShiftVar(MIPSOpcode op, Arm64Gen::ShiftType shiftType);
void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin);
void CompVV2OpCall(MIPSOpcode op);
void LoadVAfterCallFlush(Arm64Gen::ARM64Reg dest, u8 vreg);
void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz);
void ApplyPrefixD(const u8 *vregs, VectorSize sz);
+9
View File
@@ -258,6 +258,15 @@ void Arm64RegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
ReleaseSpillLockV(vt);
}
void Arm64RegCacheFPU::FlushBeforeCall() {
// Only the bottom 64 bits of D8-D15 are callee-saved, which is all a single needs.
for (int i = 0; i < 32; i++) {
if (i < 8 || i > 15) {
FlushArmReg((ARM64Reg)(S0 + i));
}
}
}
void Arm64RegCacheFPU::FlushArmReg(ARM64Reg r) {
if (r >= S0 && r <= S31) {
int reg = r - S0;
+2
View File
@@ -127,6 +127,8 @@ public:
Arm64Gen::ARM64Reg V(int vreg) { return R(vreg + 32); }
void FlushAll();
// Flushes only the registers a call doesn't preserve. S8-S15 stay mapped.
void FlushBeforeCall();
// This one is allowed at any point.
void FlushV(MIPSReg r);
+34 -5
View File
@@ -2311,6 +2311,27 @@ namespace MIPSComp {
u8 sreg[1];
GetVectorRegs(sreg, V_Single, vs);
// With both a sine and a cosine lane and no overlap, one call computes both.
bool hasSine = false, hasCosine = false;
for (int i = 0; i < n; i++) {
hasSine = hasSine || d[i] == 's';
hasCosine = hasCosine || d[i] == 'c';
}
if (hasSine && hasCosine && IsOverlapSafe(n, dregs, 1, sreg)) {
ir.Write(IROp::FSinCos, IRVTEMP_0, sreg[0]);
if (negSin)
ir.Write(IROp::FNeg, IRVTEMP_0, IRVTEMP_0);
for (int i = 0; i < n; i++) {
if (d[i] == 's')
ir.Write(IROp::FMov, dregs[i], IRVTEMP_0);
else if (d[i] == 'c')
ir.Write(IROp::FMov, dregs[i], IRVTEMP_0 + 1);
else
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f);
}
return;
}
// If there's overlap, sin is calculated without it, but cosine uses the result.
// This corresponds with prefix handling, where cosine doesn't get in prefixes.
if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) {
@@ -2335,12 +2356,20 @@ namespace MIPSComp {
}
break;
case 'c':
if (IsOverlapSafe(n, dregs, 1, sreg))
if (IsOverlapSafe(n, dregs, 1, sreg)) {
ir.Write(IROp::FCos, dregs[i], sreg[0]);
else if (dregs[sineLane] == sreg[0])
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
else
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
} else {
// The cosine is taken of what the source lane got: the sine, or zero.
int srcLane = 0;
while (dregs[srcLane] != sreg[0]) {
srcLane++;
}
if (broadcastSine || srcLane == sineLane) {
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
} else {
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
}
}
break;
}
}
+1
View File
@@ -113,6 +113,7 @@ static const IRMeta irMeta[] = {
{ IROp::FExp2, "FExp2", "FF" },
{ IROp::FLog2, "FLog2", "FF" },
{ IROp::FHalfToFloat, "FHalfToFloat", "FFI" },
{ IROp::FSinCos, "FSinCos", "2F" },
{ IROp::FNeg, "FNeg", "FF" },
{ IROp::FSign, "FSign", "FF" },
{ IROp::FAbs, "FAbs", "FF" },
+1
View File
@@ -203,6 +203,7 @@ enum class IROp : uint8_t {
FExp2, // vexp2, bit-exact with the PSP
FLog2, // vlog2, bit-exact with the PSP
FHalfToFloat, // vh2f of the lower (src2 = 0) or upper (src2 = 1) half of src1
FSinCos, // dest = sin(src1), dest + 1 = cos(src1), from one argument reduction
// Fake/System instructions
Interpret,
+8
View File
@@ -613,6 +613,14 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
case IROp::FLog2:
mips->f[inst->dest] = vfpu_log2(mips->f[inst->src1]);
break;
case IROp::FSinCos:
{
float s, c;
vfpu_sincos(mips->f[inst->src1], s, c);
mips->f[inst->dest] = s;
mips->f[inst->dest + 1] = c;
break;
}
case IROp::FHalfToFloat:
mips->fi[inst->dest] = vfpu_h2f((u16)(inst->src2 ? mips->fi[inst->src1] >> 16 : mips->fi[inst->src1] & 0xFFFF));
break;
+1
View File
@@ -431,6 +431,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) {
case IROp::FExp2:
case IROp::FLog2:
case IROp::FHalfToFloat:
case IROp::FSinCos:
CompIR_FSpecial(inst);
break;
+1
View File
@@ -769,6 +769,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::FExp2:
case IROp::FLog2:
case IROp::FHalfToFloat:
case IROp::FSinCos:
out.Write(inst);
break;
+19 -1
View File
@@ -562,7 +562,7 @@ void LoongArch64JitBackend::CompIR_RoundingMode(IRInst inst) {
void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
CONDITIONAL_DISABLE;
auto callFuncF_F = [&](float (*func)(float)) {
auto callWithF0 = [&](const u8 *func) {
regs_.FlushBeforeCall();
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
@@ -579,6 +579,10 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
FLD_S(F0, CTXREG, offset);
}
QuickCallFunction(func, SCRATCH1);
};
auto callFuncF_F = [&](float (*func)(float)) {
callWithF0((const u8 *)func);
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
// If it's already F0, we're done - MapReg doesn't actually overwrite the reg in that case.
@@ -626,6 +630,20 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
callFuncF_F(inst.src2 ? &vfpu_h2f_upper : &vfpu_h2f_lower);
break;
case IROp::FSinCos:
// The sine comes back in the low 32 bits of F0, the cosine in the high.
callWithF0((const u8 *)&vfpu_sincos_packed);
MOVFR2GR_S(SCRATCH1, F0);
MOVFRH2GR_S(SCRATCH2, F0);
regs_.SpillLockFPR(inst.dest, inst.dest + 1);
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
regs_.MapFPR(inst.dest + 1, MIPSMap::NOINIT);
regs_.ReleaseSpillLockFPR(inst.dest, inst.dest + 1);
MOVGR2FR_W(regs_.F(inst.dest), SCRATCH1);
MOVGR2FR_W(regs_.F(inst.dest + 1), SCRATCH2);
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
break;
default:
INVALIDOP;
break;
@@ -144,6 +144,12 @@ void LoongArch64RegCache::FlushBeforeCall() {
for (int i = 0; i <= 23; ++i) {
FlushNativeReg(F0 + i);
}
// F24-F31 are preserved, but only their low 64 bits, so an LSX vector there has to go.
for (int i = 24; i <= 31; ++i) {
IRNativeReg nreg = F0 + i;
if (nr[nreg].mipsReg != IRREG_INVALID && GetFPRLaneCount(nr[nreg].mipsReg - 32) > 2)
FlushNativeReg(nreg);
}
}
bool LoongArch64RegCache::IsNormalized32(IRReg mipsReg) {
+9
View File
@@ -1661,6 +1661,15 @@ void vfpu_sincos(float a, float &s, float &c) {
c = vfpu_float_from_bits(cosBits);
}
double vfpu_sincos_packed(float a) {
float s, c;
vfpu_sincos(a, s, c);
const uint64_t bits = ((uint64_t)vfpu_bits_from_float(c) << 32) | vfpu_bits_from_float(s);
double packed;
memcpy(&packed, &bits, sizeof(packed));
return packed;
}
// The tables for the fast paths (see VFPUFastSegment). bias moves the segment's binade to the
// exponent the result's bits start from.
static constexpr std::array<VFPUFastSegment, 128> vfpu_make_fast_table(const VFPUSegment (&segments)[128], int bias) {
+2
View File
@@ -52,6 +52,8 @@ inline int Xpose(int v) {
extern float vfpu_sin(float);
extern float vfpu_cos(float);
extern void vfpu_sincos(float, float&, float&);
// vfpu_sincos with the sine in the low 32 bits of the result and the cosine in the high, for the JITs.
extern double vfpu_sincos_packed(float);
extern float vfpu_asin(float);
+19 -1
View File
@@ -589,7 +589,7 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
#error Currently hard float is required.
#endif
auto callFuncF_F = [&](float (*func)(float)) {
auto callWithF10 = [&](const u8 *func) {
regs_.FlushBeforeCall();
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
@@ -602,6 +602,10 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
FL(32, F10, CTXREG, offset);
}
QuickCallFunction(func, SCRATCH1);
};
auto callFuncF_F = [&](float (*func)(float)) {
callWithF10((const u8 *)func);
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
// If it's already F10, we're done - MapReg doesn't actually overwrite the reg in that case.
@@ -649,6 +653,20 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
callFuncF_F(inst.src2 ? &vfpu_h2f_upper : &vfpu_h2f_lower);
break;
case IROp::FSinCos:
// The sine comes back in the low 32 bits of F10, the cosine in the high.
callWithF10((const u8 *)&vfpu_sincos_packed);
FMV(FMv::X, FMv::D, SCRATCH1, F10);
regs_.SpillLockFPR(inst.dest, inst.dest + 1);
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
regs_.MapFPR(inst.dest + 1, MIPSMap::NOINIT);
regs_.ReleaseSpillLockFPR(inst.dest, inst.dest + 1);
FMV(FMv::W, FMv::X, regs_.F(inst.dest), SCRATCH1);
SRLI(SCRATCH1, SCRATCH1, 32);
FMV(FMv::W, FMv::X, regs_.F(inst.dest + 1), SCRATCH1);
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
break;
default:
INVALIDOP;
break;
+7 -5
View File
@@ -60,12 +60,12 @@ const int *RiscVRegCache::GetAllocationOrder(MIPSLoc type, MIPSMap flags, int &c
if (type == MIPSLoc::REG) {
// X8 and X9 are the most ideal for static alloc because they can be used with compression.
// Otherwise we stick to saved regs - might not be necessary.
// After the compressible ones come the saved regs, which survive calls.
static const int allocationOrder[] = {
X8, X9, X12, X13, X14, X15, X5, X6, X7, X16, X17, X18, X19, X20, X21, X22, X23, X28, X29, X30, X31,
X8, X9, X12, X13, X14, X15, X18, X19, X20, X21, X22, X23, X5, X6, X7, X16, X17, X28, X29, X30, X31,
};
static const int allocationOrderStaticAlloc[] = {
X12, X13, X14, X15, X5, X6, X7, X16, X17, X21, X22, X23, X28, X29, X30, X31,
X12, X13, X14, X15, X21, X22, X23, X5, X6, X7, X16, X17, X28, X29, X30, X31,
};
if (jo_->useStaticAlloc) {
@@ -76,11 +76,13 @@ const int *RiscVRegCache::GetAllocationOrder(MIPSLoc type, MIPSMap flags, int &c
return allocationOrder;
}
} else if (type == MIPSLoc::FREG) {
// F8 through F15 are used for compression, so they are great.
// F8 through F15 are used for compression, so they are great. Then the saved regs (F18-F27),
// which survive calls.
static const int allocationOrder[] = {
F8, F9, F10, F11, F12, F13, F14, F15,
F18, F19, F20, F21, F22, F23, F24, F25, F26, F27,
F0, F1, F2, F3, F4, F5, F6, F7,
F16, F17, F18, F19, F20, F21, F22, F23, F24, F25, F26, F27, F28, F29, F30, F31,
F16, F17, F28, F29, F30, F31,
};
count = ARRAY_SIZE(allocationOrder);
+54 -4
View File
@@ -2317,6 +2317,34 @@ void RExp2(SinCosArg arg, float *output) {
output[0] = vfpu_rexp2(arg);
}
#if PPSSPP_ARCH(AMD64)
static float NegRcp(float x) {
return -vfpu_rcp(x);
}
static float NegSin(float x) {
return -vfpu_sin(x);
}
// The VV2Op math functions, called with CallProtectedLeaf. They take and return their float in XMM0.
static float (*VV2OpMathFunc(int optype))(float) {
switch (optype) {
case 16: return &vfpu_rcp;
case 17: return &vfpu_rsqrt;
case 18: return &vfpu_sin;
case 19: return &vfpu_cos;
case 20: return &vfpu_exp2;
case 21: return &vfpu_log2;
case 22: return &vfpu_sqrt;
case 23: return &vfpu_asin;
case 24: return &NegRcp;
case 26: return &NegSin;
case 28: return &vfpu_rexp2;
default: return nullptr;
}
}
#endif
void Jit::Comp_VV2Op(MIPSOpcode op) {
CONDITIONAL_DISABLE(VFPU_VEC);
@@ -2445,10 +2473,22 @@ void Jit::Comp_VV2Op(MIPSOpcode op) {
}
}
#if PPSSPP_ARCH(AMD64)
float (*mathFunc)(float) = VV2OpMathFunc((op >> 16) & 0x1f);
#endif
// Warning: sregs[i] and tempxregs[i] may be the same reg.
// Helps for vmov, hurts for vrcp, etc.
for (int i = 0; i < n; ++i)
{
#if PPSSPP_ARCH(AMD64)
if (mathFunc) {
MOVSS(XMM0, fpr.V(sregs[i]));
CallProtectedLeaf((const void *)mathFunc);
MOVSS(tempxregs[i], R(XMM0));
continue;
}
#endif
switch ((op >> 16) & 0x1f)
{
case 0: // d[i] = s[i]; break; //vmov
@@ -3719,15 +3759,22 @@ void Jit::Comp_VRot(MIPSOpcode op) {
if (vd2 >= 0)
GetVectorRegs(dregs2, sz, vd2);
GetVectorRegs(&sreg, V_Single, vs);
// With the angle in a destination lane, the cosine is taken of what was written there.
// The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot.
for (int i = 0; i < n; i++) {
if (dregs[i] == sreg) {
DISABLE;
}
if (vd2 >= 0 && dregs2[i] == sreg) {
vd2 = -1;
}
}
// Flush SIMD.
fpr.SimpleRegsV(&sreg, V_Single, 0);
int imm = (op >> 16) & 0x1f;
gpr.FlushBeforeCall();
fpr.Flush();
bool negSin1 = (imm & 0x10) ? true : false;
#if PPSSPP_ARCH(AMD64)
@@ -3737,8 +3784,11 @@ void Jit::Comp_VRot(MIPSOpcode op) {
LEA(64, RDI, MIPSSTATE_VAR(sincostemp));
#endif
MOVSS(XMM0, fpr.V(sreg));
ABI_CallFunction(negSin1 ? (const void *)&SinCosNegSin : (const void *)&SinCos);
CallProtectedLeaf(negSin1 ? (const void *)&SinCosNegSin : (const void *)&SinCos);
#else
gpr.FlushBeforeCall();
fpr.Flush();
// Sigh, passing floats with cdecl isn't pretty, ends up on the stack.
ABI_CallFunctionAC(negSin1 ? (const void *)&SinCosNegSin : (const void *)&SinCos, fpr.V(sreg), (uintptr_t)mips_->sincostemp);
#endif
+43
View File
@@ -907,6 +907,49 @@ void Jit::CallProtectedFunction(const void *func, const OpArg &arg1, const u32 a
ABI_CallFunctionAC(thunks.ProtectFunction(func, 2), arg1, arg2);
}
#if PPSSPP_ARCH(AMD64)
void Jit::CallProtectedLeaf(const void *func) {
// The registers a call may clobber, apart from RAX and XMM0-1, which are scratch.
#ifdef _WIN32
static const X64Reg callerSavedGPRs[] = { RCX, RDX, R8, R9, R10, R11 };
const int lastXMM = 5;
const int shadowSpace = 32;
#else
static const X64Reg callerSavedGPRs[] = { RCX, RDX, R8, R9, R10, R11, RSI, RDI };
const int lastXMM = 15;
const int shadowSpace = 0;
#endif
X64Reg xmms[16];
X64Reg gprs[ARRAY_SIZE(callerSavedGPRs)];
int numXMMs = 0, numGPRs = 0;
for (int i = 2; i <= lastXMM; i++) {
if (fpr.IsXRegInUse((X64Reg)(XMM0 + i)))
xmms[numXMMs++] = (X64Reg)(XMM0 + i);
}
for (X64Reg reg : callerSavedGPRs) {
if (gpr.IsXRegInUse(reg))
gprs[numGPRs++] = reg;
}
// The JIT runs with RSP 16-byte aligned, so this keeps it aligned for the call.
const int gprBase = shadowSpace + numXMMs * 16;
const int frameSize = (gprBase + numGPRs * 8 + 15) & ~15;
if (frameSize)
SUB(64, R(RSP), Imm32(frameSize));
for (int i = 0; i < numXMMs; i++)
MOVAPS(MDisp(RSP, shadowSpace + i * 16), xmms[i]);
for (int i = 0; i < numGPRs; i++)
MOV(64, MDisp(RSP, gprBase + i * 8), R(gprs[i]));
ABI_CallFunction(func);
for (int i = 0; i < numXMMs; i++)
MOVAPS(xmms[i], MDisp(RSP, shadowSpace + i * 16));
for (int i = 0; i < numGPRs; i++)
MOV(64, R(gprs[i]), MDisp(RSP, gprBase + i * 8));
if (frameSize)
ADD(64, R(RSP), Imm32(frameSize));
}
#endif
void Jit::Comp_DoNothing(MIPSOpcode op) { }
MIPSOpcode Jit::GetOriginalOp(MIPSOpcode op) {
+6
View File
@@ -246,6 +246,12 @@ private:
void CallProtectedFunction(const void *func, const Gen::OpArg &arg1, const Gen::OpArg &arg2);
void CallProtectedFunction(const void *func, const Gen::OpArg &arg1, const u32 arg2);
void CallProtectedFunction(const void *func, const u32 arg1, const u32 arg2);
#if PPSSPP_ARCH(AMD64)
// A lighter CallProtectedFunction for a function that leaves MXCSR alone: saves only the
// caller-saved registers the caches are using. Set up the arguments first - the argument
// registers are never cached.
void CallProtectedLeaf(const void *func);
#endif
template <typename Tr, typename T1>
void CallProtectedFunction(Tr (*func)(T1), const Gen::OpArg &arg1) {
+5
View File
@@ -119,6 +119,11 @@ public:
void GetState(GPRRegCacheState &state) const;
void RestoreState(const GPRRegCacheState& state);
// Whether the register holds something the cache cares about: a MIPS register or a locked temp.
bool IsXRegInUse(Gen::X64Reg reg) const {
return !xregs[reg].free || xregs[reg].allocLocked;
}
MIPSState *mips_ = nullptr;
private:
+4
View File
@@ -172,6 +172,10 @@ public:
bool IsMappedV(int v) {
return vregs[v].lane == 0 && V(v).IsSimpleReg();
}
// Whether the register holds a MIPS register or a temp.
bool IsXRegInUse(Gen::X64Reg reg) const {
return xregs[reg].mipsReg != -1;
}
bool IsMappedVS(u8 v) {
return vregs[v].lane != 0 && VS(&v).IsSimpleReg();
}
+36
View File
@@ -978,6 +978,10 @@ static float X64JIT_XMM_CALL x64_log2(float f) {
return vfpu_log2(f);
}
static double X64JIT_XMM_CALL x64_sincos(float f) {
return vfpu_sincos_packed(f);
}
static float X64JIT_XMM_CALL x64_h2f_lower(float f) {
return vfpu_h2f_lower(f);
}
@@ -1150,6 +1154,38 @@ void X64JitBackend::CompIR_FSpecial(IRInst inst) {
callFuncF_F(inst.src2 ? (const void *)&x64_h2f_upper : (const void *)&x64_h2f_lower);
break;
case IROp::FSinCos:
{
#if X64JIT_USE_XMM_CALL
// The helper returns the sine and cosine packed into the low 64 bits of XMM0. The cache
// can hand out XMM0 when mapping below, so park them in sincostemp meanwhile.
regs_.FlushBeforeCall();
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
if (regs_.IsFPRMapped(inst.src1)) {
int lane = regs_.GetFPRLane(inst.src1);
CopyVec4ToFPRLane0(XMM0, regs_.FX(inst.src1), lane);
} else {
// Account for CTXREG being increased by 128 to reduce imm sizes.
MOVSS(XMM0, MDisp(CTXREG, offsetof(MIPSState, f) + inst.src1 * 4 - 128));
}
ABI_CallFunction((const void *)&x64_sincos);
MOVSD(MDisp(CTXREG, offsetof(MIPSState, sincostemp) - 128), XMM0);
regs_.Map(inst);
MOVSD(regs_.FX(inst.dest), MDisp(CTXREG, offsetof(MIPSState, sincostemp) - 128));
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
#else
// Two calls here. The frontend makes sure dest doesn't overlap src1.
IRInst sinInst = inst;
sinInst.op = IROp::FSin;
CompIR_FSpecial(sinInst);
IRInst cosInst = inst;
cosInst.op = IROp::FCos;
cosInst.dest = inst.dest + 1;
CompIR_FSpecial(cosInst);
#endif
break;
}
default:
INVALIDOP;
break;
+1
View File
@@ -102,6 +102,7 @@ tests_good = [
"cpu/vfpu/matrix",
"cpu/vfpu/vavg",
"cpu/vfpu/exact",
"cpu/vfpu/vrot",
"cpu/icache/icache",
"cpu/lsu/lsu",
"cpu/lsu/llsc",