mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #22351 from hrydgard/jit-call-optimizations
JIT function-call optimizations
This commit is contained in:
27 files changed
+384
-66
No files matched your search
+23
-31
@@ -30,6 +30,18 @@ alignas(32) static u8 saved_gpr_state[16 * 8];
|
||||
static u16 saved_mxcsr;
|
||||
#endif
|
||||
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
// The registers a call may clobber, apart from RAX and XMM0-1, which the JIT uses as scratch.
|
||||
// The callee preserves the rest (RBX, RBP, R12-R15, and XMM6-15 on Windows), so there's no need to.
|
||||
#ifdef _WIN32
|
||||
const int THUNK_LAST_XMM = 5;
|
||||
const Gen::X64Reg thunkGPRs[] = { Gen::RCX, Gen::RDX, Gen::R8, Gen::R9, Gen::R10, Gen::R11 };
|
||||
#else
|
||||
const int THUNK_LAST_XMM = 15;
|
||||
const Gen::X64Reg thunkGPRs[] = { Gen::RCX, Gen::RDX, Gen::R8, Gen::R9, Gen::R10, Gen::R11, Gen::RSI, Gen::RDI };
|
||||
#endif
|
||||
#endif
|
||||
|
||||
} // namespace
|
||||
|
||||
using namespace Gen;
|
||||
@@ -46,21 +58,12 @@ void ThunkManager::Init()
|
||||
BeginWrite(512);
|
||||
save_regs = GetCodePtr();
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
|
||||
for (int i = 2; i <= THUNK_LAST_XMM; i++)
|
||||
MOVAPS(MDisp(RSP, stackOffset + (i - 2) * 16), (X64Reg)(XMM0 + i));
|
||||
stackPosition = (ABI_GetNumXMMRegs() - 2) * 2;
|
||||
stackPosition = (THUNK_LAST_XMM - 1) * 2;
|
||||
STMXCSR(MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RCX));
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RDX));
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R8) );
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R9) );
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R10));
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(R11));
|
||||
#ifndef _WIN32
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RSI));
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RDI));
|
||||
#endif
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(RBX));
|
||||
for (X64Reg reg : thunkGPRs)
|
||||
MOV(64, MDisp(RSP, stackOffset + (stackPosition++ * 8)), R(reg));
|
||||
#else
|
||||
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
|
||||
MOVAPS(M(saved_fp_state + i * 16), (X64Reg)(XMM0 + i));
|
||||
@@ -72,21 +75,12 @@ void ThunkManager::Init()
|
||||
|
||||
load_regs = GetCodePtr();
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
|
||||
for (int i = 2; i <= THUNK_LAST_XMM; i++)
|
||||
MOVAPS((X64Reg)(XMM0 + i), MDisp(RSP, stackOffset + (i - 2) * 16));
|
||||
stackPosition = (ABI_GetNumXMMRegs() - 2) * 2;
|
||||
stackPosition = (THUNK_LAST_XMM - 1) * 2;
|
||||
LDMXCSR(MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(RCX), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(RDX), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(R8) , MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(R9) , MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(R10), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(R11), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
#ifndef _WIN32
|
||||
MOV(64, R(RSI), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
MOV(64, R(RDI), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
#endif
|
||||
MOV(64, R(RBX), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
for (X64Reg reg : thunkGPRs)
|
||||
MOV(64, R(reg), MDisp(RSP, stackOffset + (stackPosition++ * 8)));
|
||||
#else
|
||||
LDMXCSR(M(&saved_mxcsr));
|
||||
for (int i = 2; i < ABI_GetNumXMMRegs(); i++)
|
||||
@@ -112,15 +106,13 @@ void ThunkManager::Shutdown()
|
||||
|
||||
int ThunkManager::ThunkBytesNeeded()
|
||||
{
|
||||
int space = (ABI_GetNumXMMRegs() - 2) * 16;
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
int space = (THUNK_LAST_XMM - 1) * 16;
|
||||
// MXCSR
|
||||
space += 8;
|
||||
space += 7 * 8;
|
||||
#ifndef _WIN32
|
||||
space += 2 * 8;
|
||||
#endif
|
||||
space += (int)ARRAY_SIZE(thunkGPRs) * 8;
|
||||
#else
|
||||
int space = (ABI_GetNumXMMRegs() - 2) * 16;
|
||||
// MXCSR
|
||||
space += 4;
|
||||
space += 2 * 4;
|
||||
|
||||
@@ -2250,6 +2250,16 @@ namespace MIPSComp
|
||||
if (vd2 >= 0)
|
||||
GetVectorRegs(dregs2, sz, vd2);
|
||||
GetVectorRegs(&sreg, V_Single, vs);
|
||||
// With the angle in a destination lane, the cosine is taken of what was written there.
|
||||
// The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot.
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (dregs[i] == sreg) {
|
||||
DISABLE;
|
||||
}
|
||||
if (vd2 >= 0 && dregs2[i] == sreg) {
|
||||
vd2 = -1;
|
||||
}
|
||||
}
|
||||
|
||||
int imm = (op >> 16) & 0x1f;
|
||||
|
||||
|
||||
@@ -950,6 +950,16 @@ namespace MIPSComp {
|
||||
|
||||
// The VFPU special functions call the exact C versions. The lanes stay in S8-S11 across the
|
||||
// calls (callee-saved), and the results are stored to the destinations' homes.
|
||||
// After fpr.FlushBeforeCall(), a VFPU register is either still mapped (in S8-S15) or its value is
|
||||
// in memory. Loads it into dest either way.
|
||||
void Arm64Jit::LoadVAfterCallFlush(ARM64Reg dest, u8 vreg) {
|
||||
if (fpr.IsMappedV(vreg)) {
|
||||
fp.FMOV(dest, fpr.V(vreg));
|
||||
} else {
|
||||
fp.LDR(32, INDEX_UNSIGNED, dest, CTXREG, fpr.GetMipsRegOffsetV(vreg));
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64Jit::CompVV2OpCall(MIPSOpcode op) {
|
||||
if (js.HasSPrefix()) {
|
||||
DISABLE;
|
||||
@@ -978,23 +988,39 @@ namespace MIPSComp {
|
||||
GetVectorRegs(sregs, sz, _VS);
|
||||
GetVectorRegs(dregs, sz, _VD);
|
||||
|
||||
// Values in S8-S15 survive the calls, so only the rest is flushed.
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.FlushAll();
|
||||
fpr.FlushBeforeCall();
|
||||
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.FMOV(S0, (ARM64Reg)(S8 + i));
|
||||
if (n == 1) {
|
||||
LoadVAfterCallFlush(S0, sregs[0]);
|
||||
QuickCallFunction(SCRATCH2_64, func);
|
||||
if (negate) {
|
||||
fp.FNEG((ARM64Reg)(S8 + i), S0);
|
||||
} else {
|
||||
} else {
|
||||
// The lanes wait in S8 and up between the calls, so free those too.
|
||||
for (int i = 0; i < n; i++) {
|
||||
fpr.FlushArmReg((ARM64Reg)(S8 + i));
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
LoadVAfterCallFlush((ARM64Reg)(S8 + i), sregs[i]);
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.FMOV(S0, (ARM64Reg)(S8 + i));
|
||||
QuickCallFunction(SCRATCH2_64, func);
|
||||
fp.FMOV((ARM64Reg)(S8 + i), S0);
|
||||
}
|
||||
// Into S0-S3, which the cache never allocates, before mapping the destinations.
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.FMOV((ARM64Reg)(S0 + i), (ARM64Reg)(S8 + i));
|
||||
}
|
||||
}
|
||||
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
|
||||
fpr.MapRegV(dregs[i], MAP_DIRTY | MAP_NOINIT);
|
||||
if (negate) {
|
||||
fp.FNEG(fpr.V(dregs[i]), (ARM64Reg)(S0 + i));
|
||||
} else {
|
||||
fp.FMOV(fpr.V(dregs[i]), (ARM64Reg)(S0 + i));
|
||||
}
|
||||
}
|
||||
|
||||
ApplyPrefixD(dregs, sz);
|
||||
@@ -1065,20 +1091,28 @@ namespace MIPSComp {
|
||||
GetVectorRegs(sregs, sz, _VS);
|
||||
GetVectorRegs(dregs, outSz, _VD);
|
||||
|
||||
// The inputs wait in S8-S9 and the results in S10-S13 between the calls (callee-saved), so
|
||||
// those are flushed along with the registers a call clobbers.
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.FlushAll();
|
||||
|
||||
// The inputs stay in S8-S9 and the results in S10-S13 across the calls (callee-saved).
|
||||
fpr.FlushBeforeCall();
|
||||
for (int i = 0; i < 6; i++) {
|
||||
fpr.FlushArmReg((ARM64Reg)(S8 + i));
|
||||
}
|
||||
for (int i = 0; i < nIn; i++) {
|
||||
fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
|
||||
LoadVAfterCallFlush((ARM64Reg)(S8 + i), sregs[i]);
|
||||
}
|
||||
for (int i = 0; i < nOut; i++) {
|
||||
fp.FMOV(S0, (ARM64Reg)(S8 + i / 2));
|
||||
QuickCallFunction(SCRATCH2_64, (i & 1) ? &vfpu_h2f_upper : &vfpu_h2f_lower);
|
||||
fp.FMOV((ARM64Reg)(S10 + i), S0);
|
||||
}
|
||||
// Into S0-S3, which the cache never allocates, before mapping the destinations.
|
||||
for (int i = 0; i < nOut; i++) {
|
||||
fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S10 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
|
||||
fp.FMOV((ARM64Reg)(S0 + i), (ARM64Reg)(S10 + i));
|
||||
}
|
||||
for (int i = 0; i < nOut; i++) {
|
||||
fpr.MapRegV(dregs[i], MAP_DIRTY | MAP_NOINIT);
|
||||
fp.FMOV(fpr.V(dregs[i]), (ARM64Reg)(S0 + i));
|
||||
}
|
||||
|
||||
ApplyPrefixD(dregs, outSz);
|
||||
@@ -2197,18 +2231,28 @@ namespace MIPSComp {
|
||||
if (vd2 >= 0)
|
||||
GetVectorRegs(dregs2, sz, vd2);
|
||||
GetVectorRegs(&sreg, V_Single, vs);
|
||||
// With the angle in a destination lane, the cosine is taken of what was written there.
|
||||
// The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot.
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (dregs[i] == sreg) {
|
||||
DISABLE;
|
||||
}
|
||||
if (vd2 >= 0 && dregs2[i] == sreg) {
|
||||
vd2 = -1;
|
||||
}
|
||||
}
|
||||
|
||||
int imm = (op >> 16) & 0x1f;
|
||||
|
||||
// Values in S8-S15 survive the call, so only the rest is flushed.
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.FlushAll();
|
||||
fpr.FlushBeforeCall();
|
||||
|
||||
// Don't need to SaveStaticRegs here as long as they are all in callee-save regs - this callee won't read them.
|
||||
|
||||
bool negSin1 = (imm & 0x10) ? true : false;
|
||||
|
||||
fpr.MapRegV(sreg);
|
||||
fp.FMOV(S0, fpr.V(sreg));
|
||||
LoadVAfterCallFlush(S0, sreg);
|
||||
QuickCallFunction(SCRATCH2_64, negSin1 ? (void *)&SinCosNegSin : (void *)&SinCos);
|
||||
// Here, sin and cos are stored together in Q0.d. On ARM32 we could use it directly
|
||||
// but with ARM64's register organization, we need to split it up.
|
||||
|
||||
@@ -575,6 +575,25 @@ void Arm64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
callFuncF_F(inst.src2 ? &vfpu_h2f_upper : &vfpu_h2f_lower);
|
||||
break;
|
||||
|
||||
case IROp::FSinCos:
|
||||
// The helper returns the sine and cosine packed into D0, which the cache never allocates.
|
||||
regs_.FlushBeforeCall();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
|
||||
if (regs_.IsFPRMapped(inst.src1)) {
|
||||
int lane = regs_.GetFPRLane(inst.src1);
|
||||
if (lane == 0)
|
||||
fp_.FMOV(S0, regs_.F(inst.src1));
|
||||
else
|
||||
fp_.DUP(32, Q0, regs_.F(inst.src1), lane);
|
||||
} else {
|
||||
fp_.LDR(32, INDEX_UNSIGNED, S0, CTXREG, offsetof(MIPSState, f) + inst.src1 * 4);
|
||||
}
|
||||
QuickCallFunction(SCRATCH2_64, &vfpu_sincos_packed);
|
||||
regs_.MapVec2(inst.dest, MIPSMap::NOINIT);
|
||||
fp_.FMOV(regs_.FD(inst.dest), D0);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
@@ -237,6 +237,7 @@ private:
|
||||
void CompShiftVar(MIPSOpcode op, Arm64Gen::ShiftType shiftType);
|
||||
void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin);
|
||||
void CompVV2OpCall(MIPSOpcode op);
|
||||
void LoadVAfterCallFlush(Arm64Gen::ARM64Reg dest, u8 vreg);
|
||||
|
||||
void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz);
|
||||
void ApplyPrefixD(const u8 *vregs, VectorSize sz);
|
||||
|
||||
@@ -258,6 +258,15 @@ void Arm64RegCacheFPU::MapDirtyInInV(int vd, int vs, int vt, bool avoidLoad) {
|
||||
ReleaseSpillLockV(vt);
|
||||
}
|
||||
|
||||
void Arm64RegCacheFPU::FlushBeforeCall() {
|
||||
// Only the bottom 64 bits of D8-D15 are callee-saved, which is all a single needs.
|
||||
for (int i = 0; i < 32; i++) {
|
||||
if (i < 8 || i > 15) {
|
||||
FlushArmReg((ARM64Reg)(S0 + i));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64RegCacheFPU::FlushArmReg(ARM64Reg r) {
|
||||
if (r >= S0 && r <= S31) {
|
||||
int reg = r - S0;
|
||||
|
||||
@@ -127,6 +127,8 @@ public:
|
||||
Arm64Gen::ARM64Reg V(int vreg) { return R(vreg + 32); }
|
||||
|
||||
void FlushAll();
|
||||
// Flushes only the registers a call doesn't preserve. S8-S15 stay mapped.
|
||||
void FlushBeforeCall();
|
||||
|
||||
// This one is allowed at any point.
|
||||
void FlushV(MIPSReg r);
|
||||
|
||||
@@ -2311,6 +2311,27 @@ namespace MIPSComp {
|
||||
u8 sreg[1];
|
||||
GetVectorRegs(sreg, V_Single, vs);
|
||||
|
||||
// With both a sine and a cosine lane and no overlap, one call computes both.
|
||||
bool hasSine = false, hasCosine = false;
|
||||
for (int i = 0; i < n; i++) {
|
||||
hasSine = hasSine || d[i] == 's';
|
||||
hasCosine = hasCosine || d[i] == 'c';
|
||||
}
|
||||
if (hasSine && hasCosine && IsOverlapSafe(n, dregs, 1, sreg)) {
|
||||
ir.Write(IROp::FSinCos, IRVTEMP_0, sreg[0]);
|
||||
if (negSin)
|
||||
ir.Write(IROp::FNeg, IRVTEMP_0, IRVTEMP_0);
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (d[i] == 's')
|
||||
ir.Write(IROp::FMov, dregs[i], IRVTEMP_0);
|
||||
else if (d[i] == 'c')
|
||||
ir.Write(IROp::FMov, dregs[i], IRVTEMP_0 + 1);
|
||||
else
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// If there's overlap, sin is calculated without it, but cosine uses the result.
|
||||
// This corresponds with prefix handling, where cosine doesn't get in prefixes.
|
||||
if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) {
|
||||
@@ -2335,12 +2356,20 @@ namespace MIPSComp {
|
||||
}
|
||||
break;
|
||||
case 'c':
|
||||
if (IsOverlapSafe(n, dregs, 1, sreg))
|
||||
if (IsOverlapSafe(n, dregs, 1, sreg)) {
|
||||
ir.Write(IROp::FCos, dregs[i], sreg[0]);
|
||||
else if (dregs[sineLane] == sreg[0])
|
||||
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
|
||||
else
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
|
||||
} else {
|
||||
// The cosine is taken of what the source lane got: the sine, or zero.
|
||||
int srcLane = 0;
|
||||
while (dregs[srcLane] != sreg[0]) {
|
||||
srcLane++;
|
||||
}
|
||||
if (broadcastSine || srcLane == sineLane) {
|
||||
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
|
||||
} else {
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -113,6 +113,7 @@ static const IRMeta irMeta[] = {
|
||||
{ IROp::FExp2, "FExp2", "FF" },
|
||||
{ IROp::FLog2, "FLog2", "FF" },
|
||||
{ IROp::FHalfToFloat, "FHalfToFloat", "FFI" },
|
||||
{ IROp::FSinCos, "FSinCos", "2F" },
|
||||
{ IROp::FNeg, "FNeg", "FF" },
|
||||
{ IROp::FSign, "FSign", "FF" },
|
||||
{ IROp::FAbs, "FAbs", "FF" },
|
||||
|
||||
@@ -203,6 +203,7 @@ enum class IROp : uint8_t {
|
||||
FExp2, // vexp2, bit-exact with the PSP
|
||||
FLog2, // vlog2, bit-exact with the PSP
|
||||
FHalfToFloat, // vh2f of the lower (src2 = 0) or upper (src2 = 1) half of src1
|
||||
FSinCos, // dest = sin(src1), dest + 1 = cos(src1), from one argument reduction
|
||||
|
||||
// Fake/System instructions
|
||||
Interpret,
|
||||
|
||||
@@ -613,6 +613,14 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
|
||||
case IROp::FLog2:
|
||||
mips->f[inst->dest] = vfpu_log2(mips->f[inst->src1]);
|
||||
break;
|
||||
case IROp::FSinCos:
|
||||
{
|
||||
float s, c;
|
||||
vfpu_sincos(mips->f[inst->src1], s, c);
|
||||
mips->f[inst->dest] = s;
|
||||
mips->f[inst->dest + 1] = c;
|
||||
break;
|
||||
}
|
||||
case IROp::FHalfToFloat:
|
||||
mips->fi[inst->dest] = vfpu_h2f((u16)(inst->src2 ? mips->fi[inst->src1] >> 16 : mips->fi[inst->src1] & 0xFFFF));
|
||||
break;
|
||||
|
||||
@@ -431,6 +431,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) {
|
||||
case IROp::FExp2:
|
||||
case IROp::FLog2:
|
||||
case IROp::FHalfToFloat:
|
||||
case IROp::FSinCos:
|
||||
CompIR_FSpecial(inst);
|
||||
break;
|
||||
|
||||
|
||||
@@ -769,6 +769,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::FExp2:
|
||||
case IROp::FLog2:
|
||||
case IROp::FHalfToFloat:
|
||||
case IROp::FSinCos:
|
||||
out.Write(inst);
|
||||
break;
|
||||
|
||||
|
||||
@@ -562,7 +562,7 @@ void LoongArch64JitBackend::CompIR_RoundingMode(IRInst inst) {
|
||||
void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
CONDITIONAL_DISABLE;
|
||||
|
||||
auto callFuncF_F = [&](float (*func)(float)) {
|
||||
auto callWithF0 = [&](const u8 *func) {
|
||||
regs_.FlushBeforeCall();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
|
||||
|
||||
@@ -579,6 +579,10 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
FLD_S(F0, CTXREG, offset);
|
||||
}
|
||||
QuickCallFunction(func, SCRATCH1);
|
||||
};
|
||||
|
||||
auto callFuncF_F = [&](float (*func)(float)) {
|
||||
callWithF0((const u8 *)func);
|
||||
|
||||
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
|
||||
// If it's already F0, we're done - MapReg doesn't actually overwrite the reg in that case.
|
||||
@@ -626,6 +630,20 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
callFuncF_F(inst.src2 ? &vfpu_h2f_upper : &vfpu_h2f_lower);
|
||||
break;
|
||||
|
||||
case IROp::FSinCos:
|
||||
// The sine comes back in the low 32 bits of F0, the cosine in the high.
|
||||
callWithF0((const u8 *)&vfpu_sincos_packed);
|
||||
MOVFR2GR_S(SCRATCH1, F0);
|
||||
MOVFRH2GR_S(SCRATCH2, F0);
|
||||
regs_.SpillLockFPR(inst.dest, inst.dest + 1);
|
||||
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
|
||||
regs_.MapFPR(inst.dest + 1, MIPSMap::NOINIT);
|
||||
regs_.ReleaseSpillLockFPR(inst.dest, inst.dest + 1);
|
||||
MOVGR2FR_W(regs_.F(inst.dest), SCRATCH1);
|
||||
MOVGR2FR_W(regs_.F(inst.dest + 1), SCRATCH2);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
@@ -144,6 +144,12 @@ void LoongArch64RegCache::FlushBeforeCall() {
|
||||
for (int i = 0; i <= 23; ++i) {
|
||||
FlushNativeReg(F0 + i);
|
||||
}
|
||||
// F24-F31 are preserved, but only their low 64 bits, so an LSX vector there has to go.
|
||||
for (int i = 24; i <= 31; ++i) {
|
||||
IRNativeReg nreg = F0 + i;
|
||||
if (nr[nreg].mipsReg != IRREG_INVALID && GetFPRLaneCount(nr[nreg].mipsReg - 32) > 2)
|
||||
FlushNativeReg(nreg);
|
||||
}
|
||||
}
|
||||
|
||||
bool LoongArch64RegCache::IsNormalized32(IRReg mipsReg) {
|
||||
|
||||
@@ -1661,6 +1661,15 @@ void vfpu_sincos(float a, float &s, float &c) {
|
||||
c = vfpu_float_from_bits(cosBits);
|
||||
}
|
||||
|
||||
double vfpu_sincos_packed(float a) {
|
||||
float s, c;
|
||||
vfpu_sincos(a, s, c);
|
||||
const uint64_t bits = ((uint64_t)vfpu_bits_from_float(c) << 32) | vfpu_bits_from_float(s);
|
||||
double packed;
|
||||
memcpy(&packed, &bits, sizeof(packed));
|
||||
return packed;
|
||||
}
|
||||
|
||||
// The tables for the fast paths (see VFPUFastSegment). bias moves the segment's binade to the
|
||||
// exponent the result's bits start from.
|
||||
static constexpr std::array<VFPUFastSegment, 128> vfpu_make_fast_table(const VFPUSegment (&segments)[128], int bias) {
|
||||
|
||||
@@ -52,6 +52,8 @@ inline int Xpose(int v) {
|
||||
extern float vfpu_sin(float);
|
||||
extern float vfpu_cos(float);
|
||||
extern void vfpu_sincos(float, float&, float&);
|
||||
// vfpu_sincos with the sine in the low 32 bits of the result and the cosine in the high, for the JITs.
|
||||
extern double vfpu_sincos_packed(float);
|
||||
|
||||
extern float vfpu_asin(float);
|
||||
|
||||
|
||||
@@ -589,7 +589,7 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
#error Currently hard float is required.
|
||||
#endif
|
||||
|
||||
auto callFuncF_F = [&](float (*func)(float)) {
|
||||
auto callWithF10 = [&](const u8 *func) {
|
||||
regs_.FlushBeforeCall();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
|
||||
|
||||
@@ -602,6 +602,10 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
FL(32, F10, CTXREG, offset);
|
||||
}
|
||||
QuickCallFunction(func, SCRATCH1);
|
||||
};
|
||||
|
||||
auto callFuncF_F = [&](float (*func)(float)) {
|
||||
callWithF10((const u8 *)func);
|
||||
|
||||
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
|
||||
// If it's already F10, we're done - MapReg doesn't actually overwrite the reg in that case.
|
||||
@@ -649,6 +653,20 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
callFuncF_F(inst.src2 ? &vfpu_h2f_upper : &vfpu_h2f_lower);
|
||||
break;
|
||||
|
||||
case IROp::FSinCos:
|
||||
// The sine comes back in the low 32 bits of F10, the cosine in the high.
|
||||
callWithF10((const u8 *)&vfpu_sincos_packed);
|
||||
FMV(FMv::X, FMv::D, SCRATCH1, F10);
|
||||
regs_.SpillLockFPR(inst.dest, inst.dest + 1);
|
||||
regs_.MapFPR(inst.dest, MIPSMap::NOINIT);
|
||||
regs_.MapFPR(inst.dest + 1, MIPSMap::NOINIT);
|
||||
regs_.ReleaseSpillLockFPR(inst.dest, inst.dest + 1);
|
||||
FMV(FMv::W, FMv::X, regs_.F(inst.dest), SCRATCH1);
|
||||
SRLI(SCRATCH1, SCRATCH1, 32);
|
||||
FMV(FMv::W, FMv::X, regs_.F(inst.dest + 1), SCRATCH1);
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
@@ -60,12 +60,12 @@ const int *RiscVRegCache::GetAllocationOrder(MIPSLoc type, MIPSMap flags, int &c
|
||||
|
||||
if (type == MIPSLoc::REG) {
|
||||
// X8 and X9 are the most ideal for static alloc because they can be used with compression.
|
||||
// Otherwise we stick to saved regs - might not be necessary.
|
||||
// After the compressible ones come the saved regs, which survive calls.
|
||||
static const int allocationOrder[] = {
|
||||
X8, X9, X12, X13, X14, X15, X5, X6, X7, X16, X17, X18, X19, X20, X21, X22, X23, X28, X29, X30, X31,
|
||||
X8, X9, X12, X13, X14, X15, X18, X19, X20, X21, X22, X23, X5, X6, X7, X16, X17, X28, X29, X30, X31,
|
||||
};
|
||||
static const int allocationOrderStaticAlloc[] = {
|
||||
X12, X13, X14, X15, X5, X6, X7, X16, X17, X21, X22, X23, X28, X29, X30, X31,
|
||||
X12, X13, X14, X15, X21, X22, X23, X5, X6, X7, X16, X17, X28, X29, X30, X31,
|
||||
};
|
||||
|
||||
if (jo_->useStaticAlloc) {
|
||||
@@ -76,11 +76,13 @@ const int *RiscVRegCache::GetAllocationOrder(MIPSLoc type, MIPSMap flags, int &c
|
||||
return allocationOrder;
|
||||
}
|
||||
} else if (type == MIPSLoc::FREG) {
|
||||
// F8 through F15 are used for compression, so they are great.
|
||||
// F8 through F15 are used for compression, so they are great. Then the saved regs (F18-F27),
|
||||
// which survive calls.
|
||||
static const int allocationOrder[] = {
|
||||
F8, F9, F10, F11, F12, F13, F14, F15,
|
||||
F18, F19, F20, F21, F22, F23, F24, F25, F26, F27,
|
||||
F0, F1, F2, F3, F4, F5, F6, F7,
|
||||
F16, F17, F18, F19, F20, F21, F22, F23, F24, F25, F26, F27, F28, F29, F30, F31,
|
||||
F16, F17, F28, F29, F30, F31,
|
||||
};
|
||||
|
||||
count = ARRAY_SIZE(allocationOrder);
|
||||
|
||||
@@ -2317,6 +2317,34 @@ void RExp2(SinCosArg arg, float *output) {
|
||||
output[0] = vfpu_rexp2(arg);
|
||||
}
|
||||
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
static float NegRcp(float x) {
|
||||
return -vfpu_rcp(x);
|
||||
}
|
||||
|
||||
static float NegSin(float x) {
|
||||
return -vfpu_sin(x);
|
||||
}
|
||||
|
||||
// The VV2Op math functions, called with CallProtectedLeaf. They take and return their float in XMM0.
|
||||
static float (*VV2OpMathFunc(int optype))(float) {
|
||||
switch (optype) {
|
||||
case 16: return &vfpu_rcp;
|
||||
case 17: return &vfpu_rsqrt;
|
||||
case 18: return &vfpu_sin;
|
||||
case 19: return &vfpu_cos;
|
||||
case 20: return &vfpu_exp2;
|
||||
case 21: return &vfpu_log2;
|
||||
case 22: return &vfpu_sqrt;
|
||||
case 23: return &vfpu_asin;
|
||||
case 24: return &NegRcp;
|
||||
case 26: return &NegSin;
|
||||
case 28: return &vfpu_rexp2;
|
||||
default: return nullptr;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
void Jit::Comp_VV2Op(MIPSOpcode op) {
|
||||
CONDITIONAL_DISABLE(VFPU_VEC);
|
||||
|
||||
@@ -2445,10 +2473,22 @@ void Jit::Comp_VV2Op(MIPSOpcode op) {
|
||||
}
|
||||
}
|
||||
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
float (*mathFunc)(float) = VV2OpMathFunc((op >> 16) & 0x1f);
|
||||
#endif
|
||||
|
||||
// Warning: sregs[i] and tempxregs[i] may be the same reg.
|
||||
// Helps for vmov, hurts for vrcp, etc.
|
||||
for (int i = 0; i < n; ++i)
|
||||
{
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
if (mathFunc) {
|
||||
MOVSS(XMM0, fpr.V(sregs[i]));
|
||||
CallProtectedLeaf((const void *)mathFunc);
|
||||
MOVSS(tempxregs[i], R(XMM0));
|
||||
continue;
|
||||
}
|
||||
#endif
|
||||
switch ((op >> 16) & 0x1f)
|
||||
{
|
||||
case 0: // d[i] = s[i]; break; //vmov
|
||||
@@ -3719,15 +3759,22 @@ void Jit::Comp_VRot(MIPSOpcode op) {
|
||||
if (vd2 >= 0)
|
||||
GetVectorRegs(dregs2, sz, vd2);
|
||||
GetVectorRegs(&sreg, V_Single, vs);
|
||||
// With the angle in a destination lane, the cosine is taken of what was written there.
|
||||
// The assembler refuses that, so leave it to the interpreter, and don't pair such a vrot.
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (dregs[i] == sreg) {
|
||||
DISABLE;
|
||||
}
|
||||
if (vd2 >= 0 && dregs2[i] == sreg) {
|
||||
vd2 = -1;
|
||||
}
|
||||
}
|
||||
|
||||
// Flush SIMD.
|
||||
fpr.SimpleRegsV(&sreg, V_Single, 0);
|
||||
|
||||
int imm = (op >> 16) & 0x1f;
|
||||
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.Flush();
|
||||
|
||||
bool negSin1 = (imm & 0x10) ? true : false;
|
||||
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
@@ -3737,8 +3784,11 @@ void Jit::Comp_VRot(MIPSOpcode op) {
|
||||
LEA(64, RDI, MIPSSTATE_VAR(sincostemp));
|
||||
#endif
|
||||
MOVSS(XMM0, fpr.V(sreg));
|
||||
ABI_CallFunction(negSin1 ? (const void *)&SinCosNegSin : (const void *)&SinCos);
|
||||
CallProtectedLeaf(negSin1 ? (const void *)&SinCosNegSin : (const void *)&SinCos);
|
||||
#else
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.Flush();
|
||||
|
||||
// Sigh, passing floats with cdecl isn't pretty, ends up on the stack.
|
||||
ABI_CallFunctionAC(negSin1 ? (const void *)&SinCosNegSin : (const void *)&SinCos, fpr.V(sreg), (uintptr_t)mips_->sincostemp);
|
||||
#endif
|
||||
|
||||
@@ -907,6 +907,49 @@ void Jit::CallProtectedFunction(const void *func, const OpArg &arg1, const u32 a
|
||||
ABI_CallFunctionAC(thunks.ProtectFunction(func, 2), arg1, arg2);
|
||||
}
|
||||
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
void Jit::CallProtectedLeaf(const void *func) {
|
||||
// The registers a call may clobber, apart from RAX and XMM0-1, which are scratch.
|
||||
#ifdef _WIN32
|
||||
static const X64Reg callerSavedGPRs[] = { RCX, RDX, R8, R9, R10, R11 };
|
||||
const int lastXMM = 5;
|
||||
const int shadowSpace = 32;
|
||||
#else
|
||||
static const X64Reg callerSavedGPRs[] = { RCX, RDX, R8, R9, R10, R11, RSI, RDI };
|
||||
const int lastXMM = 15;
|
||||
const int shadowSpace = 0;
|
||||
#endif
|
||||
X64Reg xmms[16];
|
||||
X64Reg gprs[ARRAY_SIZE(callerSavedGPRs)];
|
||||
int numXMMs = 0, numGPRs = 0;
|
||||
for (int i = 2; i <= lastXMM; i++) {
|
||||
if (fpr.IsXRegInUse((X64Reg)(XMM0 + i)))
|
||||
xmms[numXMMs++] = (X64Reg)(XMM0 + i);
|
||||
}
|
||||
for (X64Reg reg : callerSavedGPRs) {
|
||||
if (gpr.IsXRegInUse(reg))
|
||||
gprs[numGPRs++] = reg;
|
||||
}
|
||||
|
||||
// The JIT runs with RSP 16-byte aligned, so this keeps it aligned for the call.
|
||||
const int gprBase = shadowSpace + numXMMs * 16;
|
||||
const int frameSize = (gprBase + numGPRs * 8 + 15) & ~15;
|
||||
if (frameSize)
|
||||
SUB(64, R(RSP), Imm32(frameSize));
|
||||
for (int i = 0; i < numXMMs; i++)
|
||||
MOVAPS(MDisp(RSP, shadowSpace + i * 16), xmms[i]);
|
||||
for (int i = 0; i < numGPRs; i++)
|
||||
MOV(64, MDisp(RSP, gprBase + i * 8), R(gprs[i]));
|
||||
ABI_CallFunction(func);
|
||||
for (int i = 0; i < numXMMs; i++)
|
||||
MOVAPS(xmms[i], MDisp(RSP, shadowSpace + i * 16));
|
||||
for (int i = 0; i < numGPRs; i++)
|
||||
MOV(64, R(gprs[i]), MDisp(RSP, gprBase + i * 8));
|
||||
if (frameSize)
|
||||
ADD(64, R(RSP), Imm32(frameSize));
|
||||
}
|
||||
#endif
|
||||
|
||||
void Jit::Comp_DoNothing(MIPSOpcode op) { }
|
||||
|
||||
MIPSOpcode Jit::GetOriginalOp(MIPSOpcode op) {
|
||||
|
||||
@@ -246,6 +246,12 @@ private:
|
||||
void CallProtectedFunction(const void *func, const Gen::OpArg &arg1, const Gen::OpArg &arg2);
|
||||
void CallProtectedFunction(const void *func, const Gen::OpArg &arg1, const u32 arg2);
|
||||
void CallProtectedFunction(const void *func, const u32 arg1, const u32 arg2);
|
||||
#if PPSSPP_ARCH(AMD64)
|
||||
// A lighter CallProtectedFunction for a function that leaves MXCSR alone: saves only the
|
||||
// caller-saved registers the caches are using. Set up the arguments first - the argument
|
||||
// registers are never cached.
|
||||
void CallProtectedLeaf(const void *func);
|
||||
#endif
|
||||
|
||||
template <typename Tr, typename T1>
|
||||
void CallProtectedFunction(Tr (*func)(T1), const Gen::OpArg &arg1) {
|
||||
|
||||
@@ -119,6 +119,11 @@ public:
|
||||
void GetState(GPRRegCacheState &state) const;
|
||||
void RestoreState(const GPRRegCacheState& state);
|
||||
|
||||
// Whether the register holds something the cache cares about: a MIPS register or a locked temp.
|
||||
bool IsXRegInUse(Gen::X64Reg reg) const {
|
||||
return !xregs[reg].free || xregs[reg].allocLocked;
|
||||
}
|
||||
|
||||
MIPSState *mips_ = nullptr;
|
||||
|
||||
private:
|
||||
|
||||
@@ -172,6 +172,10 @@ public:
|
||||
bool IsMappedV(int v) {
|
||||
return vregs[v].lane == 0 && V(v).IsSimpleReg();
|
||||
}
|
||||
// Whether the register holds a MIPS register or a temp.
|
||||
bool IsXRegInUse(Gen::X64Reg reg) const {
|
||||
return xregs[reg].mipsReg != -1;
|
||||
}
|
||||
bool IsMappedVS(u8 v) {
|
||||
return vregs[v].lane != 0 && VS(&v).IsSimpleReg();
|
||||
}
|
||||
|
||||
@@ -978,6 +978,10 @@ static float X64JIT_XMM_CALL x64_log2(float f) {
|
||||
return vfpu_log2(f);
|
||||
}
|
||||
|
||||
static double X64JIT_XMM_CALL x64_sincos(float f) {
|
||||
return vfpu_sincos_packed(f);
|
||||
}
|
||||
|
||||
static float X64JIT_XMM_CALL x64_h2f_lower(float f) {
|
||||
return vfpu_h2f_lower(f);
|
||||
}
|
||||
@@ -1150,6 +1154,38 @@ void X64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
callFuncF_F(inst.src2 ? (const void *)&x64_h2f_upper : (const void *)&x64_h2f_lower);
|
||||
break;
|
||||
|
||||
case IROp::FSinCos:
|
||||
{
|
||||
#if X64JIT_USE_XMM_CALL
|
||||
// The helper returns the sine and cosine packed into the low 64 bits of XMM0. The cache
|
||||
// can hand out XMM0 when mapping below, so park them in sincostemp meanwhile.
|
||||
regs_.FlushBeforeCall();
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::MATH_HELPER);
|
||||
if (regs_.IsFPRMapped(inst.src1)) {
|
||||
int lane = regs_.GetFPRLane(inst.src1);
|
||||
CopyVec4ToFPRLane0(XMM0, regs_.FX(inst.src1), lane);
|
||||
} else {
|
||||
// Account for CTXREG being increased by 128 to reduce imm sizes.
|
||||
MOVSS(XMM0, MDisp(CTXREG, offsetof(MIPSState, f) + inst.src1 * 4 - 128));
|
||||
}
|
||||
ABI_CallFunction((const void *)&x64_sincos);
|
||||
MOVSD(MDisp(CTXREG, offsetof(MIPSState, sincostemp) - 128), XMM0);
|
||||
regs_.Map(inst);
|
||||
MOVSD(regs_.FX(inst.dest), MDisp(CTXREG, offsetof(MIPSState, sincostemp) - 128));
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
#else
|
||||
// Two calls here. The frontend makes sure dest doesn't overlap src1.
|
||||
IRInst sinInst = inst;
|
||||
sinInst.op = IROp::FSin;
|
||||
CompIR_FSpecial(sinInst);
|
||||
IRInst cosInst = inst;
|
||||
cosInst.op = IROp::FCos;
|
||||
cosInst.dest = inst.dest + 1;
|
||||
CompIR_FSpecial(cosInst);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
+1
-1
Submodule pspautotests updated: 80eb3ae038...6f03ee6457.
Reference in new issue
Block a user