mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
VFPU: vsqrt, vrsq, vrcp and vnrcp are exact in every backend
These went through the host's sqrt and division everywhere except the interpreter's vrcp and vnrcp (vsqrt and vrsq there only behind USE_VFPU_SQRT, now gone). They now always give the PSP's bits: the IR gets FVSqrt (FSqrt stays the FPU's IEEE sqrt.s), and FRSqrt and FRecip, which only the VFPU emits, become vfpu_rsqrt and vfpu_rcp; the IR interpreter and the x64, arm64, RISC-V and LoongArch backends call them. The old JITs call them directly, the ARM ones keeping the lanes in callee-saved registers across the calls. cpu/vfpu/exact now passes on every core. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
1 parent
87ad32d578
commit
a2b10778be
17 files changed
+225
-132
No files matched your search
@@ -19,6 +19,7 @@
|
||||
#if PPSSPP_ARCH(ARM)
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include "Common/CPUDetect.h"
|
||||
#include "Common/Data/Convert/SmallDataConvert.h"
|
||||
#include "Common/Math/math_util.h"
|
||||
@@ -911,6 +912,12 @@ namespace MIPSComp
|
||||
case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2
|
||||
DISABLE;
|
||||
break;
|
||||
case 16: // vrcp
|
||||
case 17: // vrsq
|
||||
case 22: // vsqrt
|
||||
case 24: // vnrcp
|
||||
CompVV2OpCall(op);
|
||||
return;
|
||||
case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
|
||||
DISABLE;
|
||||
break;
|
||||
@@ -1007,23 +1014,6 @@ namespace MIPSComp
|
||||
VMOV(fpr.V(tempregs[i]), S1);
|
||||
SetCC(CC_AL);
|
||||
break;
|
||||
case 16: // d[i] = 1.0f / s[i]; break; //vrcp
|
||||
if (i == 0) {
|
||||
MOVI2F(S0, 1.0f, SCRATCHREG1);
|
||||
}
|
||||
VDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
|
||||
break;
|
||||
case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
|
||||
if (i == 0) {
|
||||
MOVI2F(S0, 1.0f, SCRATCHREG1);
|
||||
}
|
||||
VSQRT(S1, fpr.V(sregs[i]));
|
||||
VDIV(fpr.V(tempregs[i]), S0, S1);
|
||||
break;
|
||||
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
|
||||
VSQRT(fpr.V(tempregs[i]), fpr.V(sregs[i]));
|
||||
VABS(fpr.V(tempregs[i]), fpr.V(tempregs[i]));
|
||||
break;
|
||||
case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin
|
||||
// Seems to work well enough but can disable if it becomes a problem.
|
||||
// Should be easy enough to translate to NEON. There we can load all the constants
|
||||
@@ -1052,12 +1042,6 @@ namespace MIPSComp
|
||||
MOVI2F(S1, 1.0f / (M_PI / 2), SCRATCHREG1);
|
||||
VMUL(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S1);
|
||||
break;
|
||||
case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
|
||||
if (i == 0) {
|
||||
MOVI2F(S0, -1.0f, SCRATCHREG1);
|
||||
}
|
||||
VDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
|
||||
break;
|
||||
default:
|
||||
ERROR_LOG(Log::JIT, "case missing in vfpu vv2op");
|
||||
DISABLE;
|
||||
@@ -2120,6 +2104,75 @@ namespace MIPSComp
|
||||
fpr.ReleaseSpillLocksAndDiscardTemps();
|
||||
}
|
||||
|
||||
// Float bits in and out, so the calls look the same with softfp and hardfp.
|
||||
static u32 CallWithBits(float (*func)(float), u32 x) {
|
||||
float f;
|
||||
memcpy(&f, &x, sizeof(f));
|
||||
f = func(f);
|
||||
memcpy(&x, &f, sizeof(x));
|
||||
return x;
|
||||
}
|
||||
|
||||
static u32 VRcpBits(u32 x) {
|
||||
return CallWithBits(&vfpu_rcp, x);
|
||||
}
|
||||
|
||||
static u32 VNRcpBits(u32 x) {
|
||||
return CallWithBits(&vfpu_rcp, x) ^ 0x80000000;
|
||||
}
|
||||
|
||||
static u32 VRSqrtBits(u32 x) {
|
||||
return CallWithBits(&vfpu_rsqrt, x);
|
||||
}
|
||||
|
||||
static u32 VSqrtBits(u32 x) {
|
||||
return CallWithBits(&vfpu_sqrt, x);
|
||||
}
|
||||
|
||||
// vrcp, vrsq, vsqrt and vnrcp call the exact functions. The lanes stay in S16-S19 across the
|
||||
// calls (callee-saved), and the results are stored to the destinations' homes.
|
||||
void ArmJit::CompVV2OpCall(MIPSOpcode op) {
|
||||
// Like the interpreter, these apply the prefixes to the last lane only.
|
||||
if (js.HasSPrefix() || (js.HasDPrefix() && GetVecSize(op) != V_Single)) {
|
||||
DISABLE;
|
||||
}
|
||||
|
||||
const int optype = (op >> 16) & 0x1f;
|
||||
u32 (*func)(u32) = &VRcpBits;
|
||||
if (optype == 17) {
|
||||
func = &VRSqrtBits;
|
||||
} else if (optype == 22) {
|
||||
func = &VSqrtBits;
|
||||
} else if (optype == 24) {
|
||||
func = &VNRcpBits;
|
||||
}
|
||||
|
||||
VectorSize sz = GetVecSize(op);
|
||||
int n = GetNumVectorElements(sz);
|
||||
u8 sregs[4], dregs[4];
|
||||
GetVectorRegs(sregs, sz, _VS);
|
||||
GetVectorRegs(dregs, sz, _VD);
|
||||
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.FlushAll();
|
||||
|
||||
for (int i = 0; i < n; i++) {
|
||||
VLDR((ARMReg)(S16 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
VMOV(R0, (ARMReg)(S16 + i));
|
||||
// FlushBeforeCall saves R1.
|
||||
QuickCallFunction(R1, func);
|
||||
VMOV((ARMReg)(S16 + i), R0);
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
VSTR((ARMReg)(S16 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
|
||||
}
|
||||
|
||||
ApplyPrefixD(dregs, sz);
|
||||
fpr.ReleaseSpillLocksAndDiscardTemps();
|
||||
}
|
||||
|
||||
static double SinCos(float angle) {
|
||||
union { struct { float sin; float cos; }; double out; } sincos;
|
||||
vfpu_sincos(angle, sincos.sin, sincos.cos);
|
||||
|
||||
@@ -196,6 +196,7 @@ private:
|
||||
void CompShiftImm(MIPSOpcode op, ArmGen::ShiftType shiftType, int sa);
|
||||
void CompShiftVar(MIPSOpcode op, ArmGen::ShiftType shiftType);
|
||||
void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin);
|
||||
void CompVV2OpCall(MIPSOpcode op);
|
||||
|
||||
void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz);
|
||||
void ApplyPrefixD(const u8 *vregs, VectorSize sz);
|
||||
|
||||
@@ -166,6 +166,9 @@ public:
|
||||
void SetEmitter(ArmGen::ARMXEmitter *emitter) { emit_ = emitter; }
|
||||
|
||||
int GetMipsRegOffset(MIPSReg r);
|
||||
int GetMipsRegOffsetV(MIPSReg r) {
|
||||
return GetMipsRegOffset(r + 32);
|
||||
}
|
||||
|
||||
private:
|
||||
bool Consecutive(int v1, int v2) const;
|
||||
@@ -174,9 +177,6 @@ private:
|
||||
|
||||
MIPSReg GetTempR();
|
||||
const ArmGen::ARMReg *GetMIPSAllocationOrder(int &count);
|
||||
int GetMipsRegOffsetV(MIPSReg r) {
|
||||
return GetMipsRegOffset(r + 32);
|
||||
}
|
||||
// This one WILL get a free quad as long as you haven't spill-locked them all.
|
||||
int QGetFreeQuad(int start, int count, const char *reason);
|
||||
|
||||
|
||||
@@ -861,6 +861,12 @@ namespace MIPSComp {
|
||||
case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2
|
||||
DISABLE;
|
||||
break;
|
||||
case 16: // vrcp
|
||||
case 17: // vrsq
|
||||
case 22: // vsqrt
|
||||
case 24: // vnrcp
|
||||
CompVV2OpCall(op);
|
||||
return;
|
||||
case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
|
||||
DISABLE;
|
||||
break;
|
||||
@@ -930,32 +936,9 @@ namespace MIPSComp {
|
||||
fp.FMAX(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S0);
|
||||
fp.FMIN(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S1);
|
||||
break;
|
||||
case 16: // d[i] = 1.0f / s[i]; break; //vrcp
|
||||
if (i == 0) {
|
||||
fp.MOVI2F(S0, 1.0f, SCRATCH1);
|
||||
}
|
||||
fp.FDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
|
||||
break;
|
||||
case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
|
||||
if (i == 0) {
|
||||
fp.MOVI2F(S0, 1.0f, SCRATCH1);
|
||||
}
|
||||
fp.FSQRT(S1, fpr.V(sregs[i]));
|
||||
fp.FDIV(fpr.V(tempregs[i]), S0, S1);
|
||||
break;
|
||||
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
|
||||
fp.FSQRT(fpr.V(tempregs[i]), fpr.V(sregs[i]));
|
||||
fp.FABS(fpr.V(tempregs[i]), fpr.V(tempregs[i]));
|
||||
break;
|
||||
case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin
|
||||
DISABLE;
|
||||
break;
|
||||
case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
|
||||
if (i == 0) {
|
||||
fp.MOVI2F(S0, -1.0f, SCRATCH1);
|
||||
}
|
||||
fp.FDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
|
||||
break;
|
||||
default:
|
||||
ERROR_LOG(Log::JIT, "case missing in vfpu vv2op");
|
||||
DISABLE;
|
||||
@@ -975,6 +958,45 @@ namespace MIPSComp {
|
||||
fpr.ReleaseSpillLocksAndDiscardTemps();
|
||||
}
|
||||
|
||||
// vrcp, vrsq, vsqrt and vnrcp call the exact functions. The lanes stay in S8-S11 across the
|
||||
// calls (callee-saved), and the results are stored to the destinations' homes.
|
||||
void Arm64Jit::CompVV2OpCall(MIPSOpcode op) {
|
||||
if (js.HasSPrefix()) {
|
||||
DISABLE;
|
||||
}
|
||||
|
||||
const int optype = (op >> 16) & 0x1f;
|
||||
float (*func)(float) = optype == 17 ? &vfpu_rsqrt : (optype == 22 ? &vfpu_sqrt : &vfpu_rcp);
|
||||
|
||||
VectorSize sz = GetVecSize(op);
|
||||
int n = GetNumVectorElements(sz);
|
||||
u8 sregs[4], dregs[4];
|
||||
GetVectorRegs(sregs, sz, _VS);
|
||||
GetVectorRegs(dregs, sz, _VD);
|
||||
|
||||
gpr.FlushBeforeCall();
|
||||
fpr.FlushAll();
|
||||
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.FMOV(S0, (ARM64Reg)(S8 + i));
|
||||
QuickCallFunction(SCRATCH2_64, func);
|
||||
if (optype == 24) {
|
||||
fp.FNEG((ARM64Reg)(S8 + i), S0);
|
||||
} else {
|
||||
fp.FMOV((ARM64Reg)(S8 + i), S0);
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < n; i++) {
|
||||
fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
|
||||
}
|
||||
|
||||
ApplyPrefixD(dregs, sz);
|
||||
fpr.ReleaseSpillLocksAndDiscardTemps();
|
||||
}
|
||||
|
||||
void Arm64Jit::Comp_Vi2f(MIPSOpcode op) {
|
||||
CONDITIONAL_DISABLE(VFPU_VEC);
|
||||
if (js.HasUnknownPrefix()) {
|
||||
|
||||
@@ -548,22 +548,21 @@ void Arm64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
break;
|
||||
|
||||
case IROp::FRSqrt:
|
||||
regs_.Map(inst);
|
||||
fp_.MOVI2F(SCRATCHF1, 1.0f);
|
||||
fp_.FSQRT(regs_.F(inst.dest), regs_.F(inst.src1));
|
||||
fp_.FDIV(regs_.F(inst.dest), SCRATCHF1, regs_.F(inst.dest));
|
||||
callFuncF_F(&vfpu_rsqrt);
|
||||
break;
|
||||
|
||||
case IROp::FRecip:
|
||||
regs_.Map(inst);
|
||||
fp_.MOVI2F(SCRATCHF1, 1.0f);
|
||||
fp_.FDIV(regs_.F(inst.dest), SCRATCHF1, regs_.F(inst.src1));
|
||||
callFuncF_F(&vfpu_rcp);
|
||||
break;
|
||||
|
||||
case IROp::FAsin:
|
||||
callFuncF_F(&vfpu_asin);
|
||||
break;
|
||||
|
||||
case IROp::FVSqrt:
|
||||
callFuncF_F(&vfpu_sqrt);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
@@ -236,6 +236,7 @@ private:
|
||||
void CompShiftImm(MIPSOpcode op, Arm64Gen::ShiftType shiftType, int sa);
|
||||
void CompShiftVar(MIPSOpcode op, Arm64Gen::ShiftType shiftType);
|
||||
void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin);
|
||||
void CompVV2OpCall(MIPSOpcode op);
|
||||
|
||||
void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz);
|
||||
void ApplyPrefixD(const u8 *vregs, VectorSize sz);
|
||||
|
||||
@@ -1128,7 +1128,7 @@ namespace MIPSComp {
|
||||
DISABLE;
|
||||
break;
|
||||
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
|
||||
ir.Write(IROp::FSqrt, tempregs[i], sregs[i]);
|
||||
ir.Write(IROp::FVSqrt, tempregs[i], sregs[i]);
|
||||
break;
|
||||
case 23: // d[i] = asinf(s[i]) / M_PI_2; break; //vasin
|
||||
ir.Write(IROp::FAsin, tempregs[i], sregs[i]);
|
||||
|
||||
@@ -109,6 +109,7 @@ static const IRMeta irMeta[] = {
|
||||
{ IROp::FRSqrt, "FRSqrt", "FF" },
|
||||
{ IROp::FRecip, "FRecip", "FF" },
|
||||
{ IROp::FAsin, "FAsin", "FF" },
|
||||
{ IROp::FVSqrt, "FVSqrt", "FF" },
|
||||
{ IROp::FNeg, "FNeg", "FF" },
|
||||
{ IROp::FSign, "FSign", "FF" },
|
||||
{ IROp::FAbs, "FAbs", "FF" },
|
||||
|
||||
@@ -197,9 +197,10 @@ enum class IROp : uint8_t {
|
||||
// Slow special functions. Used on singles.
|
||||
FSin,
|
||||
FCos,
|
||||
FRSqrt,
|
||||
FRecip,
|
||||
FRSqrt, // vrsq, bit-exact with the PSP
|
||||
FRecip, // vrcp, bit-exact with the PSP
|
||||
FAsin,
|
||||
FVSqrt, // vsqrt, bit-exact with the PSP (FSqrt is the FPU's IEEE sqrt.s)
|
||||
|
||||
// Fake/System instructions
|
||||
Interpret,
|
||||
|
||||
@@ -629,14 +629,17 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
|
||||
mips->f[inst->dest] = vfpu_cos(mips->f[inst->src1]);
|
||||
break;
|
||||
case IROp::FRSqrt:
|
||||
mips->f[inst->dest] = 1.0f / sqrtf(mips->f[inst->src1]);
|
||||
mips->f[inst->dest] = vfpu_rsqrt(mips->f[inst->src1]);
|
||||
break;
|
||||
case IROp::FRecip:
|
||||
mips->f[inst->dest] = 1.0f / mips->f[inst->src1];
|
||||
mips->f[inst->dest] = vfpu_rcp(mips->f[inst->src1]);
|
||||
break;
|
||||
case IROp::FAsin:
|
||||
mips->f[inst->dest] = vfpu_asin(mips->f[inst->src1]);
|
||||
break;
|
||||
case IROp::FVSqrt:
|
||||
mips->f[inst->dest] = vfpu_sqrt(mips->f[inst->src1]);
|
||||
break;
|
||||
|
||||
case IROp::ShlImm:
|
||||
mips->r[inst->dest] = mips->r[inst->src1] << (int)inst->src2;
|
||||
|
||||
@@ -432,6 +432,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) {
|
||||
case IROp::FRSqrt:
|
||||
case IROp::FRecip:
|
||||
case IROp::FAsin:
|
||||
case IROp::FVSqrt:
|
||||
CompIR_FSpecial(inst);
|
||||
break;
|
||||
|
||||
|
||||
@@ -765,6 +765,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::FRSqrt:
|
||||
case IROp::FRecip:
|
||||
case IROp::FAsin:
|
||||
case IROp::FVSqrt:
|
||||
out.Write(inst);
|
||||
break;
|
||||
|
||||
|
||||
@@ -75,7 +75,6 @@
|
||||
#endif
|
||||
|
||||
static const bool USE_VFPU_DOT = false;
|
||||
static const bool USE_VFPU_SQRT = false;
|
||||
|
||||
union FloatBits {
|
||||
float f[4];
|
||||
@@ -665,13 +664,13 @@ namespace MIPSInt
|
||||
case 4: if (s[i] <= 0) d[i] = 0; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat0
|
||||
case 5: if (s[i] < -1.0f) d[i] = -1.0f; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat1
|
||||
case 16: { d[i] = vfpu_rcp(s[i]); } break; //vrcp
|
||||
case 17: d[i] = USE_VFPU_SQRT ? vfpu_rsqrt(s[i]) : 1.0f / sqrtf(s[i]); break; //vrsq
|
||||
case 17: d[i] = vfpu_rsqrt(s[i]); break; //vrsq
|
||||
|
||||
case 18: { d[i] = vfpu_sin(s[i]); } break; //vsin
|
||||
case 19: { d[i] = vfpu_cos(s[i]); } break; //vcos
|
||||
case 20: { d[i] = vfpu_exp2(s[i]); } break; //vexp2
|
||||
case 21: { d[i] = vfpu_log2(s[i]); } break; //vlog2
|
||||
case 22: d[i] = USE_VFPU_SQRT ? vfpu_sqrt(s[i]) : fabsf(sqrtf(s[i])); break; //vsqrt
|
||||
case 22: d[i] = vfpu_sqrt(s[i]); break; //vsqrt
|
||||
case 23: { d[i] = vfpu_asin(s[i]); } break; //vasin
|
||||
case 24: { d[i] = -vfpu_rcp(s[i]); } break; // vnrcp
|
||||
case 26: { d[i] = -vfpu_sin(s[i]); } break; // vnsin
|
||||
|
||||
@@ -593,19 +593,21 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
break;
|
||||
|
||||
case IROp::FRSqrt:
|
||||
regs_.Map(inst);
|
||||
FRSQRT_S(regs_.F(inst.dest), regs_.F(inst.src1));
|
||||
callFuncF_F(&vfpu_rsqrt);
|
||||
break;
|
||||
|
||||
case IROp::FRecip:
|
||||
regs_.Map(inst);
|
||||
FRECIP_S(regs_.F(inst.dest), regs_.F(inst.src1));
|
||||
callFuncF_F(&vfpu_rcp);
|
||||
break;
|
||||
|
||||
case IROp::FAsin:
|
||||
callFuncF_F(&vfpu_asin);
|
||||
break;
|
||||
|
||||
case IROp::FVSqrt:
|
||||
callFuncF_F(&vfpu_sqrt);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
@@ -608,7 +608,6 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
|
||||
};
|
||||
|
||||
RiscVReg tempReg = INVALID_REG;
|
||||
switch (inst.op) {
|
||||
case IROp::FSin:
|
||||
callFuncF_F(&vfpu_sin);
|
||||
@@ -619,32 +618,21 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
break;
|
||||
|
||||
case IROp::FRSqrt:
|
||||
tempReg = regs_.MapWithFPRTemp(inst);
|
||||
FSQRT(32, regs_.F(inst.dest), regs_.F(inst.src1));
|
||||
|
||||
// Ugh, we can't really avoid a temp here. Probably not worth a permanent one.
|
||||
QuickFLI(32, tempReg, 1.0f, SCRATCH1);
|
||||
FDIV(32, regs_.F(inst.dest), tempReg, regs_.F(inst.dest));
|
||||
callFuncF_F(&vfpu_rsqrt);
|
||||
break;
|
||||
|
||||
case IROp::FRecip:
|
||||
if (inst.dest != inst.src1) {
|
||||
// This is the easy case.
|
||||
regs_.Map(inst);
|
||||
LI(SCRATCH1, 1.0f);
|
||||
FMV(FMv::W, FMv::X, regs_.F(inst.dest), SCRATCH1);
|
||||
FDIV(32, regs_.F(inst.dest), regs_.F(inst.dest), regs_.F(inst.src1));
|
||||
} else {
|
||||
tempReg = regs_.MapWithFPRTemp(inst);
|
||||
QuickFLI(32, tempReg, 1.0f, SCRATCH1);
|
||||
FDIV(32, regs_.F(inst.dest), tempReg, regs_.F(inst.src1));
|
||||
}
|
||||
callFuncF_F(&vfpu_rcp);
|
||||
break;
|
||||
|
||||
case IROp::FAsin:
|
||||
callFuncF_F(&vfpu_asin);
|
||||
break;
|
||||
|
||||
case IROp::FVSqrt:
|
||||
callFuncF_F(&vfpu_sqrt);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
+24
-24
@@ -2289,6 +2289,22 @@ void SinCosNegSin(SinCosArg angle, float *output) {
|
||||
output[0] = -output[0];
|
||||
}
|
||||
|
||||
void VSqrt(SinCosArg arg, float *output) {
|
||||
output[0] = vfpu_sqrt(arg);
|
||||
}
|
||||
|
||||
void VRSqrt(SinCosArg arg, float *output) {
|
||||
output[0] = vfpu_rsqrt(arg);
|
||||
}
|
||||
|
||||
void VRcp(SinCosArg arg, float *output) {
|
||||
output[0] = vfpu_rcp(arg);
|
||||
}
|
||||
|
||||
void VNRcp(SinCosArg arg, float *output) {
|
||||
output[0] = -vfpu_rcp(arg);
|
||||
}
|
||||
|
||||
void Exp2(SinCosArg arg, float *output) {
|
||||
output[0] = vfpu_exp2(arg);
|
||||
}
|
||||
@@ -2496,24 +2512,12 @@ void Jit::Comp_VV2Op(MIPSOpcode op) {
|
||||
MINSS(tempxregs[i], R(XMM0));
|
||||
break;
|
||||
case 16: // d[i] = 1.0f / s[i]; break; //vrcp
|
||||
if (RipAccessible(&one)) {
|
||||
MOVSS(XMM0, M(&one)); // rip accessible
|
||||
} else {
|
||||
MOV(PTRBITS, R(TEMPREG), ImmPtr(&one));
|
||||
MOVSS(XMM0, MatR(TEMPREG));
|
||||
}
|
||||
DIVSS(XMM0, fpr.V(sregs[i]));
|
||||
MOVSS(tempxregs[i], R(XMM0));
|
||||
specialFuncCallHelper(&VRcp, sregs[i]);
|
||||
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
|
||||
break;
|
||||
case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
|
||||
SQRTSS(XMM0, fpr.V(sregs[i]));
|
||||
if (RipAccessible(&one)) {
|
||||
MOVSS(tempxregs[i], M(&one)); // rip accessible
|
||||
} else {
|
||||
MOV(PTRBITS, R(TEMPREG), ImmPtr(&one));
|
||||
MOVSS(tempxregs[i], MatR(TEMPREG));
|
||||
}
|
||||
DIVSS(tempxregs[i], R(XMM0));
|
||||
specialFuncCallHelper(&VRSqrt, sregs[i]);
|
||||
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
|
||||
break;
|
||||
case 18: // d[i] = sinf((float)M_PI_2 * s[i]); break; //vsin
|
||||
specialFuncCallHelper(&SinOnly, sregs[i]);
|
||||
@@ -2532,20 +2536,16 @@ void Jit::Comp_VV2Op(MIPSOpcode op) {
|
||||
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
|
||||
break;
|
||||
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
|
||||
SQRTSS(tempxregs[i], fpr.V(sregs[i]));
|
||||
MOV(PTRBITS, R(TEMPREG), ImmPtr(&noSignMask));
|
||||
ANDPS(tempxregs[i], MatR(TEMPREG));
|
||||
specialFuncCallHelper(&VSqrt, sregs[i]);
|
||||
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
|
||||
break;
|
||||
case 23: // d[i] = asinf(s[i]) / M_PI_2; break; //vasin
|
||||
specialFuncCallHelper(&ASinScaled, sregs[i]);
|
||||
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
|
||||
break;
|
||||
case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
|
||||
// Rare so let's not bother checking for RipAccessible.
|
||||
MOV(PTRBITS, R(TEMPREG), ImmPtr(&minus_one));
|
||||
MOVSS(XMM0, MatR(TEMPREG));
|
||||
DIVSS(XMM0, fpr.V(sregs[i]));
|
||||
MOVSS(tempxregs[i], R(XMM0));
|
||||
specialFuncCallHelper(&VNRcp, sregs[i]);
|
||||
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
|
||||
break;
|
||||
case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
|
||||
specialFuncCallHelper(&NegSinOnly, sregs[i]);
|
||||
|
||||
@@ -969,6 +969,18 @@ static float X64JIT_XMM_CALL x64_cos(float f) {
|
||||
static float X64JIT_XMM_CALL x64_asin(float f) {
|
||||
return vfpu_asin(f);
|
||||
}
|
||||
|
||||
static float X64JIT_XMM_CALL x64_vsqrt(float f) {
|
||||
return vfpu_sqrt(f);
|
||||
}
|
||||
|
||||
static float X64JIT_XMM_CALL x64_rsqrt(float f) {
|
||||
return vfpu_rsqrt(f);
|
||||
}
|
||||
|
||||
static float X64JIT_XMM_CALL x64_rcp(float f) {
|
||||
return vfpu_rcp(f);
|
||||
}
|
||||
#else
|
||||
static uint32_t x64_sin(uint32_t v) {
|
||||
float f;
|
||||
@@ -993,6 +1005,30 @@ static uint32_t x64_asin(uint32_t v) {
|
||||
memcpy(&v, &f, sizeof(v));
|
||||
return v;
|
||||
}
|
||||
|
||||
static uint32_t x64_vsqrt(uint32_t v) {
|
||||
float f;
|
||||
memcpy(&f, &v, sizeof(v));
|
||||
f = vfpu_sqrt(f);
|
||||
memcpy(&v, &f, sizeof(v));
|
||||
return v;
|
||||
}
|
||||
|
||||
static uint32_t x64_rsqrt(uint32_t v) {
|
||||
float f;
|
||||
memcpy(&f, &v, sizeof(v));
|
||||
f = vfpu_rsqrt(f);
|
||||
memcpy(&v, &f, sizeof(v));
|
||||
return v;
|
||||
}
|
||||
|
||||
static uint32_t x64_rcp(uint32_t v) {
|
||||
float f;
|
||||
memcpy(&f, &v, sizeof(v));
|
||||
f = vfpu_rcp(f);
|
||||
memcpy(&v, &f, sizeof(v));
|
||||
return v;
|
||||
}
|
||||
#endif
|
||||
|
||||
void X64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
@@ -1047,36 +1083,21 @@ void X64JitBackend::CompIR_FSpecial(IRInst inst) {
|
||||
break;
|
||||
|
||||
case IROp::FRSqrt:
|
||||
{
|
||||
X64Reg tempReg = regs_.MapWithFPRTemp(inst);
|
||||
SQRTSS(tempReg, regs_.F(inst.src1));
|
||||
|
||||
MOVSS(regs_.FX(inst.dest), M(constants.positiveOnes)); // rip accessible
|
||||
DIVSS(regs_.FX(inst.dest), R(tempReg));
|
||||
break;
|
||||
}
|
||||
callFuncF_F((const void *)&x64_rsqrt);
|
||||
break;
|
||||
|
||||
case IROp::FRecip:
|
||||
if (inst.dest != inst.src1) {
|
||||
regs_.Map(inst);
|
||||
MOVSS(regs_.FX(inst.dest), M(constants.positiveOnes)); // rip accessible
|
||||
DIVSS(regs_.FX(inst.dest), regs_.F(inst.src1));
|
||||
} else {
|
||||
X64Reg tempReg = regs_.MapWithFPRTemp(inst);
|
||||
MOVSS(tempReg, M(constants.positiveOnes)); // rip accessible
|
||||
if (cpu_info.bAVX) {
|
||||
VDIVSS(regs_.FX(inst.dest), tempReg, regs_.F(inst.src1));
|
||||
} else {
|
||||
DIVSS(tempReg, regs_.F(inst.src1));
|
||||
MOVSS(regs_.FX(inst.dest), R(tempReg));
|
||||
}
|
||||
}
|
||||
callFuncF_F((const void *)&x64_rcp);
|
||||
break;
|
||||
|
||||
case IROp::FAsin:
|
||||
callFuncF_F((const void *)&x64_asin);
|
||||
break;
|
||||
|
||||
case IROp::FVSqrt:
|
||||
callFuncF_F((const void *)&x64_vsqrt);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
Reference in new issue
Block a user