VFPU: vsqrt, vrsq, vrcp and vnrcp are exact in every backend

These went through the host's sqrt and division everywhere except the
interpreter's vrcp and vnrcp (vsqrt and vrsq there only behind
USE_VFPU_SQRT, now gone). They now always give the PSP's bits: the IR
gets FVSqrt (FSqrt stays the FPU's IEEE sqrt.s), and FRSqrt and FRecip,
which only the VFPU emits, become vfpu_rsqrt and vfpu_rcp; the IR
interpreter and the x64, arm64, RISC-V and LoongArch backends call them.
The old JITs call them directly, the ARM ones keeping the lanes in
callee-saved registers across the calls. cpu/vfpu/exact now passes on every core.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
Henrik RydgårdandClaude Opus 5.5 committed 2026-09-24 12:24:42 -06:00
1 parent 87ad32d578
commit a2b10778be
17 files changed
+225 -132

No files matched your search

+76 -23
View File
@@ -19,6 +19,7 @@
#if PPSSPP_ARCH(ARM)
#include <cmath>
#include <cstring>
#include "Common/CPUDetect.h"
#include "Common/Data/Convert/SmallDataConvert.h"
#include "Common/Math/math_util.h"
@@ -911,6 +912,12 @@ namespace MIPSComp
case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2
DISABLE;
break;
case 16: // vrcp
case 17: // vrsq
case 22: // vsqrt
case 24: // vnrcp
CompVV2OpCall(op);
return;
case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
DISABLE;
break;
@@ -1007,23 +1014,6 @@ namespace MIPSComp
VMOV(fpr.V(tempregs[i]), S1);
SetCC(CC_AL);
break;
case 16: // d[i] = 1.0f / s[i]; break; //vrcp
if (i == 0) {
MOVI2F(S0, 1.0f, SCRATCHREG1);
}
VDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
break;
case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
if (i == 0) {
MOVI2F(S0, 1.0f, SCRATCHREG1);
}
VSQRT(S1, fpr.V(sregs[i]));
VDIV(fpr.V(tempregs[i]), S0, S1);
break;
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
VSQRT(fpr.V(tempregs[i]), fpr.V(sregs[i]));
VABS(fpr.V(tempregs[i]), fpr.V(tempregs[i]));
break;
case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin
// Seems to work well enough but can disable if it becomes a problem.
// Should be easy enough to translate to NEON. There we can load all the constants
@@ -1052,12 +1042,6 @@ namespace MIPSComp
MOVI2F(S1, 1.0f / (M_PI / 2), SCRATCHREG1);
VMUL(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S1);
break;
case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
if (i == 0) {
MOVI2F(S0, -1.0f, SCRATCHREG1);
}
VDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
break;
default:
ERROR_LOG(Log::JIT, "case missing in vfpu vv2op");
DISABLE;
@@ -2120,6 +2104,75 @@ namespace MIPSComp
fpr.ReleaseSpillLocksAndDiscardTemps();
}
// Float bits in and out, so the calls look the same with softfp and hardfp.
static u32 CallWithBits(float (*func)(float), u32 x) {
float f;
memcpy(&f, &x, sizeof(f));
f = func(f);
memcpy(&x, &f, sizeof(x));
return x;
}
static u32 VRcpBits(u32 x) {
return CallWithBits(&vfpu_rcp, x);
}
static u32 VNRcpBits(u32 x) {
return CallWithBits(&vfpu_rcp, x) ^ 0x80000000;
}
static u32 VRSqrtBits(u32 x) {
return CallWithBits(&vfpu_rsqrt, x);
}
static u32 VSqrtBits(u32 x) {
return CallWithBits(&vfpu_sqrt, x);
}
// vrcp, vrsq, vsqrt and vnrcp call the exact functions. The lanes stay in S16-S19 across the
// calls (callee-saved), and the results are stored to the destinations' homes.
void ArmJit::CompVV2OpCall(MIPSOpcode op) {
// Like the interpreter, these apply the prefixes to the last lane only.
if (js.HasSPrefix() || (js.HasDPrefix() && GetVecSize(op) != V_Single)) {
DISABLE;
}
const int optype = (op >> 16) & 0x1f;
u32 (*func)(u32) = &VRcpBits;
if (optype == 17) {
func = &VRSqrtBits;
} else if (optype == 22) {
func = &VSqrtBits;
} else if (optype == 24) {
func = &VNRcpBits;
}
VectorSize sz = GetVecSize(op);
int n = GetNumVectorElements(sz);
u8 sregs[4], dregs[4];
GetVectorRegs(sregs, sz, _VS);
GetVectorRegs(dregs, sz, _VD);
gpr.FlushBeforeCall();
fpr.FlushAll();
for (int i = 0; i < n; i++) {
VLDR((ARMReg)(S16 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
}
for (int i = 0; i < n; i++) {
VMOV(R0, (ARMReg)(S16 + i));
// FlushBeforeCall saves R1.
QuickCallFunction(R1, func);
VMOV((ARMReg)(S16 + i), R0);
}
for (int i = 0; i < n; i++) {
VSTR((ARMReg)(S16 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
}
ApplyPrefixD(dregs, sz);
fpr.ReleaseSpillLocksAndDiscardTemps();
}
static double SinCos(float angle) {
union { struct { float sin; float cos; }; double out; } sincos;
vfpu_sincos(angle, sincos.sin, sincos.cos);
+1
View File
@@ -196,6 +196,7 @@ private:
void CompShiftImm(MIPSOpcode op, ArmGen::ShiftType shiftType, int sa);
void CompShiftVar(MIPSOpcode op, ArmGen::ShiftType shiftType);
void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin);
void CompVV2OpCall(MIPSOpcode op);
void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz);
void ApplyPrefixD(const u8 *vregs, VectorSize sz);
+3 -3
View File
@@ -166,6 +166,9 @@ public:
void SetEmitter(ArmGen::ARMXEmitter *emitter) { emit_ = emitter; }
int GetMipsRegOffset(MIPSReg r);
int GetMipsRegOffsetV(MIPSReg r) {
return GetMipsRegOffset(r + 32);
}
private:
bool Consecutive(int v1, int v2) const;
@@ -174,9 +177,6 @@ private:
MIPSReg GetTempR();
const ArmGen::ARMReg *GetMIPSAllocationOrder(int &count);
int GetMipsRegOffsetV(MIPSReg r) {
return GetMipsRegOffset(r + 32);
}
// This one WILL get a free quad as long as you haven't spill-locked them all.
int QGetFreeQuad(int start, int count, const char *reason);
+45 -23
View File
@@ -861,6 +861,12 @@ namespace MIPSComp {
case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2
DISABLE;
break;
case 16: // vrcp
case 17: // vrsq
case 22: // vsqrt
case 24: // vnrcp
CompVV2OpCall(op);
return;
case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
DISABLE;
break;
@@ -930,32 +936,9 @@ namespace MIPSComp {
fp.FMAX(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S0);
fp.FMIN(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S1);
break;
case 16: // d[i] = 1.0f / s[i]; break; //vrcp
if (i == 0) {
fp.MOVI2F(S0, 1.0f, SCRATCH1);
}
fp.FDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
break;
case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
if (i == 0) {
fp.MOVI2F(S0, 1.0f, SCRATCH1);
}
fp.FSQRT(S1, fpr.V(sregs[i]));
fp.FDIV(fpr.V(tempregs[i]), S0, S1);
break;
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
fp.FSQRT(fpr.V(tempregs[i]), fpr.V(sregs[i]));
fp.FABS(fpr.V(tempregs[i]), fpr.V(tempregs[i]));
break;
case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin
DISABLE;
break;
case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
if (i == 0) {
fp.MOVI2F(S0, -1.0f, SCRATCH1);
}
fp.FDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i]));
break;
default:
ERROR_LOG(Log::JIT, "case missing in vfpu vv2op");
DISABLE;
@@ -975,6 +958,45 @@ namespace MIPSComp {
fpr.ReleaseSpillLocksAndDiscardTemps();
}
// vrcp, vrsq, vsqrt and vnrcp call the exact functions. The lanes stay in S8-S11 across the
// calls (callee-saved), and the results are stored to the destinations' homes.
void Arm64Jit::CompVV2OpCall(MIPSOpcode op) {
if (js.HasSPrefix()) {
DISABLE;
}
const int optype = (op >> 16) & 0x1f;
float (*func)(float) = optype == 17 ? &vfpu_rsqrt : (optype == 22 ? &vfpu_sqrt : &vfpu_rcp);
VectorSize sz = GetVecSize(op);
int n = GetNumVectorElements(sz);
u8 sregs[4], dregs[4];
GetVectorRegs(sregs, sz, _VS);
GetVectorRegs(dregs, sz, _VD);
gpr.FlushBeforeCall();
fpr.FlushAll();
for (int i = 0; i < n; i++) {
fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i]));
}
for (int i = 0; i < n; i++) {
fp.FMOV(S0, (ARM64Reg)(S8 + i));
QuickCallFunction(SCRATCH2_64, func);
if (optype == 24) {
fp.FNEG((ARM64Reg)(S8 + i), S0);
} else {
fp.FMOV((ARM64Reg)(S8 + i), S0);
}
}
for (int i = 0; i < n; i++) {
fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i]));
}
ApplyPrefixD(dregs, sz);
fpr.ReleaseSpillLocksAndDiscardTemps();
}
void Arm64Jit::Comp_Vi2f(MIPSOpcode op) {
CONDITIONAL_DISABLE(VFPU_VEC);
if (js.HasUnknownPrefix()) {
+6 -7
View File
@@ -548,22 +548,21 @@ void Arm64JitBackend::CompIR_FSpecial(IRInst inst) {
break;
case IROp::FRSqrt:
regs_.Map(inst);
fp_.MOVI2F(SCRATCHF1, 1.0f);
fp_.FSQRT(regs_.F(inst.dest), regs_.F(inst.src1));
fp_.FDIV(regs_.F(inst.dest), SCRATCHF1, regs_.F(inst.dest));
callFuncF_F(&vfpu_rsqrt);
break;
case IROp::FRecip:
regs_.Map(inst);
fp_.MOVI2F(SCRATCHF1, 1.0f);
fp_.FDIV(regs_.F(inst.dest), SCRATCHF1, regs_.F(inst.src1));
callFuncF_F(&vfpu_rcp);
break;
case IROp::FAsin:
callFuncF_F(&vfpu_asin);
break;
case IROp::FVSqrt:
callFuncF_F(&vfpu_sqrt);
break;
default:
INVALIDOP;
break;
+1
View File
@@ -236,6 +236,7 @@ private:
void CompShiftImm(MIPSOpcode op, Arm64Gen::ShiftType shiftType, int sa);
void CompShiftVar(MIPSOpcode op, Arm64Gen::ShiftType shiftType);
void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin);
void CompVV2OpCall(MIPSOpcode op);
void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz);
void ApplyPrefixD(const u8 *vregs, VectorSize sz);
+1 -1
View File
@@ -1128,7 +1128,7 @@ namespace MIPSComp {
DISABLE;
break;
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
ir.Write(IROp::FSqrt, tempregs[i], sregs[i]);
ir.Write(IROp::FVSqrt, tempregs[i], sregs[i]);
break;
case 23: // d[i] = asinf(s[i]) / M_PI_2; break; //vasin
ir.Write(IROp::FAsin, tempregs[i], sregs[i]);
+1
View File
@@ -109,6 +109,7 @@ static const IRMeta irMeta[] = {
{ IROp::FRSqrt, "FRSqrt", "FF" },
{ IROp::FRecip, "FRecip", "FF" },
{ IROp::FAsin, "FAsin", "FF" },
{ IROp::FVSqrt, "FVSqrt", "FF" },
{ IROp::FNeg, "FNeg", "FF" },
{ IROp::FSign, "FSign", "FF" },
{ IROp::FAbs, "FAbs", "FF" },
+3 -2
View File
@@ -197,9 +197,10 @@ enum class IROp : uint8_t {
// Slow special functions. Used on singles.
FSin,
FCos,
FRSqrt,
FRecip,
FRSqrt, // vrsq, bit-exact with the PSP
FRecip, // vrcp, bit-exact with the PSP
FAsin,
FVSqrt, // vsqrt, bit-exact with the PSP (FSqrt is the FPU's IEEE sqrt.s)
// Fake/System instructions
Interpret,
+5 -2
View File
@@ -629,14 +629,17 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) {
mips->f[inst->dest] = vfpu_cos(mips->f[inst->src1]);
break;
case IROp::FRSqrt:
mips->f[inst->dest] = 1.0f / sqrtf(mips->f[inst->src1]);
mips->f[inst->dest] = vfpu_rsqrt(mips->f[inst->src1]);
break;
case IROp::FRecip:
mips->f[inst->dest] = 1.0f / mips->f[inst->src1];
mips->f[inst->dest] = vfpu_rcp(mips->f[inst->src1]);
break;
case IROp::FAsin:
mips->f[inst->dest] = vfpu_asin(mips->f[inst->src1]);
break;
case IROp::FVSqrt:
mips->f[inst->dest] = vfpu_sqrt(mips->f[inst->src1]);
break;
case IROp::ShlImm:
mips->r[inst->dest] = mips->r[inst->src1] << (int)inst->src2;
+1
View File
@@ -432,6 +432,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) {
case IROp::FRSqrt:
case IROp::FRecip:
case IROp::FAsin:
case IROp::FVSqrt:
CompIR_FSpecial(inst);
break;
+1
View File
@@ -765,6 +765,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::FRSqrt:
case IROp::FRecip:
case IROp::FAsin:
case IROp::FVSqrt:
out.Write(inst);
break;
+2 -3
View File
@@ -75,7 +75,6 @@
#endif
static const bool USE_VFPU_DOT = false;
static const bool USE_VFPU_SQRT = false;
union FloatBits {
float f[4];
@@ -665,13 +664,13 @@ namespace MIPSInt
case 4: if (s[i] <= 0) d[i] = 0; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat0
case 5: if (s[i] < -1.0f) d[i] = -1.0f; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat1
case 16: { d[i] = vfpu_rcp(s[i]); } break; //vrcp
case 17: d[i] = USE_VFPU_SQRT ? vfpu_rsqrt(s[i]) : 1.0f / sqrtf(s[i]); break; //vrsq
case 17: d[i] = vfpu_rsqrt(s[i]); break; //vrsq
case 18: { d[i] = vfpu_sin(s[i]); } break; //vsin
case 19: { d[i] = vfpu_cos(s[i]); } break; //vcos
case 20: { d[i] = vfpu_exp2(s[i]); } break; //vexp2
case 21: { d[i] = vfpu_log2(s[i]); } break; //vlog2
case 22: d[i] = USE_VFPU_SQRT ? vfpu_sqrt(s[i]) : fabsf(sqrtf(s[i])); break; //vsqrt
case 22: d[i] = vfpu_sqrt(s[i]); break; //vsqrt
case 23: { d[i] = vfpu_asin(s[i]); } break; //vasin
case 24: { d[i] = -vfpu_rcp(s[i]); } break; // vnrcp
case 26: { d[i] = -vfpu_sin(s[i]); } break; // vnsin
+6 -4
View File
@@ -593,19 +593,21 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) {
break;
case IROp::FRSqrt:
regs_.Map(inst);
FRSQRT_S(regs_.F(inst.dest), regs_.F(inst.src1));
callFuncF_F(&vfpu_rsqrt);
break;
case IROp::FRecip:
regs_.Map(inst);
FRECIP_S(regs_.F(inst.dest), regs_.F(inst.src1));
callFuncF_F(&vfpu_rcp);
break;
case IROp::FAsin:
callFuncF_F(&vfpu_asin);
break;
case IROp::FVSqrt:
callFuncF_F(&vfpu_sqrt);
break;
default:
INVALIDOP;
break;
+6 -18
View File
@@ -608,7 +608,6 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT);
};
RiscVReg tempReg = INVALID_REG;
switch (inst.op) {
case IROp::FSin:
callFuncF_F(&vfpu_sin);
@@ -619,32 +618,21 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) {
break;
case IROp::FRSqrt:
tempReg = regs_.MapWithFPRTemp(inst);
FSQRT(32, regs_.F(inst.dest), regs_.F(inst.src1));
// Ugh, we can't really avoid a temp here. Probably not worth a permanent one.
QuickFLI(32, tempReg, 1.0f, SCRATCH1);
FDIV(32, regs_.F(inst.dest), tempReg, regs_.F(inst.dest));
callFuncF_F(&vfpu_rsqrt);
break;
case IROp::FRecip:
if (inst.dest != inst.src1) {
// This is the easy case.
regs_.Map(inst);
LI(SCRATCH1, 1.0f);
FMV(FMv::W, FMv::X, regs_.F(inst.dest), SCRATCH1);
FDIV(32, regs_.F(inst.dest), regs_.F(inst.dest), regs_.F(inst.src1));
} else {
tempReg = regs_.MapWithFPRTemp(inst);
QuickFLI(32, tempReg, 1.0f, SCRATCH1);
FDIV(32, regs_.F(inst.dest), tempReg, regs_.F(inst.src1));
}
callFuncF_F(&vfpu_rcp);
break;
case IROp::FAsin:
callFuncF_F(&vfpu_asin);
break;
case IROp::FVSqrt:
callFuncF_F(&vfpu_sqrt);
break;
default:
INVALIDOP;
break;
+24 -24
View File
@@ -2289,6 +2289,22 @@ void SinCosNegSin(SinCosArg angle, float *output) {
output[0] = -output[0];
}
void VSqrt(SinCosArg arg, float *output) {
output[0] = vfpu_sqrt(arg);
}
void VRSqrt(SinCosArg arg, float *output) {
output[0] = vfpu_rsqrt(arg);
}
void VRcp(SinCosArg arg, float *output) {
output[0] = vfpu_rcp(arg);
}
void VNRcp(SinCosArg arg, float *output) {
output[0] = -vfpu_rcp(arg);
}
void Exp2(SinCosArg arg, float *output) {
output[0] = vfpu_exp2(arg);
}
@@ -2496,24 +2512,12 @@ void Jit::Comp_VV2Op(MIPSOpcode op) {
MINSS(tempxregs[i], R(XMM0));
break;
case 16: // d[i] = 1.0f / s[i]; break; //vrcp
if (RipAccessible(&one)) {
MOVSS(XMM0, M(&one)); // rip accessible
} else {
MOV(PTRBITS, R(TEMPREG), ImmPtr(&one));
MOVSS(XMM0, MatR(TEMPREG));
}
DIVSS(XMM0, fpr.V(sregs[i]));
MOVSS(tempxregs[i], R(XMM0));
specialFuncCallHelper(&VRcp, sregs[i]);
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
break;
case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq
SQRTSS(XMM0, fpr.V(sregs[i]));
if (RipAccessible(&one)) {
MOVSS(tempxregs[i], M(&one)); // rip accessible
} else {
MOV(PTRBITS, R(TEMPREG), ImmPtr(&one));
MOVSS(tempxregs[i], MatR(TEMPREG));
}
DIVSS(tempxregs[i], R(XMM0));
specialFuncCallHelper(&VRSqrt, sregs[i]);
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
break;
case 18: // d[i] = sinf((float)M_PI_2 * s[i]); break; //vsin
specialFuncCallHelper(&SinOnly, sregs[i]);
@@ -2532,20 +2536,16 @@ void Jit::Comp_VV2Op(MIPSOpcode op) {
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
break;
case 22: // d[i] = sqrtf(s[i]); break; //vsqrt
SQRTSS(tempxregs[i], fpr.V(sregs[i]));
MOV(PTRBITS, R(TEMPREG), ImmPtr(&noSignMask));
ANDPS(tempxregs[i], MatR(TEMPREG));
specialFuncCallHelper(&VSqrt, sregs[i]);
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
break;
case 23: // d[i] = asinf(s[i]) / M_PI_2; break; //vasin
specialFuncCallHelper(&ASinScaled, sregs[i]);
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
break;
case 24: // d[i] = -1.0f / s[i]; break; // vnrcp
// Rare so let's not bother checking for RipAccessible.
MOV(PTRBITS, R(TEMPREG), ImmPtr(&minus_one));
MOVSS(XMM0, MatR(TEMPREG));
DIVSS(XMM0, fpr.V(sregs[i]));
MOVSS(tempxregs[i], R(XMM0));
specialFuncCallHelper(&VNRcp, sregs[i]);
MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0]));
break;
case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin
specialFuncCallHelper(&NegSinOnly, sregs[i]);
+43 -22
View File
@@ -969,6 +969,18 @@ static float X64JIT_XMM_CALL x64_cos(float f) {
static float X64JIT_XMM_CALL x64_asin(float f) {
return vfpu_asin(f);
}
static float X64JIT_XMM_CALL x64_vsqrt(float f) {
return vfpu_sqrt(f);
}
static float X64JIT_XMM_CALL x64_rsqrt(float f) {
return vfpu_rsqrt(f);
}
static float X64JIT_XMM_CALL x64_rcp(float f) {
return vfpu_rcp(f);
}
#else
static uint32_t x64_sin(uint32_t v) {
float f;
@@ -993,6 +1005,30 @@ static uint32_t x64_asin(uint32_t v) {
memcpy(&v, &f, sizeof(v));
return v;
}
static uint32_t x64_vsqrt(uint32_t v) {
float f;
memcpy(&f, &v, sizeof(v));
f = vfpu_sqrt(f);
memcpy(&v, &f, sizeof(v));
return v;
}
static uint32_t x64_rsqrt(uint32_t v) {
float f;
memcpy(&f, &v, sizeof(v));
f = vfpu_rsqrt(f);
memcpy(&v, &f, sizeof(v));
return v;
}
static uint32_t x64_rcp(uint32_t v) {
float f;
memcpy(&f, &v, sizeof(v));
f = vfpu_rcp(f);
memcpy(&v, &f, sizeof(v));
return v;
}
#endif
void X64JitBackend::CompIR_FSpecial(IRInst inst) {
@@ -1047,36 +1083,21 @@ void X64JitBackend::CompIR_FSpecial(IRInst inst) {
break;
case IROp::FRSqrt:
{
X64Reg tempReg = regs_.MapWithFPRTemp(inst);
SQRTSS(tempReg, regs_.F(inst.src1));
MOVSS(regs_.FX(inst.dest), M(constants.positiveOnes)); // rip accessible
DIVSS(regs_.FX(inst.dest), R(tempReg));
break;
}
callFuncF_F((const void *)&x64_rsqrt);
break;
case IROp::FRecip:
if (inst.dest != inst.src1) {
regs_.Map(inst);
MOVSS(regs_.FX(inst.dest), M(constants.positiveOnes)); // rip accessible
DIVSS(regs_.FX(inst.dest), regs_.F(inst.src1));
} else {
X64Reg tempReg = regs_.MapWithFPRTemp(inst);
MOVSS(tempReg, M(constants.positiveOnes)); // rip accessible
if (cpu_info.bAVX) {
VDIVSS(regs_.FX(inst.dest), tempReg, regs_.F(inst.src1));
} else {
DIVSS(tempReg, regs_.F(inst.src1));
MOVSS(regs_.FX(inst.dest), R(tempReg));
}
}
callFuncF_F((const void *)&x64_rcp);
break;
case IROp::FAsin:
callFuncF_F((const void *)&x64_asin);
break;
case IROp::FVSqrt:
callFuncF_F((const void *)&x64_vsqrt);
break;
default:
INVALIDOP;
break;