diff --git a/Core/MIPS/ARM/ArmCompVFPU.cpp b/Core/MIPS/ARM/ArmCompVFPU.cpp index e54ef3e99a..52a7273adc 100644 --- a/Core/MIPS/ARM/ArmCompVFPU.cpp +++ b/Core/MIPS/ARM/ArmCompVFPU.cpp @@ -19,6 +19,7 @@ #if PPSSPP_ARCH(ARM) #include +#include #include "Common/CPUDetect.h" #include "Common/Data/Convert/SmallDataConvert.h" #include "Common/Math/math_util.h" @@ -911,6 +912,12 @@ namespace MIPSComp case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2 DISABLE; break; + case 16: // vrcp + case 17: // vrsq + case 22: // vsqrt + case 24: // vnrcp + CompVV2OpCall(op); + return; case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin DISABLE; break; @@ -1007,23 +1014,6 @@ namespace MIPSComp VMOV(fpr.V(tempregs[i]), S1); SetCC(CC_AL); break; - case 16: // d[i] = 1.0f / s[i]; break; //vrcp - if (i == 0) { - MOVI2F(S0, 1.0f, SCRATCHREG1); - } - VDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i])); - break; - case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq - if (i == 0) { - MOVI2F(S0, 1.0f, SCRATCHREG1); - } - VSQRT(S1, fpr.V(sregs[i])); - VDIV(fpr.V(tempregs[i]), S0, S1); - break; - case 22: // d[i] = sqrtf(s[i]); break; //vsqrt - VSQRT(fpr.V(tempregs[i]), fpr.V(sregs[i])); - VABS(fpr.V(tempregs[i]), fpr.V(tempregs[i])); - break; case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin // Seems to work well enough but can disable if it becomes a problem. // Should be easy enough to translate to NEON. There we can load all the constants @@ -1052,12 +1042,6 @@ namespace MIPSComp MOVI2F(S1, 1.0f / (M_PI / 2), SCRATCHREG1); VMUL(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S1); break; - case 24: // d[i] = -1.0f / s[i]; break; // vnrcp - if (i == 0) { - MOVI2F(S0, -1.0f, SCRATCHREG1); - } - VDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i])); - break; default: ERROR_LOG(Log::JIT, "case missing in vfpu vv2op"); DISABLE; @@ -2120,6 +2104,75 @@ namespace MIPSComp fpr.ReleaseSpillLocksAndDiscardTemps(); } + // Float bits in and out, so the calls look the same with softfp and hardfp. + static u32 CallWithBits(float (*func)(float), u32 x) { + float f; + memcpy(&f, &x, sizeof(f)); + f = func(f); + memcpy(&x, &f, sizeof(x)); + return x; + } + + static u32 VRcpBits(u32 x) { + return CallWithBits(&vfpu_rcp, x); + } + + static u32 VNRcpBits(u32 x) { + return CallWithBits(&vfpu_rcp, x) ^ 0x80000000; + } + + static u32 VRSqrtBits(u32 x) { + return CallWithBits(&vfpu_rsqrt, x); + } + + static u32 VSqrtBits(u32 x) { + return CallWithBits(&vfpu_sqrt, x); + } + + // vrcp, vrsq, vsqrt and vnrcp call the exact functions. The lanes stay in S16-S19 across the + // calls (callee-saved), and the results are stored to the destinations' homes. + void ArmJit::CompVV2OpCall(MIPSOpcode op) { + // Like the interpreter, these apply the prefixes to the last lane only. + if (js.HasSPrefix() || (js.HasDPrefix() && GetVecSize(op) != V_Single)) { + DISABLE; + } + + const int optype = (op >> 16) & 0x1f; + u32 (*func)(u32) = &VRcpBits; + if (optype == 17) { + func = &VRSqrtBits; + } else if (optype == 22) { + func = &VSqrtBits; + } else if (optype == 24) { + func = &VNRcpBits; + } + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + u8 sregs[4], dregs[4]; + GetVectorRegs(sregs, sz, _VS); + GetVectorRegs(dregs, sz, _VD); + + gpr.FlushBeforeCall(); + fpr.FlushAll(); + + for (int i = 0; i < n; i++) { + VLDR((ARMReg)(S16 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i])); + } + for (int i = 0; i < n; i++) { + VMOV(R0, (ARMReg)(S16 + i)); + // FlushBeforeCall saves R1. + QuickCallFunction(R1, func); + VMOV((ARMReg)(S16 + i), R0); + } + for (int i = 0; i < n; i++) { + VSTR((ARMReg)(S16 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i])); + } + + ApplyPrefixD(dregs, sz); + fpr.ReleaseSpillLocksAndDiscardTemps(); + } + static double SinCos(float angle) { union { struct { float sin; float cos; }; double out; } sincos; vfpu_sincos(angle, sincos.sin, sincos.cos); diff --git a/Core/MIPS/ARM/ArmJit.h b/Core/MIPS/ARM/ArmJit.h index cc7ef2907d..f01b22922a 100644 --- a/Core/MIPS/ARM/ArmJit.h +++ b/Core/MIPS/ARM/ArmJit.h @@ -196,6 +196,7 @@ private: void CompShiftImm(MIPSOpcode op, ArmGen::ShiftType shiftType, int sa); void CompShiftVar(MIPSOpcode op, ArmGen::ShiftType shiftType); void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin); + void CompVV2OpCall(MIPSOpcode op); void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz); void ApplyPrefixD(const u8 *vregs, VectorSize sz); diff --git a/Core/MIPS/ARM/ArmRegCacheFPU.h b/Core/MIPS/ARM/ArmRegCacheFPU.h index 5551833aa4..8fb2492cd5 100644 --- a/Core/MIPS/ARM/ArmRegCacheFPU.h +++ b/Core/MIPS/ARM/ArmRegCacheFPU.h @@ -166,6 +166,9 @@ public: void SetEmitter(ArmGen::ARMXEmitter *emitter) { emit_ = emitter; } int GetMipsRegOffset(MIPSReg r); + int GetMipsRegOffsetV(MIPSReg r) { + return GetMipsRegOffset(r + 32); + } private: bool Consecutive(int v1, int v2) const; @@ -174,9 +177,6 @@ private: MIPSReg GetTempR(); const ArmGen::ARMReg *GetMIPSAllocationOrder(int &count); - int GetMipsRegOffsetV(MIPSReg r) { - return GetMipsRegOffset(r + 32); - } // This one WILL get a free quad as long as you haven't spill-locked them all. int QGetFreeQuad(int start, int count, const char *reason); diff --git a/Core/MIPS/ARM64/Arm64CompVFPU.cpp b/Core/MIPS/ARM64/Arm64CompVFPU.cpp index 2afed15f90..f5471518c4 100644 --- a/Core/MIPS/ARM64/Arm64CompVFPU.cpp +++ b/Core/MIPS/ARM64/Arm64CompVFPU.cpp @@ -861,6 +861,12 @@ namespace MIPSComp { case 21: // d[i] = logf(s[i])/log(2.0f); break; //vlog2 DISABLE; break; + case 16: // vrcp + case 17: // vrsq + case 22: // vsqrt + case 24: // vnrcp + CompVV2OpCall(op); + return; case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin DISABLE; break; @@ -930,32 +936,9 @@ namespace MIPSComp { fp.FMAX(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S0); fp.FMIN(fpr.V(tempregs[i]), fpr.V(tempregs[i]), S1); break; - case 16: // d[i] = 1.0f / s[i]; break; //vrcp - if (i == 0) { - fp.MOVI2F(S0, 1.0f, SCRATCH1); - } - fp.FDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i])); - break; - case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq - if (i == 0) { - fp.MOVI2F(S0, 1.0f, SCRATCH1); - } - fp.FSQRT(S1, fpr.V(sregs[i])); - fp.FDIV(fpr.V(tempregs[i]), S0, S1); - break; - case 22: // d[i] = sqrtf(s[i]); break; //vsqrt - fp.FSQRT(fpr.V(tempregs[i]), fpr.V(sregs[i])); - fp.FABS(fpr.V(tempregs[i]), fpr.V(tempregs[i])); - break; case 23: // d[i] = asinf(s[i] * (float)M_2_PI); break; //vasin DISABLE; break; - case 24: // d[i] = -1.0f / s[i]; break; // vnrcp - if (i == 0) { - fp.MOVI2F(S0, -1.0f, SCRATCH1); - } - fp.FDIV(fpr.V(tempregs[i]), S0, fpr.V(sregs[i])); - break; default: ERROR_LOG(Log::JIT, "case missing in vfpu vv2op"); DISABLE; @@ -975,6 +958,45 @@ namespace MIPSComp { fpr.ReleaseSpillLocksAndDiscardTemps(); } + // vrcp, vrsq, vsqrt and vnrcp call the exact functions. The lanes stay in S8-S11 across the + // calls (callee-saved), and the results are stored to the destinations' homes. + void Arm64Jit::CompVV2OpCall(MIPSOpcode op) { + if (js.HasSPrefix()) { + DISABLE; + } + + const int optype = (op >> 16) & 0x1f; + float (*func)(float) = optype == 17 ? &vfpu_rsqrt : (optype == 22 ? &vfpu_sqrt : &vfpu_rcp); + + VectorSize sz = GetVecSize(op); + int n = GetNumVectorElements(sz); + u8 sregs[4], dregs[4]; + GetVectorRegs(sregs, sz, _VS); + GetVectorRegs(dregs, sz, _VD); + + gpr.FlushBeforeCall(); + fpr.FlushAll(); + + for (int i = 0; i < n; i++) { + fp.LDR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(sregs[i])); + } + for (int i = 0; i < n; i++) { + fp.FMOV(S0, (ARM64Reg)(S8 + i)); + QuickCallFunction(SCRATCH2_64, func); + if (optype == 24) { + fp.FNEG((ARM64Reg)(S8 + i), S0); + } else { + fp.FMOV((ARM64Reg)(S8 + i), S0); + } + } + for (int i = 0; i < n; i++) { + fp.STR(32, INDEX_UNSIGNED, (ARM64Reg)(S8 + i), CTXREG, fpr.GetMipsRegOffsetV(dregs[i])); + } + + ApplyPrefixD(dregs, sz); + fpr.ReleaseSpillLocksAndDiscardTemps(); + } + void Arm64Jit::Comp_Vi2f(MIPSOpcode op) { CONDITIONAL_DISABLE(VFPU_VEC); if (js.HasUnknownPrefix()) { diff --git a/Core/MIPS/ARM64/Arm64IRCompFPU.cpp b/Core/MIPS/ARM64/Arm64IRCompFPU.cpp index 01fdeed9e7..f806a86659 100644 --- a/Core/MIPS/ARM64/Arm64IRCompFPU.cpp +++ b/Core/MIPS/ARM64/Arm64IRCompFPU.cpp @@ -548,22 +548,21 @@ void Arm64JitBackend::CompIR_FSpecial(IRInst inst) { break; case IROp::FRSqrt: - regs_.Map(inst); - fp_.MOVI2F(SCRATCHF1, 1.0f); - fp_.FSQRT(regs_.F(inst.dest), regs_.F(inst.src1)); - fp_.FDIV(regs_.F(inst.dest), SCRATCHF1, regs_.F(inst.dest)); + callFuncF_F(&vfpu_rsqrt); break; case IROp::FRecip: - regs_.Map(inst); - fp_.MOVI2F(SCRATCHF1, 1.0f); - fp_.FDIV(regs_.F(inst.dest), SCRATCHF1, regs_.F(inst.src1)); + callFuncF_F(&vfpu_rcp); break; case IROp::FAsin: callFuncF_F(&vfpu_asin); break; + case IROp::FVSqrt: + callFuncF_F(&vfpu_sqrt); + break; + default: INVALIDOP; break; diff --git a/Core/MIPS/ARM64/Arm64Jit.h b/Core/MIPS/ARM64/Arm64Jit.h index f0d99ac904..9f87a948b6 100644 --- a/Core/MIPS/ARM64/Arm64Jit.h +++ b/Core/MIPS/ARM64/Arm64Jit.h @@ -236,6 +236,7 @@ private: void CompShiftImm(MIPSOpcode op, Arm64Gen::ShiftType shiftType, int sa); void CompShiftVar(MIPSOpcode op, Arm64Gen::ShiftType shiftType); void CompVrotShuffle(u8 *dregs, int imm, VectorSize sz, bool negSin); + void CompVV2OpCall(MIPSOpcode op); void ApplyPrefixST(u8 *vregs, u32 prefix, VectorSize sz); void ApplyPrefixD(const u8 *vregs, VectorSize sz); diff --git a/Core/MIPS/IR/IRCompVFPU.cpp b/Core/MIPS/IR/IRCompVFPU.cpp index 0c216b9e2a..8c9587ce78 100644 --- a/Core/MIPS/IR/IRCompVFPU.cpp +++ b/Core/MIPS/IR/IRCompVFPU.cpp @@ -1128,7 +1128,7 @@ namespace MIPSComp { DISABLE; break; case 22: // d[i] = sqrtf(s[i]); break; //vsqrt - ir.Write(IROp::FSqrt, tempregs[i], sregs[i]); + ir.Write(IROp::FVSqrt, tempregs[i], sregs[i]); break; case 23: // d[i] = asinf(s[i]) / M_PI_2; break; //vasin ir.Write(IROp::FAsin, tempregs[i], sregs[i]); diff --git a/Core/MIPS/IR/IRInst.cpp b/Core/MIPS/IR/IRInst.cpp index 53a00ef8c9..be9db73e7d 100644 --- a/Core/MIPS/IR/IRInst.cpp +++ b/Core/MIPS/IR/IRInst.cpp @@ -109,6 +109,7 @@ static const IRMeta irMeta[] = { { IROp::FRSqrt, "FRSqrt", "FF" }, { IROp::FRecip, "FRecip", "FF" }, { IROp::FAsin, "FAsin", "FF" }, + { IROp::FVSqrt, "FVSqrt", "FF" }, { IROp::FNeg, "FNeg", "FF" }, { IROp::FSign, "FSign", "FF" }, { IROp::FAbs, "FAbs", "FF" }, diff --git a/Core/MIPS/IR/IRInst.h b/Core/MIPS/IR/IRInst.h index 313f7aa525..0e41ca3894 100644 --- a/Core/MIPS/IR/IRInst.h +++ b/Core/MIPS/IR/IRInst.h @@ -197,9 +197,10 @@ enum class IROp : uint8_t { // Slow special functions. Used on singles. FSin, FCos, - FRSqrt, - FRecip, + FRSqrt, // vrsq, bit-exact with the PSP + FRecip, // vrcp, bit-exact with the PSP FAsin, + FVSqrt, // vsqrt, bit-exact with the PSP (FSqrt is the FPU's IEEE sqrt.s) // Fake/System instructions Interpret, diff --git a/Core/MIPS/IR/IRInterpreter.cpp b/Core/MIPS/IR/IRInterpreter.cpp index 89cf7c1a23..76e3e8e29b 100644 --- a/Core/MIPS/IR/IRInterpreter.cpp +++ b/Core/MIPS/IR/IRInterpreter.cpp @@ -629,14 +629,17 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) { mips->f[inst->dest] = vfpu_cos(mips->f[inst->src1]); break; case IROp::FRSqrt: - mips->f[inst->dest] = 1.0f / sqrtf(mips->f[inst->src1]); + mips->f[inst->dest] = vfpu_rsqrt(mips->f[inst->src1]); break; case IROp::FRecip: - mips->f[inst->dest] = 1.0f / mips->f[inst->src1]; + mips->f[inst->dest] = vfpu_rcp(mips->f[inst->src1]); break; case IROp::FAsin: mips->f[inst->dest] = vfpu_asin(mips->f[inst->src1]); break; + case IROp::FVSqrt: + mips->f[inst->dest] = vfpu_sqrt(mips->f[inst->src1]); + break; case IROp::ShlImm: mips->r[inst->dest] = mips->r[inst->src1] << (int)inst->src2; diff --git a/Core/MIPS/IR/IRNativeCommon.cpp b/Core/MIPS/IR/IRNativeCommon.cpp index 20641a1c91..4201fe5f41 100644 --- a/Core/MIPS/IR/IRNativeCommon.cpp +++ b/Core/MIPS/IR/IRNativeCommon.cpp @@ -432,6 +432,7 @@ void IRNativeBackend::CompileIRInst(IRInst inst) { case IROp::FRSqrt: case IROp::FRecip: case IROp::FAsin: + case IROp::FVSqrt: CompIR_FSpecial(inst); break; diff --git a/Core/MIPS/IR/IRPassSimplify.cpp b/Core/MIPS/IR/IRPassSimplify.cpp index 1ea6fef644..c8b603dc90 100644 --- a/Core/MIPS/IR/IRPassSimplify.cpp +++ b/Core/MIPS/IR/IRPassSimplify.cpp @@ -765,6 +765,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts case IROp::FRSqrt: case IROp::FRecip: case IROp::FAsin: + case IROp::FVSqrt: out.Write(inst); break; diff --git a/Core/MIPS/InterpreterVFPU.cpp b/Core/MIPS/InterpreterVFPU.cpp index d944c7c256..40787b3c65 100644 --- a/Core/MIPS/InterpreterVFPU.cpp +++ b/Core/MIPS/InterpreterVFPU.cpp @@ -75,7 +75,6 @@ #endif static const bool USE_VFPU_DOT = false; -static const bool USE_VFPU_SQRT = false; union FloatBits { float f[4]; @@ -665,13 +664,13 @@ namespace MIPSInt case 4: if (s[i] <= 0) d[i] = 0; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat0 case 5: if (s[i] < -1.0f) d[i] = -1.0f; else {if(s[i] > 1.0f) d[i] = 1.0f; else d[i] = s[i];} break; // vsat1 case 16: { d[i] = vfpu_rcp(s[i]); } break; //vrcp - case 17: d[i] = USE_VFPU_SQRT ? vfpu_rsqrt(s[i]) : 1.0f / sqrtf(s[i]); break; //vrsq + case 17: d[i] = vfpu_rsqrt(s[i]); break; //vrsq case 18: { d[i] = vfpu_sin(s[i]); } break; //vsin case 19: { d[i] = vfpu_cos(s[i]); } break; //vcos case 20: { d[i] = vfpu_exp2(s[i]); } break; //vexp2 case 21: { d[i] = vfpu_log2(s[i]); } break; //vlog2 - case 22: d[i] = USE_VFPU_SQRT ? vfpu_sqrt(s[i]) : fabsf(sqrtf(s[i])); break; //vsqrt + case 22: d[i] = vfpu_sqrt(s[i]); break; //vsqrt case 23: { d[i] = vfpu_asin(s[i]); } break; //vasin case 24: { d[i] = -vfpu_rcp(s[i]); } break; // vnrcp case 26: { d[i] = -vfpu_sin(s[i]); } break; // vnsin diff --git a/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp b/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp index 7839072816..33fcb0279a 100644 --- a/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp +++ b/Core/MIPS/LoongArch64/LoongArch64CompFPU.cpp @@ -593,19 +593,21 @@ void LoongArch64JitBackend::CompIR_FSpecial(IRInst inst) { break; case IROp::FRSqrt: - regs_.Map(inst); - FRSQRT_S(regs_.F(inst.dest), regs_.F(inst.src1)); + callFuncF_F(&vfpu_rsqrt); break; case IROp::FRecip: - regs_.Map(inst); - FRECIP_S(regs_.F(inst.dest), regs_.F(inst.src1)); + callFuncF_F(&vfpu_rcp); break; case IROp::FAsin: callFuncF_F(&vfpu_asin); break; + case IROp::FVSqrt: + callFuncF_F(&vfpu_sqrt); + break; + default: INVALIDOP; break; diff --git a/Core/MIPS/RiscV/RiscVCompFPU.cpp b/Core/MIPS/RiscV/RiscVCompFPU.cpp index f4a967df5d..0473d73c7d 100644 --- a/Core/MIPS/RiscV/RiscVCompFPU.cpp +++ b/Core/MIPS/RiscV/RiscVCompFPU.cpp @@ -608,7 +608,6 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) { WriteDebugProfilerStatus(IRProfilerStatus::IN_JIT); }; - RiscVReg tempReg = INVALID_REG; switch (inst.op) { case IROp::FSin: callFuncF_F(&vfpu_sin); @@ -619,32 +618,21 @@ void RiscVJitBackend::CompIR_FSpecial(IRInst inst) { break; case IROp::FRSqrt: - tempReg = regs_.MapWithFPRTemp(inst); - FSQRT(32, regs_.F(inst.dest), regs_.F(inst.src1)); - - // Ugh, we can't really avoid a temp here. Probably not worth a permanent one. - QuickFLI(32, tempReg, 1.0f, SCRATCH1); - FDIV(32, regs_.F(inst.dest), tempReg, regs_.F(inst.dest)); + callFuncF_F(&vfpu_rsqrt); break; case IROp::FRecip: - if (inst.dest != inst.src1) { - // This is the easy case. - regs_.Map(inst); - LI(SCRATCH1, 1.0f); - FMV(FMv::W, FMv::X, regs_.F(inst.dest), SCRATCH1); - FDIV(32, regs_.F(inst.dest), regs_.F(inst.dest), regs_.F(inst.src1)); - } else { - tempReg = regs_.MapWithFPRTemp(inst); - QuickFLI(32, tempReg, 1.0f, SCRATCH1); - FDIV(32, regs_.F(inst.dest), tempReg, regs_.F(inst.src1)); - } + callFuncF_F(&vfpu_rcp); break; case IROp::FAsin: callFuncF_F(&vfpu_asin); break; + case IROp::FVSqrt: + callFuncF_F(&vfpu_sqrt); + break; + default: INVALIDOP; break; diff --git a/Core/MIPS/x86/CompVFPU.cpp b/Core/MIPS/x86/CompVFPU.cpp index b20d6e3d5d..a0e7ae443b 100644 --- a/Core/MIPS/x86/CompVFPU.cpp +++ b/Core/MIPS/x86/CompVFPU.cpp @@ -2289,6 +2289,22 @@ void SinCosNegSin(SinCosArg angle, float *output) { output[0] = -output[0]; } +void VSqrt(SinCosArg arg, float *output) { + output[0] = vfpu_sqrt(arg); +} + +void VRSqrt(SinCosArg arg, float *output) { + output[0] = vfpu_rsqrt(arg); +} + +void VRcp(SinCosArg arg, float *output) { + output[0] = vfpu_rcp(arg); +} + +void VNRcp(SinCosArg arg, float *output) { + output[0] = -vfpu_rcp(arg); +} + void Exp2(SinCosArg arg, float *output) { output[0] = vfpu_exp2(arg); } @@ -2496,24 +2512,12 @@ void Jit::Comp_VV2Op(MIPSOpcode op) { MINSS(tempxregs[i], R(XMM0)); break; case 16: // d[i] = 1.0f / s[i]; break; //vrcp - if (RipAccessible(&one)) { - MOVSS(XMM0, M(&one)); // rip accessible - } else { - MOV(PTRBITS, R(TEMPREG), ImmPtr(&one)); - MOVSS(XMM0, MatR(TEMPREG)); - } - DIVSS(XMM0, fpr.V(sregs[i])); - MOVSS(tempxregs[i], R(XMM0)); + specialFuncCallHelper(&VRcp, sregs[i]); + MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0])); break; case 17: // d[i] = 1.0f / sqrtf(s[i]); break; //vrsq - SQRTSS(XMM0, fpr.V(sregs[i])); - if (RipAccessible(&one)) { - MOVSS(tempxregs[i], M(&one)); // rip accessible - } else { - MOV(PTRBITS, R(TEMPREG), ImmPtr(&one)); - MOVSS(tempxregs[i], MatR(TEMPREG)); - } - DIVSS(tempxregs[i], R(XMM0)); + specialFuncCallHelper(&VRSqrt, sregs[i]); + MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0])); break; case 18: // d[i] = sinf((float)M_PI_2 * s[i]); break; //vsin specialFuncCallHelper(&SinOnly, sregs[i]); @@ -2532,20 +2536,16 @@ void Jit::Comp_VV2Op(MIPSOpcode op) { MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0])); break; case 22: // d[i] = sqrtf(s[i]); break; //vsqrt - SQRTSS(tempxregs[i], fpr.V(sregs[i])); - MOV(PTRBITS, R(TEMPREG), ImmPtr(&noSignMask)); - ANDPS(tempxregs[i], MatR(TEMPREG)); + specialFuncCallHelper(&VSqrt, sregs[i]); + MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0])); break; case 23: // d[i] = asinf(s[i]) / M_PI_2; break; //vasin specialFuncCallHelper(&ASinScaled, sregs[i]); MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0])); break; case 24: // d[i] = -1.0f / s[i]; break; // vnrcp - // Rare so let's not bother checking for RipAccessible. - MOV(PTRBITS, R(TEMPREG), ImmPtr(&minus_one)); - MOVSS(XMM0, MatR(TEMPREG)); - DIVSS(XMM0, fpr.V(sregs[i])); - MOVSS(tempxregs[i], R(XMM0)); + specialFuncCallHelper(&VNRcp, sregs[i]); + MOVSS(tempxregs[i], MIPSSTATE_VAR(sincostemp[0])); break; case 26: // d[i] = -sinf((float)M_PI_2 * s[i]); break; // vnsin specialFuncCallHelper(&NegSinOnly, sregs[i]); diff --git a/Core/MIPS/x86/X64IRCompFPU.cpp b/Core/MIPS/x86/X64IRCompFPU.cpp index 71b1c734dd..221565b248 100644 --- a/Core/MIPS/x86/X64IRCompFPU.cpp +++ b/Core/MIPS/x86/X64IRCompFPU.cpp @@ -969,6 +969,18 @@ static float X64JIT_XMM_CALL x64_cos(float f) { static float X64JIT_XMM_CALL x64_asin(float f) { return vfpu_asin(f); } + +static float X64JIT_XMM_CALL x64_vsqrt(float f) { + return vfpu_sqrt(f); +} + +static float X64JIT_XMM_CALL x64_rsqrt(float f) { + return vfpu_rsqrt(f); +} + +static float X64JIT_XMM_CALL x64_rcp(float f) { + return vfpu_rcp(f); +} #else static uint32_t x64_sin(uint32_t v) { float f; @@ -993,6 +1005,30 @@ static uint32_t x64_asin(uint32_t v) { memcpy(&v, &f, sizeof(v)); return v; } + +static uint32_t x64_vsqrt(uint32_t v) { + float f; + memcpy(&f, &v, sizeof(v)); + f = vfpu_sqrt(f); + memcpy(&v, &f, sizeof(v)); + return v; +} + +static uint32_t x64_rsqrt(uint32_t v) { + float f; + memcpy(&f, &v, sizeof(v)); + f = vfpu_rsqrt(f); + memcpy(&v, &f, sizeof(v)); + return v; +} + +static uint32_t x64_rcp(uint32_t v) { + float f; + memcpy(&f, &v, sizeof(v)); + f = vfpu_rcp(f); + memcpy(&v, &f, sizeof(v)); + return v; +} #endif void X64JitBackend::CompIR_FSpecial(IRInst inst) { @@ -1047,36 +1083,21 @@ void X64JitBackend::CompIR_FSpecial(IRInst inst) { break; case IROp::FRSqrt: - { - X64Reg tempReg = regs_.MapWithFPRTemp(inst); - SQRTSS(tempReg, regs_.F(inst.src1)); - - MOVSS(regs_.FX(inst.dest), M(constants.positiveOnes)); // rip accessible - DIVSS(regs_.FX(inst.dest), R(tempReg)); - break; - } + callFuncF_F((const void *)&x64_rsqrt); + break; case IROp::FRecip: - if (inst.dest != inst.src1) { - regs_.Map(inst); - MOVSS(regs_.FX(inst.dest), M(constants.positiveOnes)); // rip accessible - DIVSS(regs_.FX(inst.dest), regs_.F(inst.src1)); - } else { - X64Reg tempReg = regs_.MapWithFPRTemp(inst); - MOVSS(tempReg, M(constants.positiveOnes)); // rip accessible - if (cpu_info.bAVX) { - VDIVSS(regs_.FX(inst.dest), tempReg, regs_.F(inst.src1)); - } else { - DIVSS(tempReg, regs_.F(inst.src1)); - MOVSS(regs_.FX(inst.dest), R(tempReg)); - } - } + callFuncF_F((const void *)&x64_rcp); break; case IROp::FAsin: callFuncF_F((const void *)&x64_asin); break; + case IROp::FVSqrt: + callFuncF_F((const void *)&x64_vsqrt); + break; + default: INVALIDOP; break;