diff --git a/Core/MIPS/IR/IRCompVFPU.cpp b/Core/MIPS/IR/IRCompVFPU.cpp index e5bf32f4e6..8d01ec3dcc 100644 --- a/Core/MIPS/IR/IRCompVFPU.cpp +++ b/Core/MIPS/IR/IRCompVFPU.cpp @@ -971,7 +971,7 @@ namespace MIPSComp { } if (type == VecDo3Op::VSGE || type == VecDo3Op::VSLT) { - ir.Write(IROp::FpCondFromReg, IRTEMP_0); + ir.Write(IROp::FpCondFromReg, 0, IRTEMP_0); } for (int i = 0; i < n; i++) { diff --git a/Core/MIPS/IR/IRInterpreter.cpp b/Core/MIPS/IR/IRInterpreter.cpp index 7a6a764951..a3bb3bcd82 100644 --- a/Core/MIPS/IR/IRInterpreter.cpp +++ b/Core/MIPS/IR/IRInterpreter.cpp @@ -486,8 +486,9 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) { #if PPSSPP_ARCH(SSE2) __m128i src = _mm_loadu_si128((__m128i *) & mips->fi[inst->src1]); - // Shift each 32-bit lane right by 24 bits. Then left by 1. This matches the rather weird behavior. - src = _mm_slli_epi32(_mm_srli_epi32(src, 24), 1); + // Shift each 32-bit lane left by 1, then take the top byte - that is, (v >> 23) & 0xFF. + // Shifting right by 24 first would drop bit 23. + src = _mm_srli_epi32(_mm_slli_epi32(src, 1), 24); // Pack 32-bit lanes to 16-bit, then 16-bit to 8-bit // This moves our target bytes to the bottom of the XMM register src = _mm_packs_epi32(src, src); @@ -905,7 +906,8 @@ u32 IRInterpret(MIPSState *mips, const IRInst *inst) { } case IROp::FpCondFromReg: - mips->fpcond = mips->r[inst->dest]; + // Note: the register is in src1, see the "_G" meta - the native backends read it there. + mips->fpcond = mips->r[inst->src1]; break; case IROp::FpCondToReg: mips->r[inst->dest] = mips->fpcond; diff --git a/Core/MIPS/IR/IRPassSimplify.cpp b/Core/MIPS/IR/IRPassSimplify.cpp index 373093904d..1ea6fef644 100644 --- a/Core/MIPS/IR/IRPassSimplify.cpp +++ b/Core/MIPS/IR/IRPassSimplify.cpp @@ -23,10 +23,11 @@ u32 Evaluate(u32 a, u32 b, IROp op) { case IROp::And: case IROp::AndConst: return a & b; case IROp::Or: case IROp::OrConst: return a | b; case IROp::Xor: case IROp::XorConst: return a ^ b; - case IROp::Shr: case IROp::ShrImm: return a >> b; - case IROp::Sar: case IROp::SarImm: return (s32)a >> b; - case IROp::Ror: case IROp::RorImm: return (a >> b) | (a << (32 - b)); - case IROp::Shl: case IROp::ShlImm: return a << b; + // MIPS (and IRInterpret) only look at the low 5 bits of a variable shift amount. + case IROp::Shr: case IROp::ShrImm: return a >> (b & 31); + case IROp::Sar: case IROp::SarImm: return (s32)a >> (b & 31); + case IROp::Ror: case IROp::RorImm: return __rotr(a, b & 31); + case IROp::Shl: case IROp::ShlImm: return a << (b & 31); case IROp::Slt: case IROp::SltConst: return ((s32)a < (s32)b); case IROp::SltU: case IROp::SltUConst: return (a < b); default: @@ -1886,7 +1887,11 @@ bool ApplyMemoryValidation(const IRWriter &in, IRWriter &out, const IROptions &o } const IRMeta *m = GetIRMeta(inst.op); - if (m->types[0] == 'G' && (m->flags & IRFLAG_SRC3) == 0) { + if ((m->flags & IRFLAG_BARRIER) != 0) { + // Interpret runs an arbitrary instruction and CallReplacement a whole function, so + // either can write any GPR - no address we validated earlier can be trusted after one. + checks.clear(); + } else if (m->types[0] == 'G' && (m->flags & IRFLAG_SRC3) == 0) { uint64_t key = (uint64_t)inst.dest << 32; // Wipe out all the already done checks since this was modified. checks.erase(checks.lower_bound(key), checks.upper_bound(key | 0xFFFFFFFFULL)); @@ -2148,7 +2153,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) { out.Write(inst.op, inst.dest, inst.src1, temp, inst.constant); skip = true; inst.src2 = IRREG_INVALID; - } else if (isVec4[inst.src2 & 3] && usedLaterAsVec4(inst.src2 & ~3) && findAvailTempVec4()) { + } else if (isVec4[inst.src2 & ~3] && usedLaterAsVec4(inst.src2 & ~3) && findAvailTempVec4()) { out.Write(IROp::FMov, temp, inst.src2); out.Write(inst.op, inst.dest, inst.src1, temp, inst.constant); skip = true;