IRWriter: Remove the confusing and inefficient AddConstant

This commit is contained in:
Henrik Rydgård committed 2026-08-15 18:31:20 +02:00
1 parent 2096179dce
commit ae14ebb6ac
12 files changed
+178 -175

No files matched your search

+2 -2
View File
@@ -98,8 +98,8 @@ struct LogChannel {
#endif
bool enabled = true;
bool IsEnabled(LogLevel level) const {
if (level > this->level || !this->enabled)
bool IsEnabled(LogLevel logLevel) const {
if (logLevel > level || !enabled)
return false;
return true;
}
+6 -6
View File
@@ -197,17 +197,17 @@ inline __m128i _mm_mullo_epi32_SSE2(const __m128i v0, const __m128i v1) {
inline __m128i _mm_max_epu16_SSE2(const __m128i v0, const __m128i v1) {
return _mm_xor_si128(
_mm_max_epi16(
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
_mm_set1_epi16((int16_t)0x8000));
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
}
inline __m128i _mm_min_epu16_SSE2(const __m128i v0, const __m128i v1) {
return _mm_xor_si128(
_mm_min_epi16(
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
_mm_set1_epi16((int16_t)0x8000));
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
}
// SSE2 replacement for half of a _mm_packus_epi32 but without the saturation.
+1 -1
View File
@@ -34,7 +34,7 @@ inline constexpr uint32_t RoundDownToMultipleOf(uint32_t v, uint32_t multiple) {
// TODO: this should just use a bitscan.
inline uint32_t log2i(uint32_t val) {
unsigned int ret = -1;
unsigned int ret = (unsigned int)-1;
while (val != 0) {
val >>= 1; ret++;
}
+11 -11
View File
@@ -61,19 +61,19 @@ void IRFrontend::Comp_IType(MIPSOpcode op) {
switch (op >> 26) {
case 8: // same as addiu?
case 9: // R(rt) = R(rs) + simm; break; //addiu
ir.Write(IROp::AddConst, rt, rs, ir.AddConstant(simm));
ir.Write(IROp::AddConst, rt, rs, 0, (u32)simm);
break;
case 12: ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(uimm)); break;
case 13: ir.Write(IROp::OrConst, rt, rs, ir.AddConstant(uimm)); break;
case 14: ir.Write(IROp::XorConst, rt, rs, ir.AddConstant(uimm)); break;
case 12: ir.Write(IROp::AndConst, rt, rs, 0, uimm); break;
case 13: ir.Write(IROp::OrConst, rt, rs, 0, uimm); break;
case 14: ir.Write(IROp::XorConst, rt, rs, 0, uimm); break;
case 10: // R(rt) = (s32)R(rs) < simm; break; //slti
ir.Write(IROp::SltConst, rt, rs, ir.AddConstant(simm));
ir.Write(IROp::SltConst, rt, rs, 0, (u32)simm);
break;
case 11: // R(rt) = R(rs) < suimm; break; //sltiu
ir.Write(IROp::SltUConst, rt, rs, ir.AddConstant(suimm));
ir.Write(IROp::SltUConst, rt, rs, 0, suimm);
break;
case 15: // R(rt) = uimm << 16; //lui
@@ -197,7 +197,7 @@ void IRFrontend::CompShiftVar(MIPSOpcode op, IROp shiftOp) {
// The interpreter already masks where needed, don't need to generate extra ops.
ir.Write(shiftOp, rd, rt, rs);
} else {
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(31));
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, (u32)31);
ir.Write(shiftOp, rd, rt, IRTEMP_0);
}
}
@@ -244,9 +244,9 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
case 0x0: // ext
if (pos != 0) {
ir.Write(IROp::ShrImm, rt, rs, pos);
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(mask));
ir.Write(IROp::AndConst, rt, rt, 0, mask);
} else {
ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(mask));
ir.Write(IROp::AndConst, rt, rs, 0, mask);
}
break;
@@ -259,7 +259,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
if (size != 32) {
// Need to use the sourcemask.
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(sourcemask));
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, sourcemask);
if (pos != 0) {
ir.Write(IROp::ShlImm, IRTEMP_0, IRTEMP_0, pos);
}
@@ -271,7 +271,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
ir.Write(IROp::Mov, IRTEMP_0, rs);
}
}
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(destmask));
ir.Write(IROp::AndConst, rt, rt, 0, destmask);
ir.Write(IROp::Or, rt, rt, IRTEMP_0);
}
break;
+20 -20
View File
@@ -96,11 +96,11 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs, rhs);
ir.Write(ComparisonToExit(cc), 0, lhs, rhs, ResolveNotTakenTarget(branchInfo));
// This makes the block "impure" :(
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
@@ -114,7 +114,7 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
}
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -147,11 +147,11 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs);
ir.Write(ComparisonToExit(cc), 0, lhs, 0, ResolveNotTakenTarget(branchInfo));
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
if (branchInfo.delaySlotIsBranch) {
@@ -165,7 +165,7 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
// Taken
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -225,12 +225,12 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
// Not taken
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
// Taken
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
@@ -244,7 +244,7 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
}
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -284,14 +284,14 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
CompileDelaySlot();
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
int imm3 = (op >> 18) & 7;
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, ir.AddConstant(1 << imm3));
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, 0, 1 << imm3);
FlushAll();
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
if (likely && !branchInfo.delaySlotIsBranch)
CompileDelaySlot();
@@ -306,7 +306,7 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
// Taken
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -355,11 +355,11 @@ void IRFrontend::Comp_Jump(MIPSOpcode op) {
}
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
FlushAll();
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
// Account for the delay slot.
js.compilerPC += 4;
@@ -420,7 +420,7 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
}
int dcAmount = js.downcountAmount;
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
ir.Write(IROp::ExitToReg, 0, destReg, 0);
@@ -433,18 +433,18 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
void IRFrontend::Comp_Syscall(MIPSOpcode op) {
// Note: If we're in a delay slot, this is off by one compared to the interpreter.
int dcAmount = js.downcountAmount + (js.inDelaySlot ? -1 : 0);
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
js.downcountAmount = 0;
// If not in a delay slot, we need to update PC.
if (!js.inDelaySlot) {
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC() + 4));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC() + 4);
}
FlushAll();
RestoreRoundingMode();
ir.Write(IROp::Syscall, 0, ir.AddConstant(op.encoding));
ir.Write(IROp::Syscall, 0, 0, 0, op.encoding);
ApplyRoundingMode();
ir.Write(IROp::ExitToPC);
@@ -452,7 +452,7 @@ void IRFrontend::Comp_Syscall(MIPSOpcode op) {
}
void IRFrontend::Comp_Break(MIPSOpcode op) {
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
ir.Write(IROp::Break);
js.compiling = false;
}
+4 -4
View File
@@ -83,11 +83,11 @@ void IRFrontend::Comp_FPULS(MIPSOpcode op) {
switch (op >> 26) {
case 49: // lwc1
ir.Write(IROp::LoadFloat, ft, rs, ir.AddConstant(offset));
ir.Write(IROp::LoadFloat, ft, rs, 0, offset);
break;
case 57: // swc1
ir.Write(IROp::StoreFloat, ft, rs, ir.AddConstant(offset));
ir.Write(IROp::StoreFloat, ft, rs, 0, offset);
break;
default:
@@ -207,10 +207,10 @@ void IRFrontend::Comp_mxc1(MIPSOpcode op) {
// This needs to insert fpcond.
ir.Write(IROp::FpCtrlToReg, rt);
} else if (fs == 0) {
ir.Write(IROp::SetConst, rt, ir.AddConstant(MIPSState::FCR0_VALUE));
ir.WriteSetConstant(rt, MIPSState::FCR0_VALUE);
} else {
// Unsupported regs are always 0.
ir.Write(IROp::SetConst, rt, ir.AddConstant(0));
ir.WriteSetConstant(rt, 0);
}
return;
+14 -14
View File
@@ -61,42 +61,42 @@ namespace MIPSComp {
switch (o) {
// Load
case 35:
ir.Write(IROp::Load32, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32, rt, rs, 0, offset);
break;
case 37:
ir.Write(IROp::Load16, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load16, rt, rs, 0, offset);
break;
case 33:
ir.Write(IROp::Load16Ext, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load16Ext, rt, rs, 0, offset);
break;
case 36:
ir.Write(IROp::Load8, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load8, rt, rs, 0, offset);
break;
case 32:
ir.Write(IROp::Load8Ext, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load8Ext, rt, rs, 0, offset);
break;
// Store
case 43:
ir.Write(IROp::Store32, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32, rt, rs, 0, offset);
break;
case 41:
ir.Write(IROp::Store16, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store16, rt, rs, 0, offset);
break;
case 40:
ir.Write(IROp::Store8, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store8, rt, rs, 0, offset);
break;
case 34: //lwl
ir.Write(IROp::Load32Left, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32Left, rt, rs, 0, offset);
break;
case 38: //lwr
ir.Write(IROp::Load32Right, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32Right, rt, rs, 0, offset);
break;
case 42: //swl
ir.Write(IROp::Store32Left, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32Left, rt, rs, 0, offset);
break;
case 46: //swr
ir.Write(IROp::Store32Right, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32Right, rt, rs, 0, offset);
break;
default:
@@ -117,11 +117,11 @@ namespace MIPSComp {
switch (op >> 26) {
case 48: // ll
ir.Write(IROp::Load32Linked, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Load32Linked, rt, rs, 0, offset);
break;
case 56: // sc
ir.Write(IROp::Store32Conditional, rt, rs, ir.AddConstant(offset));
ir.Write(IROp::Store32Conditional, rt, rs, 0, offset);
break;
default:
+41 -40
View File
@@ -256,11 +256,11 @@ namespace MIPSComp {
}
// Nope, it has something else going on.
zeroedLanes = -1;
zeroedLanes = (u32)-1;
break;
}
if (zeroedLanes != -1) {
if (zeroedLanes != (u32)-1) {
InitRegs(vregs, tempReg);
ir.Write(IROp::Vec4Init, vregs[0], (int)Vec4Init::AllZERO);
ir.Write(IROp::Vec4Blend, vregs[0], origV[0], vregs[0], zeroedLanes);
@@ -284,8 +284,9 @@ namespace MIPSComp {
if (!constants) {
if (regnum >= n) {
// Depends on the op, but often zero.
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(vregs[i], 0.0f);
} else if (abs) {
// Could have a FNAbs op, but probably not worth it.
ir.Write(IROp::FAbs, vregs[i], origV[regnum]);
if (negate)
ir.Write(IROp::FNeg, vregs[i], vregs[i]);
@@ -297,9 +298,9 @@ namespace MIPSComp {
}
} else {
if (negate) {
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(-constantArray[regnum + (abs << 2)]));
ir.WriteSetConstantFloat(vregs[i], -constantArray[regnum + (abs << 2)]);
} else {
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(constantArray[regnum + (abs << 2)]));
ir.WriteSetConstantFloat(vregs[i], constantArray[regnum + (abs << 2)]);
}
}
}
@@ -401,11 +402,11 @@ namespace MIPSComp {
switch (op >> 26) {
case 50: //lv.s
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, 0, offset);
break;
case 58: //sv.s
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, 0, offset);
break;
default:
@@ -461,29 +462,29 @@ namespace MIPSComp {
switch (optype) {
case LSVType::LVQ:
if (IsVec4(V_Quad, vregs)) {
ir.Write(IROp::LoadVec4, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::LoadVec4, vregs[0], rs, 0, imm);
} else {
// Let's not even bother with "vertical" loads for now.
if (!g_Config.bFastMemory)
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 0, (u32)imm);
ir.Write(IROp::LoadFloat, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::LoadFloat, vregs[1], rs, ir.AddConstant(imm + 4));
ir.Write(IROp::LoadFloat, vregs[2], rs, ir.AddConstant(imm + 8));
ir.Write(IROp::LoadFloat, vregs[3], rs, ir.AddConstant(imm + 12));
ir.Write(IROp::LoadFloat, vregs[0], rs, 0, imm);
ir.Write(IROp::LoadFloat, vregs[1], rs, 0, imm + 4);
ir.Write(IROp::LoadFloat, vregs[2], rs, 0, imm + 8);
ir.Write(IROp::LoadFloat, vregs[3], rs, 0, imm + 12);
}
break;
case LSVType::SVQ:
if (IsVec4(V_Quad, vregs)) {
ir.Write(IROp::StoreVec4, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::StoreVec4, vregs[0], rs, 0, imm);
} else {
// Let's not even bother with "vertical" stores for now.
if (!g_Config.bFastMemory)
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 1, (u32)imm);
ir.Write(IROp::StoreFloat, vregs[0], rs, ir.AddConstant(imm));
ir.Write(IROp::StoreFloat, vregs[1], rs, ir.AddConstant(imm + 4));
ir.Write(IROp::StoreFloat, vregs[2], rs, ir.AddConstant(imm + 8));
ir.Write(IROp::StoreFloat, vregs[3], rs, ir.AddConstant(imm + 12));
ir.Write(IROp::StoreFloat, vregs[0], rs, 0, imm);
ir.Write(IROp::StoreFloat, vregs[1], rs, 0, imm + 4);
ir.Write(IROp::StoreFloat, vregs[2], rs, 0, imm + 8);
ir.Write(IROp::StoreFloat, vregs[3], rs, 0, imm + 12);
}
break;
@@ -521,7 +522,7 @@ namespace MIPSComp {
ir.Write(IROp::Vec4Init, dregs[0], (int)(type == 6 ? Vec4Init::AllZERO : Vec4Init::AllONE));
} else {
for (int i = 0; i < n; i++) {
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(type == 6 ? 0.0f : 1.0f));
ir.WriteSetConstantFloat(dregs[i], type == 6 ? 0.0f : 1.0f);
}
}
ApplyPrefixD(dregs, sz, vd);
@@ -549,14 +550,14 @@ namespace MIPSComp {
} else {
switch (sz) {
case V_Pair:
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 1) == 0 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 1) == 1 ? 1.0f : 0.0f));
ir.WriteSetConstantFloat(dregs[0], (vd & 1) == 0 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[1], (vd & 1) == 1 ? 1.0f : 0.0f);
break;
case V_Quad:
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 3) == 0 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 3) == 1 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[2], ir.AddConstantFloat((vd & 3) == 2 ? 1.0f : 0.0f));
ir.Write(IROp::SetConstF, dregs[3], ir.AddConstantFloat((vd & 3) == 3 ? 1.0f : 0.0f));
ir.WriteSetConstantFloat(dregs[0], (vd & 3) == 0 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[1], (vd & 3) == 1 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[2], (vd & 3) == 2 ? 1.0f : 0.0f);
ir.WriteSetConstantFloat(dregs[3], (vd & 3) == 3 ? 1.0f : 0.0f);
break;
default:
INVALIDOP;
@@ -594,19 +595,19 @@ namespace MIPSComp {
switch ((op >> 16) & 0xF) {
case 3: // vmidt
if (x == 0 && y == 0)
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
else if (x == y)
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
else
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
break;
case 6: // vmzero
// Likely to be fast.
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
break;
case 7: // vmone
if (x == 0 && y == 0)
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
else
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
break;
@@ -707,7 +708,7 @@ namespace MIPSComp {
GetVectorRegsPrefixD(dregs, V_Single, _VD);
// We have to start at +0.000 in case any values are -0.000.
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(0.0f));
ir.WriteSetConstantFloat(IRVTEMP_0, 0.0f);
for (int i = 0; i < n; ++i) {
ir.Write(IROp::FAdd, IRVTEMP_0, IRVTEMP_0, sregs[i]);
}
@@ -717,7 +718,7 @@ namespace MIPSComp {
ir.Write(IROp::FMov, dregs[0], IRVTEMP_0);
break;
case 7: // vavg
ir.Write(IROp::SetConstF, IRVTEMP_0 + 1, ir.AddConstantFloat(vavg_table[n - 1]));
ir.WriteSetConstantFloat(IRVTEMP_0 + 1, vavg_table[n - 1]);
ir.Write(IROp::FMul, dregs[0], IRVTEMP_0, IRVTEMP_0 + 1);
break;
}
@@ -939,7 +940,7 @@ namespace MIPSComp {
case VecDo3Op::VSGE: // vsge
ir.Write(IROp::FCmp, (int)IRFpCompareMode::LessUnordered, sregs[i], tregs[i]);
ir.Write(IROp::FpCondToReg, IRTEMP_1);
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, ir.AddConstant(1));
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, 0, 1);
ir.Write(IROp::FMovFromGPR, tempregs[i], IRTEMP_1);
ir.Write(IROp::FCvtSW, tempregs[i], tempregs[i]);
break;
@@ -1266,7 +1267,7 @@ namespace MIPSComp {
u32 mask;
if (GetVFPUCtrlMask(imm - 128, &mask)) {
if (mask != 0xFFFFFFFF) {
ir.Write(IROp::AndConst, IRTEMP_0, rt, ir.AddConstant(mask));
ir.Write(IROp::AndConst, IRTEMP_0, rt, 0, mask);
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, IRTEMP_0);
} else {
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, rt);
@@ -1322,7 +1323,7 @@ namespace MIPSComp {
if (GetVFPUCtrlMask(imm, &mask)) {
if (mask != 0xFFFFFFFF) {
ir.Write(IROp::FMovToGPR, IRTEMP_0, vfpuBase + voffset[imm]);
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, ir.AddConstant(mask));
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, 0, mask);
ir.Write(IROp::SetCtrlVFPUReg, imm, IRTEMP_0);
} else {
ir.Write(IROp::SetCtrlVFPUFReg, imm, vfpuBase + voffset[vs]);
@@ -2146,7 +2147,7 @@ namespace MIPSComp {
s32 imm = SignExtend16ToS32(op);
u8 dreg;
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat((float)imm));
ir.WriteSetConstantFloat(dreg, (float)imm);
ApplyPrefixD(&dreg, V_Single, _VT);
}
@@ -2164,7 +2165,7 @@ namespace MIPSComp {
u8 dreg;
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat(fval.f));
ir.WriteSetConstantFloat(dreg, fval.f);
ApplyPrefixD(&dreg, V_Single, _VT);
}
@@ -2186,17 +2187,17 @@ namespace MIPSComp {
GetVectorRegsPrefixD(dregs, sz, vd);
if (IsVec4(sz, dregs)) {
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
ir.Write(IROp::Vec4Shuffle, dregs[0], IRVTEMP_0, 0);
} else if (IsVec3of4(sz, dregs) && opts.preferVec4) {
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
ir.Write(IROp::Vec4Shuffle, IRVTEMP_0, IRVTEMP_0, 0);
ir.Write(IROp::Vec4Blend, dregs[0], dregs[0], IRVTEMP_0, 0x7);
} else {
for (int i = 0; i < n; i++) {
// Most of the time, materializing a float is slower than copying from another float.
if (i == 0)
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(cst_constants[conNum]));
ir.WriteSetConstantFloat(dregs[i], cst_constants[conNum]);
else
ir.Write(IROp::FMov, dregs[i], dregs[0]);
}
@@ -2253,7 +2254,7 @@ namespace MIPSComp {
for (int i = 0; i < n; i++) {
switch (d[i]) {
case '0':
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(0.0f));
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f);
break;
case 's':
if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) {
@@ -2271,7 +2272,7 @@ namespace MIPSComp {
else if (dregs[sineLane] == sreg[0])
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
else
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(1.0f));
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
break;
}
}
+16 -15
View File
@@ -76,17 +76,17 @@ void IRFrontend::FlushPrefixV() {
}
if ((js.prefixSFlag & JitState::PREFIX_DIRTY) != 0) {
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, ir.AddConstant(js.prefixS));
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, 0, 0, js.prefixS);
js.prefixSFlag = (JitState::PrefixState) (js.prefixSFlag & ~JitState::PREFIX_DIRTY);
}
if ((js.prefixTFlag & JitState::PREFIX_DIRTY) != 0) {
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, ir.AddConstant(js.prefixT));
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, 0, 0, js.prefixT);
js.prefixTFlag = (JitState::PrefixState) (js.prefixTFlag & ~JitState::PREFIX_DIRTY);
}
if ((js.prefixDFlag & JitState::PREFIX_DIRTY) != 0) {
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, ir.AddConstant(js.prefixD));
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, 0, 0, js.prefixD);
js.prefixDFlag = (JitState::PrefixState) (js.prefixDFlag & ~JitState::PREFIX_DIRTY);
}
@@ -164,8 +164,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
} else if (entry->replaceFunc) {
FlushAll();
RestoreRoundingMode();
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::CallReplacement, IRTEMP_0, ir.AddConstant(index));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
ir.Write(IROp::CallReplacement, IRTEMP_0, 0, 0, index);
if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) {
// Compile the original instruction at this address. We ignore cycles for hooks.
@@ -175,8 +175,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
ApplyRoundingMode();
// If IRTEMP_0 was set to 1, it means the replacement needs to run again (sliced.)
// This is necessary for replacements that take a lot of cycles.
ir.Write(IROp::Downcount, 0, ir.AddConstant(js.downcountAmount));
ir.Write(IROp::ExitToConstIfNeq, ir.AddConstant(GetCompilerPC()), IRTEMP_0, MIPS_REG_ZERO);
ir.Write(IROp::Downcount, 0, 0, 0, js.downcountAmount);
ir.Write(IROp::ExitToConstIfNeq, 0, IRTEMP_0, MIPS_REG_ZERO, GetCompilerPC());
ir.Write(IROp::ExitToReg, 0, MIPS_REG_RA, 0);
js.compiling = false;
}
@@ -187,7 +187,7 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
void IRFrontend::Comp_Generic(MIPSOpcode op) {
FlushAll();
ir.Write(IROp::Interpret, 0, ir.AddConstant(op.encoding));
ir.Write(IROp::Interpret, 0, 0, 0, op.encoding);
const MIPSInfo info = MIPSGetInfo(op);
if ((info & IS_VFPU) != 0 && (info & VFPU_NO_PREFIX) == 0) {
// If it does eat them, it'll happen in MIPSCompileOp().
@@ -359,7 +359,7 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
FlushAll();
// Can't skip this even at the start of a block, might impact block linking.
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
RestoreRoundingMode();
// At this point, downcount HAS the delay slot, but not the instruction itself.
@@ -374,11 +374,12 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
}
}
int downcountAmount = js.downcountAmount + downcountOffset;
if (downcountAmount != 0)
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
if (downcountAmount != 0) {
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
}
// Note that this means downcount can't be metadata on the block.
js.downcountAmount = -downcountOffset;
ir.Write(IROp::Breakpoint, 0, ir.AddConstant(addr));
ir.Write(IROp::Breakpoint, 0, 0, 0, addr);
ApplyRoundingMode();
js.hadBreakpoints = true;
@@ -390,7 +391,7 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
FlushAll();
// Can't skip this even at the start of a block, might impact block linking.
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
RestoreRoundingMode();
// At this point, downcount HAS the delay slot, but not the instruction itself.
@@ -407,10 +408,10 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
}
int downcountAmount = js.downcountAmount + downcountOffset;
if (downcountAmount != 0)
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
// Note that this means downcount can't be metadata on the block.
js.downcountAmount = -downcountOffset;
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, ir.AddConstant(offset));
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, 0, offset);
ApplyRoundingMode();
js.hadBreakpoints = true;
+17 -13
View File
@@ -206,31 +206,35 @@ void InitIR() {
}
}
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2) {
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2, u32 constant) {
IRInst inst;
inst.op = op;
inst.dest = dst;
inst.src1 = src1;
inst.src2 = src2;
inst.constant = nextConst_;
inst.constant = constant;
insts_.push_back(inst);
}
nextConst_ = 0;
void IRWriter::WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant) {
u32 constant;
memcpy(&constant, &fconstant, sizeof(u32));
IRInst inst;
inst.op = op;
inst.dest = dst;
inst.src1 = src1;
inst.src2 = src2;
inst.constant = constant;
insts_.push_back(inst);
}
void IRWriter::WriteSetConstant(u8 dst, u32 value) {
Write(IROp::SetConst, dst, AddConstant(value));
Write(IROp::SetConst, dst, 0, 0, value);
}
int IRWriter::AddConstant(u32 value) {
nextConst_ = value;
return 255;
}
int IRWriter::AddConstantFloat(float value) {
u32 val;
memcpy(&val, &value, 4);
return AddConstant(val);
void IRWriter::WriteSetConstantFloat(u8 dst, float value) {
WriteFC(IROp::SetConstF, dst, 0, 0, value);
}
void IRWriter::ReplaceConstant(size_t instNumber, u32 newConstant) {
+6 -11
View File
@@ -222,7 +222,7 @@ enum class IROp : uint8_t {
ExitToConstIfFpFalse,
ExitToPC, // Used after a syscall to give us a way to do things before returning.
Syscall,
Syscall, // puts the address of the syscall instruction in the constant - we need both, but we can use the address to look up the value.
SetPC, // hack to make syscall returns work
SetPCConst, // hack to make replacement know PC
CallReplacement,
@@ -384,18 +384,14 @@ public:
return *this;
}
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0);
void Write(IROp op, IRReg dst, IRReg src1, IRReg src2, uint32_t c) {
AddConstant(c);
Write(op, dst, src1, src2);
}
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0, u32 constant = 0);
void WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant);
void WriteSetConstant(u8 dst, u32 value);
void WriteSetConstantFloat(u8 dst, float value);
void Write(IRInst inst) {
insts_.push_back(inst);
}
void WriteSetConstant(u8 dst, u32 value);
int AddConstant(u32 value);
int AddConstantFloat(float value);
void Reserve(size_t s) {
insts_.reserve(s);
@@ -409,7 +405,6 @@ public:
private:
std::vector<IRInst> insts_;
u32 nextConst_ = 0;
};
struct IROptions {
+40 -38
View File
@@ -258,21 +258,21 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
if (opts.unalignedLoadStore) {
// Write out one unaligned op.
out.Write(replaceOp, inst.dest, inst.src1, out.AddConstant(inst.constant + replaceOff));
out.Write(replaceOp, inst.dest, inst.src1, 0, inst.constant + replaceOff);
} else if (replaceOp == IROp::Load32) {
// We can still combine to a simpler set of two loads.
// We start by isolating the address and shift amount.
// IRTEMP_LR_ADDR = rs + imm
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant + replaceOff));
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant + replaceOff);
// IRTEMP_LR_SHIFT = (addr & 3) * 8
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
// IRTEMP_LR_ADDR = addr & 0xfffffffc
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
// IRTEMP_LR_VALUE = low_word, dest = high_word
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, out.AddConstant(0));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(4));
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, 0, 0);
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 4);
// Now we just need to adjust and combine dest and IRTEMP_LR_VALUE.
// inst.dest >>= shift (putting its bits in the right spot.)
@@ -281,7 +281,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::ShlImm, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, 8);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
// IRTEMP_LR_VALUE <<= (24 - shift)
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
@@ -297,18 +297,18 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
auto addCommonProlog = [&]() {
// IRTEMP_LR_ADDR = rs + imm
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant));
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant);
// IRTEMP_LR_SHIFT = (addr & 3) * 8
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
// IRTEMP_LR_ADDR = addr & 0xfffffffc (for stores, later)
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR)
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(0));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 0);
};
auto addCommonStore = [&](int off = 0) {
// RAM(IRTEMP_LR_ADDR) = IRTEMP_LR_VALUE
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(off));
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, off);
};
switch (inst.op) {
@@ -327,7 +327,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::And, inst.dest, inst.dest, IRTEMP_LR_MASK);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
// IRTEMP_LR_VALUE <<= (24 - shift)
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
// dest |= IRTEMP_LR_VALUE
@@ -336,7 +336,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
bool src1Dirty = inst.dest == inst.src1;
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, nextOp().constant - inst.constant);
// dest &= IRTEMP_LR_MASK
out.Write(IROp::And, nextOp().dest, nextOp().dest, IRTEMP_LR_MASK);
@@ -362,7 +362,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::Shr, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
// dest &= (0xffffff00 << (24 - shift))
// Alternatively, could shift to a wall and back (but would require two shifts each way.)
out.WriteSetConstant(IRTEMP_LR_MASK, 0xffffff00);
@@ -377,12 +377,12 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
bool src1Dirty = inst.dest == inst.src1;
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, (u32)(nextOp().constant - inst.constant));
if (shiftNeedsReverse) {
// IRTEMP_LR_SHIFT = shift again
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
shiftNeedsReverse = false;
}
// IRTEMP_LR_VALUE >>= IRTEMP_LR_SHIFT
@@ -411,7 +411,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
// IRTEMP_LR_SHIFT = 24 - shift
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
// IRTEMP_LR_VALUE |= src3 >> (24 - shift)
out.Write(IROp::Shr, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
@@ -429,11 +429,11 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
// IRTEMP_LR_VALUE &= 0x00ffffff << (24 - shift)
out.WriteSetConstant(IRTEMP_LR_MASK, 0x00ffffff);
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
out.Write(IROp::Shr, IRTEMP_LR_MASK, IRTEMP_LR_MASK, IRTEMP_LR_SHIFT);
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
// IRTEMP_LR_VALUE |= src3 << shift
out.Write(IROp::Shl, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
@@ -508,7 +508,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
if (inst.dest != inst.src1)
out.Write(IROp::Mov, inst.dest, inst.src1);
} else {
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, out.AddConstant(imm2));
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, 0, imm2);
}
} else if (symmetric && gpr.IsImm(inst.src1)) {
const u32 imm1 = gpr.GetImm(inst.src1);
@@ -518,7 +518,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
if (inst.dest != inst.src2)
out.Write(IROp::Mov, inst.dest, inst.src2);
} else {
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, out.AddConstant(imm1));
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, 0, imm1);
}
} else {
gpr.MapDirtyInIn(inst.dest, inst.src1, inst.src2);
@@ -632,7 +632,8 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::FMovFromGPR:
if (gpr.IsImm(inst.src1)) {
out.Write(IROp::SetConstF, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
// NOTE: SetConstantFloat doesn't work here since we actually want the bits.
out.Write(IROp::SetConstF, inst.dest, 0, 0, gpr.GetImm(inst.src1));
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -661,7 +662,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::Store32Conditional:
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
gpr.MapIn(inst.dest);
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapInIn(inst.dest, inst.src1);
goto doDefault;
@@ -670,7 +671,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::StoreFloat:
case IROp::StoreVec4:
if (gpr.IsImm(inst.src1)) {
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -685,7 +686,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::Load32Linked:
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
gpr.MapDirty(inst.dest);
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapDirtyIn(inst.dest, inst.src1);
goto doDefault;
@@ -694,7 +695,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::LoadFloat:
case IROp::LoadVec4:
if (gpr.IsImm(inst.src1)) {
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -704,7 +705,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::Load32Right:
if (gpr.IsImm(inst.src1)) {
gpr.MapIn(inst.dest);
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapInIn(inst.dest, inst.src1);
goto doDefault;
@@ -716,7 +717,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::ValidateAddress32:
case IROp::ValidateAddress128:
if (gpr.IsImm(inst.src1)) {
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -729,7 +730,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::SetPC:
if (gpr.IsImm(inst.src1)) {
out.Write(IROp::SetPCConst, out.AddConstant(gpr.GetImm(inst.src1)));
out.Write(IROp::SetPCConst, 0, 0, 0, gpr.GetImm(inst.src1));
} else {
gpr.MapIn(inst.src1);
goto doDefault;
@@ -772,7 +773,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
case IROp::SetCtrlVFPUReg:
if (gpr.IsImm(inst.src1)) {
out.Write(IROp::SetCtrlVFPU, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
out.Write(IROp::SetCtrlVFPU, inst.dest, 0, 0, gpr.GetImm(inst.src1));
} else {
gpr.MapDirtyIn(IRREG_VFPU_CTRL_BASE + inst.dest, inst.src1);
out.Write(inst);
@@ -872,7 +873,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
// Reduce bloat by skipping on fail, and const exit on pass.
if (passed) {
gpr.FlushAll();
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
skipNextExitToConst = true;
}
break;
@@ -896,7 +897,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
if (passed) {
gpr.FlushAll();
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
skipNextExitToConst = true;
}
break;
@@ -918,7 +919,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
// Prefer ExitToConst to allow block linking.
u32 dest = gpr.GetImm(inst.src1);
gpr.FlushAll();
out.Write(IROp::ExitToConst, out.AddConstant(dest));
out.Write(IROp::ExitToConst, 0, 0, 0, dest);
break;
}
gpr.FlushAll();
@@ -2040,10 +2041,11 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
} else if (inst.constant == 0xBF800000) {
out.Write(IROp::Vec4Init, temp, (int)Vec4Init::AllMinusONE);
} else {
out.Write(IROp::SetConstF, temp, out.AddConstant(inst.constant));
// NOTE: WriteSetConstantFloat doesn't work here since we actually want the bits.
out.Write(IROp::SetConstF, temp, 0, 0, inst.constant);
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
}
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
isVec4Dirty[inst.dest & ~3] = true;
continue;
}
@@ -2054,7 +2056,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
u8 blendMask = 1 << (inst.dest & 3);
out.Write(IROp::FMovFromGPR, temp, inst.src1);
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
isVec4Dirty[inst.dest & ~3] = true;
continue;
}
@@ -2065,7 +2067,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
u8 blendMask = 1 << (inst.dest & 3);
out.Write(inst.op, temp, inst.src1, inst.src2, inst.constant);
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
isVec4Dirty[inst.dest & ~3] = true;
continue;
}