mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
IRWriter: Remove the confusing and inefficient AddConstant
This commit is contained in:
1 parent
2096179dce
commit
ae14ebb6ac
12 files changed
+178
-175
No files matched your search
+2
-2
@@ -98,8 +98,8 @@ struct LogChannel {
|
||||
#endif
|
||||
bool enabled = true;
|
||||
|
||||
bool IsEnabled(LogLevel level) const {
|
||||
if (level > this->level || !this->enabled)
|
||||
bool IsEnabled(LogLevel logLevel) const {
|
||||
if (logLevel > level || !enabled)
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -197,17 +197,17 @@ inline __m128i _mm_mullo_epi32_SSE2(const __m128i v0, const __m128i v1) {
|
||||
inline __m128i _mm_max_epu16_SSE2(const __m128i v0, const __m128i v1) {
|
||||
return _mm_xor_si128(
|
||||
_mm_max_epi16(
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)0x8000));
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
|
||||
}
|
||||
|
||||
inline __m128i _mm_min_epu16_SSE2(const __m128i v0, const __m128i v1) {
|
||||
return _mm_xor_si128(
|
||||
_mm_min_epi16(
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)0x8000));
|
||||
_mm_xor_si128(v0, _mm_set1_epi16((int16_t)(uint16_t)0x8000)),
|
||||
_mm_xor_si128(v1, _mm_set1_epi16((int16_t)(uint16_t)0x8000))),
|
||||
_mm_set1_epi16((int16_t)(uint16_t)0x8000));
|
||||
}
|
||||
|
||||
// SSE2 replacement for half of a _mm_packus_epi32 but without the saturation.
|
||||
|
||||
@@ -34,7 +34,7 @@ inline constexpr uint32_t RoundDownToMultipleOf(uint32_t v, uint32_t multiple) {
|
||||
|
||||
// TODO: this should just use a bitscan.
|
||||
inline uint32_t log2i(uint32_t val) {
|
||||
unsigned int ret = -1;
|
||||
unsigned int ret = (unsigned int)-1;
|
||||
while (val != 0) {
|
||||
val >>= 1; ret++;
|
||||
}
|
||||
|
||||
+11
-11
@@ -61,19 +61,19 @@ void IRFrontend::Comp_IType(MIPSOpcode op) {
|
||||
switch (op >> 26) {
|
||||
case 8: // same as addiu?
|
||||
case 9: // R(rt) = R(rs) + simm; break; //addiu
|
||||
ir.Write(IROp::AddConst, rt, rs, ir.AddConstant(simm));
|
||||
ir.Write(IROp::AddConst, rt, rs, 0, (u32)simm);
|
||||
break;
|
||||
|
||||
case 12: ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(uimm)); break;
|
||||
case 13: ir.Write(IROp::OrConst, rt, rs, ir.AddConstant(uimm)); break;
|
||||
case 14: ir.Write(IROp::XorConst, rt, rs, ir.AddConstant(uimm)); break;
|
||||
case 12: ir.Write(IROp::AndConst, rt, rs, 0, uimm); break;
|
||||
case 13: ir.Write(IROp::OrConst, rt, rs, 0, uimm); break;
|
||||
case 14: ir.Write(IROp::XorConst, rt, rs, 0, uimm); break;
|
||||
|
||||
case 10: // R(rt) = (s32)R(rs) < simm; break; //slti
|
||||
ir.Write(IROp::SltConst, rt, rs, ir.AddConstant(simm));
|
||||
ir.Write(IROp::SltConst, rt, rs, 0, (u32)simm);
|
||||
break;
|
||||
|
||||
case 11: // R(rt) = R(rs) < suimm; break; //sltiu
|
||||
ir.Write(IROp::SltUConst, rt, rs, ir.AddConstant(suimm));
|
||||
ir.Write(IROp::SltUConst, rt, rs, 0, suimm);
|
||||
break;
|
||||
|
||||
case 15: // R(rt) = uimm << 16; //lui
|
||||
@@ -197,7 +197,7 @@ void IRFrontend::CompShiftVar(MIPSOpcode op, IROp shiftOp) {
|
||||
// The interpreter already masks where needed, don't need to generate extra ops.
|
||||
ir.Write(shiftOp, rd, rt, rs);
|
||||
} else {
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(31));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, (u32)31);
|
||||
ir.Write(shiftOp, rd, rt, IRTEMP_0);
|
||||
}
|
||||
}
|
||||
@@ -244,9 +244,9 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
|
||||
case 0x0: // ext
|
||||
if (pos != 0) {
|
||||
ir.Write(IROp::ShrImm, rt, rs, pos);
|
||||
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, rt, rt, 0, mask);
|
||||
} else {
|
||||
ir.Write(IROp::AndConst, rt, rs, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, rt, rs, 0, mask);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -259,7 +259,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
|
||||
|
||||
if (size != 32) {
|
||||
// Need to use the sourcemask.
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, ir.AddConstant(sourcemask));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rs, 0, sourcemask);
|
||||
if (pos != 0) {
|
||||
ir.Write(IROp::ShlImm, IRTEMP_0, IRTEMP_0, pos);
|
||||
}
|
||||
@@ -271,7 +271,7 @@ void IRFrontend::Comp_Special3(MIPSOpcode op) {
|
||||
ir.Write(IROp::Mov, IRTEMP_0, rs);
|
||||
}
|
||||
}
|
||||
ir.Write(IROp::AndConst, rt, rt, ir.AddConstant(destmask));
|
||||
ir.Write(IROp::AndConst, rt, rt, 0, destmask);
|
||||
ir.Write(IROp::Or, rt, rt, IRTEMP_0);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -96,11 +96,11 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs, rhs);
|
||||
ir.Write(ComparisonToExit(cc), 0, lhs, rhs, ResolveNotTakenTarget(branchInfo));
|
||||
// This makes the block "impure" :(
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
@@ -114,7 +114,7 @@ void IRFrontend::BranchRSRTComp(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
}
|
||||
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -147,11 +147,11 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), lhs);
|
||||
ir.Write(ComparisonToExit(cc), 0, lhs, 0, ResolveNotTakenTarget(branchInfo));
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
if (branchInfo.delaySlotIsBranch) {
|
||||
@@ -165,7 +165,7 @@ void IRFrontend::BranchRSZeroComp(MIPSOpcode op, IRComparison cc, bool andLink,
|
||||
|
||||
// Taken
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -225,12 +225,12 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
// Not taken
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
|
||||
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
|
||||
// Taken
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
@@ -244,7 +244,7 @@ void IRFrontend::BranchFPFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
}
|
||||
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -284,14 +284,14 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
CompileDelaySlot();
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
int imm3 = (op >> 18) & 7;
|
||||
|
||||
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, ir.AddConstant(1 << imm3));
|
||||
ir.Write(IROp::AndConst, IRTEMP_LHS, IRTEMP_LHS, 0, 1 << imm3);
|
||||
FlushAll();
|
||||
ir.Write(ComparisonToExit(cc), ir.AddConstant(ResolveNotTakenTarget(branchInfo)), IRTEMP_LHS, 0);
|
||||
ir.Write(ComparisonToExit(cc), 0, IRTEMP_LHS, 0, ResolveNotTakenTarget(branchInfo));
|
||||
|
||||
if (likely && !branchInfo.delaySlotIsBranch)
|
||||
CompileDelaySlot();
|
||||
@@ -306,7 +306,7 @@ void IRFrontend::BranchVFPUFlag(MIPSOpcode op, IRComparison cc, bool likely) {
|
||||
|
||||
// Taken
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -355,11 +355,11 @@ void IRFrontend::Comp_Jump(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
FlushAll();
|
||||
ir.Write(IROp::ExitToConst, ir.AddConstant(targetAddr));
|
||||
ir.Write(IROp::ExitToConst, 0, 0, 0, targetAddr);
|
||||
|
||||
// Account for the delay slot.
|
||||
js.compilerPC += 4;
|
||||
@@ -420,7 +420,7 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
int dcAmount = js.downcountAmount;
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
ir.Write(IROp::ExitToReg, 0, destReg, 0);
|
||||
@@ -433,18 +433,18 @@ void IRFrontend::Comp_JumpReg(MIPSOpcode op) {
|
||||
void IRFrontend::Comp_Syscall(MIPSOpcode op) {
|
||||
// Note: If we're in a delay slot, this is off by one compared to the interpreter.
|
||||
int dcAmount = js.downcountAmount + (js.inDelaySlot ? -1 : 0);
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(dcAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, dcAmount);
|
||||
js.downcountAmount = 0;
|
||||
|
||||
// If not in a delay slot, we need to update PC.
|
||||
if (!js.inDelaySlot) {
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC() + 4));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC() + 4);
|
||||
}
|
||||
|
||||
FlushAll();
|
||||
|
||||
RestoreRoundingMode();
|
||||
ir.Write(IROp::Syscall, 0, ir.AddConstant(op.encoding));
|
||||
ir.Write(IROp::Syscall, 0, 0, 0, op.encoding);
|
||||
ApplyRoundingMode();
|
||||
ir.Write(IROp::ExitToPC);
|
||||
|
||||
@@ -452,7 +452,7 @@ void IRFrontend::Comp_Syscall(MIPSOpcode op) {
|
||||
}
|
||||
|
||||
void IRFrontend::Comp_Break(MIPSOpcode op) {
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
ir.Write(IROp::Break);
|
||||
js.compiling = false;
|
||||
}
|
||||
|
||||
@@ -83,11 +83,11 @@ void IRFrontend::Comp_FPULS(MIPSOpcode op) {
|
||||
|
||||
switch (op >> 26) {
|
||||
case 49: // lwc1
|
||||
ir.Write(IROp::LoadFloat, ft, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::LoadFloat, ft, rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 57: // swc1
|
||||
ir.Write(IROp::StoreFloat, ft, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::StoreFloat, ft, rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -207,10 +207,10 @@ void IRFrontend::Comp_mxc1(MIPSOpcode op) {
|
||||
// This needs to insert fpcond.
|
||||
ir.Write(IROp::FpCtrlToReg, rt);
|
||||
} else if (fs == 0) {
|
||||
ir.Write(IROp::SetConst, rt, ir.AddConstant(MIPSState::FCR0_VALUE));
|
||||
ir.WriteSetConstant(rt, MIPSState::FCR0_VALUE);
|
||||
} else {
|
||||
// Unsupported regs are always 0.
|
||||
ir.Write(IROp::SetConst, rt, ir.AddConstant(0));
|
||||
ir.WriteSetConstant(rt, 0);
|
||||
}
|
||||
return;
|
||||
|
||||
|
||||
@@ -61,42 +61,42 @@ namespace MIPSComp {
|
||||
switch (o) {
|
||||
// Load
|
||||
case 35:
|
||||
ir.Write(IROp::Load32, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32, rt, rs, 0, offset);
|
||||
break;
|
||||
case 37:
|
||||
ir.Write(IROp::Load16, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load16, rt, rs, 0, offset);
|
||||
break;
|
||||
case 33:
|
||||
ir.Write(IROp::Load16Ext, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load16Ext, rt, rs, 0, offset);
|
||||
break;
|
||||
case 36:
|
||||
ir.Write(IROp::Load8, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load8, rt, rs, 0, offset);
|
||||
break;
|
||||
case 32:
|
||||
ir.Write(IROp::Load8Ext, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load8Ext, rt, rs, 0, offset);
|
||||
break;
|
||||
// Store
|
||||
case 43:
|
||||
ir.Write(IROp::Store32, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32, rt, rs, 0, offset);
|
||||
break;
|
||||
case 41:
|
||||
ir.Write(IROp::Store16, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store16, rt, rs, 0, offset);
|
||||
break;
|
||||
case 40:
|
||||
ir.Write(IROp::Store8, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store8, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 34: //lwl
|
||||
ir.Write(IROp::Load32Left, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32Left, rt, rs, 0, offset);
|
||||
break;
|
||||
case 38: //lwr
|
||||
ir.Write(IROp::Load32Right, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32Right, rt, rs, 0, offset);
|
||||
break;
|
||||
case 42: //swl
|
||||
ir.Write(IROp::Store32Left, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32Left, rt, rs, 0, offset);
|
||||
break;
|
||||
case 46: //swr
|
||||
ir.Write(IROp::Store32Right, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32Right, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -117,11 +117,11 @@ namespace MIPSComp {
|
||||
|
||||
switch (op >> 26) {
|
||||
case 48: // ll
|
||||
ir.Write(IROp::Load32Linked, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Load32Linked, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 56: // sc
|
||||
ir.Write(IROp::Store32Conditional, rt, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::Store32Conditional, rt, rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
|
||||
+41
-40
@@ -256,11 +256,11 @@ namespace MIPSComp {
|
||||
}
|
||||
|
||||
// Nope, it has something else going on.
|
||||
zeroedLanes = -1;
|
||||
zeroedLanes = (u32)-1;
|
||||
break;
|
||||
}
|
||||
|
||||
if (zeroedLanes != -1) {
|
||||
if (zeroedLanes != (u32)-1) {
|
||||
InitRegs(vregs, tempReg);
|
||||
ir.Write(IROp::Vec4Init, vregs[0], (int)Vec4Init::AllZERO);
|
||||
ir.Write(IROp::Vec4Blend, vregs[0], origV[0], vregs[0], zeroedLanes);
|
||||
@@ -284,8 +284,9 @@ namespace MIPSComp {
|
||||
if (!constants) {
|
||||
if (regnum >= n) {
|
||||
// Depends on the op, but often zero.
|
||||
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(vregs[i], 0.0f);
|
||||
} else if (abs) {
|
||||
// Could have a FNAbs op, but probably not worth it.
|
||||
ir.Write(IROp::FAbs, vregs[i], origV[regnum]);
|
||||
if (negate)
|
||||
ir.Write(IROp::FNeg, vregs[i], vregs[i]);
|
||||
@@ -297,9 +298,9 @@ namespace MIPSComp {
|
||||
}
|
||||
} else {
|
||||
if (negate) {
|
||||
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(-constantArray[regnum + (abs << 2)]));
|
||||
ir.WriteSetConstantFloat(vregs[i], -constantArray[regnum + (abs << 2)]);
|
||||
} else {
|
||||
ir.Write(IROp::SetConstF, vregs[i], ir.AddConstantFloat(constantArray[regnum + (abs << 2)]));
|
||||
ir.WriteSetConstantFloat(vregs[i], constantArray[regnum + (abs << 2)]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -401,11 +402,11 @@ namespace MIPSComp {
|
||||
|
||||
switch (op >> 26) {
|
||||
case 50: //lv.s
|
||||
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::LoadFloat, vfpuBase + voffset[vt], rs, 0, offset);
|
||||
break;
|
||||
|
||||
case 58: //sv.s
|
||||
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::StoreFloat, vfpuBase + voffset[vt], rs, 0, offset);
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -461,29 +462,29 @@ namespace MIPSComp {
|
||||
switch (optype) {
|
||||
case LSVType::LVQ:
|
||||
if (IsVec4(V_Quad, vregs)) {
|
||||
ir.Write(IROp::LoadVec4, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::LoadVec4, vregs[0], rs, 0, imm);
|
||||
} else {
|
||||
// Let's not even bother with "vertical" loads for now.
|
||||
if (!g_Config.bFastMemory)
|
||||
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 0, (u32)imm);
|
||||
ir.Write(IROp::LoadFloat, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::LoadFloat, vregs[1], rs, ir.AddConstant(imm + 4));
|
||||
ir.Write(IROp::LoadFloat, vregs[2], rs, ir.AddConstant(imm + 8));
|
||||
ir.Write(IROp::LoadFloat, vregs[3], rs, ir.AddConstant(imm + 12));
|
||||
ir.Write(IROp::LoadFloat, vregs[0], rs, 0, imm);
|
||||
ir.Write(IROp::LoadFloat, vregs[1], rs, 0, imm + 4);
|
||||
ir.Write(IROp::LoadFloat, vregs[2], rs, 0, imm + 8);
|
||||
ir.Write(IROp::LoadFloat, vregs[3], rs, 0, imm + 12);
|
||||
}
|
||||
break;
|
||||
|
||||
case LSVType::SVQ:
|
||||
if (IsVec4(V_Quad, vregs)) {
|
||||
ir.Write(IROp::StoreVec4, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::StoreVec4, vregs[0], rs, 0, imm);
|
||||
} else {
|
||||
// Let's not even bother with "vertical" stores for now.
|
||||
if (!g_Config.bFastMemory)
|
||||
ir.Write(IROp::ValidateAddress128, 0, (u8)rs, 1, (u32)imm);
|
||||
ir.Write(IROp::StoreFloat, vregs[0], rs, ir.AddConstant(imm));
|
||||
ir.Write(IROp::StoreFloat, vregs[1], rs, ir.AddConstant(imm + 4));
|
||||
ir.Write(IROp::StoreFloat, vregs[2], rs, ir.AddConstant(imm + 8));
|
||||
ir.Write(IROp::StoreFloat, vregs[3], rs, ir.AddConstant(imm + 12));
|
||||
ir.Write(IROp::StoreFloat, vregs[0], rs, 0, imm);
|
||||
ir.Write(IROp::StoreFloat, vregs[1], rs, 0, imm + 4);
|
||||
ir.Write(IROp::StoreFloat, vregs[2], rs, 0, imm + 8);
|
||||
ir.Write(IROp::StoreFloat, vregs[3], rs, 0, imm + 12);
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -521,7 +522,7 @@ namespace MIPSComp {
|
||||
ir.Write(IROp::Vec4Init, dregs[0], (int)(type == 6 ? Vec4Init::AllZERO : Vec4Init::AllONE));
|
||||
} else {
|
||||
for (int i = 0; i < n; i++) {
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(type == 6 ? 0.0f : 1.0f));
|
||||
ir.WriteSetConstantFloat(dregs[i], type == 6 ? 0.0f : 1.0f);
|
||||
}
|
||||
}
|
||||
ApplyPrefixD(dregs, sz, vd);
|
||||
@@ -549,14 +550,14 @@ namespace MIPSComp {
|
||||
} else {
|
||||
switch (sz) {
|
||||
case V_Pair:
|
||||
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 1) == 0 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 1) == 1 ? 1.0f : 0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[0], (vd & 1) == 0 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[1], (vd & 1) == 1 ? 1.0f : 0.0f);
|
||||
break;
|
||||
case V_Quad:
|
||||
ir.Write(IROp::SetConstF, dregs[0], ir.AddConstantFloat((vd & 3) == 0 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[1], ir.AddConstantFloat((vd & 3) == 1 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[2], ir.AddConstantFloat((vd & 3) == 2 ? 1.0f : 0.0f));
|
||||
ir.Write(IROp::SetConstF, dregs[3], ir.AddConstantFloat((vd & 3) == 3 ? 1.0f : 0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[0], (vd & 3) == 0 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[1], (vd & 3) == 1 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[2], (vd & 3) == 2 ? 1.0f : 0.0f);
|
||||
ir.WriteSetConstantFloat(dregs[3], (vd & 3) == 3 ? 1.0f : 0.0f);
|
||||
break;
|
||||
default:
|
||||
INVALIDOP;
|
||||
@@ -594,19 +595,19 @@ namespace MIPSComp {
|
||||
switch ((op >> 16) & 0xF) {
|
||||
case 3: // vmidt
|
||||
if (x == 0 && y == 0)
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
|
||||
else if (x == y)
|
||||
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
|
||||
else
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
|
||||
break;
|
||||
case 6: // vmzero
|
||||
// Likely to be fast.
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 0.0f);
|
||||
break;
|
||||
case 7: // vmone
|
||||
if (x == 0 && y == 0)
|
||||
ir.Write(IROp::SetConstF, dregs[y * 4 + x], ir.AddConstantFloat(1.0f));
|
||||
ir.WriteSetConstantFloat(dregs[y * 4 + x], 1.0f);
|
||||
else
|
||||
ir.Write(IROp::FMov, dregs[y * 4 + x], dregs[0]);
|
||||
break;
|
||||
@@ -707,7 +708,7 @@ namespace MIPSComp {
|
||||
GetVectorRegsPrefixD(dregs, V_Single, _VD);
|
||||
|
||||
// We have to start at +0.000 in case any values are -0.000.
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(0.0f));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0, 0.0f);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
ir.Write(IROp::FAdd, IRVTEMP_0, IRVTEMP_0, sregs[i]);
|
||||
}
|
||||
@@ -717,7 +718,7 @@ namespace MIPSComp {
|
||||
ir.Write(IROp::FMov, dregs[0], IRVTEMP_0);
|
||||
break;
|
||||
case 7: // vavg
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0 + 1, ir.AddConstantFloat(vavg_table[n - 1]));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0 + 1, vavg_table[n - 1]);
|
||||
ir.Write(IROp::FMul, dregs[0], IRVTEMP_0, IRVTEMP_0 + 1);
|
||||
break;
|
||||
}
|
||||
@@ -939,7 +940,7 @@ namespace MIPSComp {
|
||||
case VecDo3Op::VSGE: // vsge
|
||||
ir.Write(IROp::FCmp, (int)IRFpCompareMode::LessUnordered, sregs[i], tregs[i]);
|
||||
ir.Write(IROp::FpCondToReg, IRTEMP_1);
|
||||
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, ir.AddConstant(1));
|
||||
ir.Write(IROp::XorConst, IRTEMP_1, IRTEMP_1, 0, 1);
|
||||
ir.Write(IROp::FMovFromGPR, tempregs[i], IRTEMP_1);
|
||||
ir.Write(IROp::FCvtSW, tempregs[i], tempregs[i]);
|
||||
break;
|
||||
@@ -1266,7 +1267,7 @@ namespace MIPSComp {
|
||||
u32 mask;
|
||||
if (GetVFPUCtrlMask(imm - 128, &mask)) {
|
||||
if (mask != 0xFFFFFFFF) {
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rt, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, rt, 0, mask);
|
||||
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, IRTEMP_0);
|
||||
} else {
|
||||
ir.Write(IROp::SetCtrlVFPUReg, imm - 128, rt);
|
||||
@@ -1322,7 +1323,7 @@ namespace MIPSComp {
|
||||
if (GetVFPUCtrlMask(imm, &mask)) {
|
||||
if (mask != 0xFFFFFFFF) {
|
||||
ir.Write(IROp::FMovToGPR, IRTEMP_0, vfpuBase + voffset[imm]);
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, ir.AddConstant(mask));
|
||||
ir.Write(IROp::AndConst, IRTEMP_0, IRTEMP_0, 0, mask);
|
||||
ir.Write(IROp::SetCtrlVFPUReg, imm, IRTEMP_0);
|
||||
} else {
|
||||
ir.Write(IROp::SetCtrlVFPUFReg, imm, vfpuBase + voffset[vs]);
|
||||
@@ -2146,7 +2147,7 @@ namespace MIPSComp {
|
||||
s32 imm = SignExtend16ToS32(op);
|
||||
u8 dreg;
|
||||
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
|
||||
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat((float)imm));
|
||||
ir.WriteSetConstantFloat(dreg, (float)imm);
|
||||
ApplyPrefixD(&dreg, V_Single, _VT);
|
||||
}
|
||||
|
||||
@@ -2164,7 +2165,7 @@ namespace MIPSComp {
|
||||
|
||||
u8 dreg;
|
||||
GetVectorRegsPrefixD(&dreg, V_Single, _VT);
|
||||
ir.Write(IROp::SetConstF, dreg, ir.AddConstantFloat(fval.f));
|
||||
ir.WriteSetConstantFloat(dreg, fval.f);
|
||||
ApplyPrefixD(&dreg, V_Single, _VT);
|
||||
}
|
||||
|
||||
@@ -2186,17 +2187,17 @@ namespace MIPSComp {
|
||||
GetVectorRegsPrefixD(dregs, sz, vd);
|
||||
|
||||
if (IsVec4(sz, dregs)) {
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
|
||||
ir.Write(IROp::Vec4Shuffle, dregs[0], IRVTEMP_0, 0);
|
||||
} else if (IsVec3of4(sz, dregs) && opts.preferVec4) {
|
||||
ir.Write(IROp::SetConstF, IRVTEMP_0, ir.AddConstantFloat(cst_constants[conNum]));
|
||||
ir.WriteSetConstantFloat(IRVTEMP_0, cst_constants[conNum]);
|
||||
ir.Write(IROp::Vec4Shuffle, IRVTEMP_0, IRVTEMP_0, 0);
|
||||
ir.Write(IROp::Vec4Blend, dregs[0], dregs[0], IRVTEMP_0, 0x7);
|
||||
} else {
|
||||
for (int i = 0; i < n; i++) {
|
||||
// Most of the time, materializing a float is slower than copying from another float.
|
||||
if (i == 0)
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(cst_constants[conNum]));
|
||||
ir.WriteSetConstantFloat(dregs[i], cst_constants[conNum]);
|
||||
else
|
||||
ir.Write(IROp::FMov, dregs[i], dregs[0]);
|
||||
}
|
||||
@@ -2253,7 +2254,7 @@ namespace MIPSComp {
|
||||
for (int i = 0; i < n; i++) {
|
||||
switch (d[i]) {
|
||||
case '0':
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(0.0f));
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 0.0f);
|
||||
break;
|
||||
case 's':
|
||||
if (broadcastSine || !IsOverlapSafe(n, dregs, 1, sreg)) {
|
||||
@@ -2271,7 +2272,7 @@ namespace MIPSComp {
|
||||
else if (dregs[sineLane] == sreg[0])
|
||||
ir.Write(IROp::FCos, dregs[i], IRVTEMP_0);
|
||||
else
|
||||
ir.Write(IROp::SetConstF, dregs[i], ir.AddConstantFloat(1.0f));
|
||||
ir.WriteFC(IROp::SetConstF, dregs[i], 0, 0, 1.0f);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
+16
-15
@@ -76,17 +76,17 @@ void IRFrontend::FlushPrefixV() {
|
||||
}
|
||||
|
||||
if ((js.prefixSFlag & JitState::PREFIX_DIRTY) != 0) {
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, ir.AddConstant(js.prefixS));
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_SPREFIX, 0, 0, js.prefixS);
|
||||
js.prefixSFlag = (JitState::PrefixState) (js.prefixSFlag & ~JitState::PREFIX_DIRTY);
|
||||
}
|
||||
|
||||
if ((js.prefixTFlag & JitState::PREFIX_DIRTY) != 0) {
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, ir.AddConstant(js.prefixT));
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_TPREFIX, 0, 0, js.prefixT);
|
||||
js.prefixTFlag = (JitState::PrefixState) (js.prefixTFlag & ~JitState::PREFIX_DIRTY);
|
||||
}
|
||||
|
||||
if ((js.prefixDFlag & JitState::PREFIX_DIRTY) != 0) {
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, ir.AddConstant(js.prefixD));
|
||||
ir.Write(IROp::SetCtrlVFPU, VFPU_CTRL_DPREFIX, 0, 0, js.prefixD);
|
||||
js.prefixDFlag = (JitState::PrefixState) (js.prefixDFlag & ~JitState::PREFIX_DIRTY);
|
||||
}
|
||||
|
||||
@@ -164,8 +164,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
|
||||
} else if (entry->replaceFunc) {
|
||||
FlushAll();
|
||||
RestoreRoundingMode();
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::CallReplacement, IRTEMP_0, ir.AddConstant(index));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
ir.Write(IROp::CallReplacement, IRTEMP_0, 0, 0, index);
|
||||
|
||||
if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) {
|
||||
// Compile the original instruction at this address. We ignore cycles for hooks.
|
||||
@@ -175,8 +175,8 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
|
||||
ApplyRoundingMode();
|
||||
// If IRTEMP_0 was set to 1, it means the replacement needs to run again (sliced.)
|
||||
// This is necessary for replacements that take a lot of cycles.
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(js.downcountAmount));
|
||||
ir.Write(IROp::ExitToConstIfNeq, ir.AddConstant(GetCompilerPC()), IRTEMP_0, MIPS_REG_ZERO);
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, js.downcountAmount);
|
||||
ir.Write(IROp::ExitToConstIfNeq, 0, IRTEMP_0, MIPS_REG_ZERO, GetCompilerPC());
|
||||
ir.Write(IROp::ExitToReg, 0, MIPS_REG_RA, 0);
|
||||
js.compiling = false;
|
||||
}
|
||||
@@ -187,7 +187,7 @@ void IRFrontend::Comp_ReplacementFunc(MIPSOpcode op) {
|
||||
|
||||
void IRFrontend::Comp_Generic(MIPSOpcode op) {
|
||||
FlushAll();
|
||||
ir.Write(IROp::Interpret, 0, ir.AddConstant(op.encoding));
|
||||
ir.Write(IROp::Interpret, 0, 0, 0, op.encoding);
|
||||
const MIPSInfo info = MIPSGetInfo(op);
|
||||
if ((info & IS_VFPU) != 0 && (info & VFPU_NO_PREFIX) == 0) {
|
||||
// If it does eat them, it'll happen in MIPSCompileOp().
|
||||
@@ -359,7 +359,7 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
|
||||
FlushAll();
|
||||
|
||||
// Can't skip this even at the start of a block, might impact block linking.
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
|
||||
RestoreRoundingMode();
|
||||
// At this point, downcount HAS the delay slot, but not the instruction itself.
|
||||
@@ -374,11 +374,12 @@ void IRFrontend::CheckBreakpoint(u32 addr) {
|
||||
}
|
||||
}
|
||||
int downcountAmount = js.downcountAmount + downcountOffset;
|
||||
if (downcountAmount != 0)
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
|
||||
if (downcountAmount != 0) {
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
|
||||
}
|
||||
// Note that this means downcount can't be metadata on the block.
|
||||
js.downcountAmount = -downcountOffset;
|
||||
ir.Write(IROp::Breakpoint, 0, ir.AddConstant(addr));
|
||||
ir.Write(IROp::Breakpoint, 0, 0, 0, addr);
|
||||
ApplyRoundingMode();
|
||||
|
||||
js.hadBreakpoints = true;
|
||||
@@ -390,7 +391,7 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
|
||||
FlushAll();
|
||||
|
||||
// Can't skip this even at the start of a block, might impact block linking.
|
||||
ir.Write(IROp::SetPCConst, 0, ir.AddConstant(GetCompilerPC()));
|
||||
ir.Write(IROp::SetPCConst, 0, 0, 0, GetCompilerPC());
|
||||
|
||||
RestoreRoundingMode();
|
||||
// At this point, downcount HAS the delay slot, but not the instruction itself.
|
||||
@@ -407,10 +408,10 @@ void IRFrontend::CheckMemoryBreakpoint(int rs, int offset) {
|
||||
}
|
||||
int downcountAmount = js.downcountAmount + downcountOffset;
|
||||
if (downcountAmount != 0)
|
||||
ir.Write(IROp::Downcount, 0, ir.AddConstant(downcountAmount));
|
||||
ir.Write(IROp::Downcount, 0, 0, 0, downcountAmount);
|
||||
// Note that this means downcount can't be metadata on the block.
|
||||
js.downcountAmount = -downcountOffset;
|
||||
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, ir.AddConstant(offset));
|
||||
ir.Write(IROp::MemoryCheck, js.inDelaySlot ? 4 : 0, rs, 0, offset);
|
||||
ApplyRoundingMode();
|
||||
|
||||
js.hadBreakpoints = true;
|
||||
|
||||
+17
-13
@@ -206,31 +206,35 @@ void InitIR() {
|
||||
}
|
||||
}
|
||||
|
||||
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2) {
|
||||
void IRWriter::Write(IROp op, u8 dst, u8 src1, u8 src2, u32 constant) {
|
||||
IRInst inst;
|
||||
inst.op = op;
|
||||
inst.dest = dst;
|
||||
inst.src1 = src1;
|
||||
inst.src2 = src2;
|
||||
inst.constant = nextConst_;
|
||||
inst.constant = constant;
|
||||
insts_.push_back(inst);
|
||||
}
|
||||
|
||||
nextConst_ = 0;
|
||||
void IRWriter::WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant) {
|
||||
u32 constant;
|
||||
memcpy(&constant, &fconstant, sizeof(u32));
|
||||
|
||||
IRInst inst;
|
||||
inst.op = op;
|
||||
inst.dest = dst;
|
||||
inst.src1 = src1;
|
||||
inst.src2 = src2;
|
||||
inst.constant = constant;
|
||||
insts_.push_back(inst);
|
||||
}
|
||||
|
||||
void IRWriter::WriteSetConstant(u8 dst, u32 value) {
|
||||
Write(IROp::SetConst, dst, AddConstant(value));
|
||||
Write(IROp::SetConst, dst, 0, 0, value);
|
||||
}
|
||||
|
||||
int IRWriter::AddConstant(u32 value) {
|
||||
nextConst_ = value;
|
||||
return 255;
|
||||
}
|
||||
|
||||
int IRWriter::AddConstantFloat(float value) {
|
||||
u32 val;
|
||||
memcpy(&val, &value, 4);
|
||||
return AddConstant(val);
|
||||
void IRWriter::WriteSetConstantFloat(u8 dst, float value) {
|
||||
WriteFC(IROp::SetConstF, dst, 0, 0, value);
|
||||
}
|
||||
|
||||
void IRWriter::ReplaceConstant(size_t instNumber, u32 newConstant) {
|
||||
|
||||
+6
-11
@@ -222,7 +222,7 @@ enum class IROp : uint8_t {
|
||||
ExitToConstIfFpFalse,
|
||||
ExitToPC, // Used after a syscall to give us a way to do things before returning.
|
||||
|
||||
Syscall,
|
||||
Syscall, // puts the address of the syscall instruction in the constant - we need both, but we can use the address to look up the value.
|
||||
SetPC, // hack to make syscall returns work
|
||||
SetPCConst, // hack to make replacement know PC
|
||||
CallReplacement,
|
||||
@@ -384,18 +384,14 @@ public:
|
||||
return *this;
|
||||
}
|
||||
|
||||
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0);
|
||||
void Write(IROp op, IRReg dst, IRReg src1, IRReg src2, uint32_t c) {
|
||||
AddConstant(c);
|
||||
Write(op, dst, src1, src2);
|
||||
}
|
||||
void Write(IROp op, u8 dst = 0, u8 src1 = 0, u8 src2 = 0, u32 constant = 0);
|
||||
void WriteFC(IROp op, u8 dst, u8 src1, u8 src2, float fconstant);
|
||||
|
||||
void WriteSetConstant(u8 dst, u32 value);
|
||||
void WriteSetConstantFloat(u8 dst, float value);
|
||||
void Write(IRInst inst) {
|
||||
insts_.push_back(inst);
|
||||
}
|
||||
void WriteSetConstant(u8 dst, u32 value);
|
||||
|
||||
int AddConstant(u32 value);
|
||||
int AddConstantFloat(float value);
|
||||
|
||||
void Reserve(size_t s) {
|
||||
insts_.reserve(s);
|
||||
@@ -409,7 +405,6 @@ public:
|
||||
|
||||
private:
|
||||
std::vector<IRInst> insts_;
|
||||
u32 nextConst_ = 0;
|
||||
};
|
||||
|
||||
struct IROptions {
|
||||
|
||||
@@ -258,21 +258,21 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
|
||||
if (opts.unalignedLoadStore) {
|
||||
// Write out one unaligned op.
|
||||
out.Write(replaceOp, inst.dest, inst.src1, out.AddConstant(inst.constant + replaceOff));
|
||||
out.Write(replaceOp, inst.dest, inst.src1, 0, inst.constant + replaceOff);
|
||||
} else if (replaceOp == IROp::Load32) {
|
||||
// We can still combine to a simpler set of two loads.
|
||||
// We start by isolating the address and shift amount.
|
||||
|
||||
// IRTEMP_LR_ADDR = rs + imm
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant + replaceOff));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant + replaceOff);
|
||||
// IRTEMP_LR_SHIFT = (addr & 3) * 8
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
|
||||
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
|
||||
// IRTEMP_LR_ADDR = addr & 0xfffffffc
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
|
||||
// IRTEMP_LR_VALUE = low_word, dest = high_word
|
||||
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, out.AddConstant(0));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(4));
|
||||
out.Write(IROp::Load32, inst.dest, IRTEMP_LR_ADDR, 0, 0);
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 4);
|
||||
|
||||
// Now we just need to adjust and combine dest and IRTEMP_LR_VALUE.
|
||||
// inst.dest >>= shift (putting its bits in the right spot.)
|
||||
@@ -281,7 +281,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::ShlImm, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, 8);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
|
||||
// IRTEMP_LR_VALUE <<= (24 - shift)
|
||||
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
|
||||
|
||||
@@ -297,18 +297,18 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
|
||||
auto addCommonProlog = [&]() {
|
||||
// IRTEMP_LR_ADDR = rs + imm
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, out.AddConstant(inst.constant));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_ADDR, inst.src1, 0, inst.constant);
|
||||
// IRTEMP_LR_SHIFT = (addr & 3) * 8
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, out.AddConstant(3));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_SHIFT, IRTEMP_LR_ADDR, 0, 3);
|
||||
out.Write(IROp::ShlImm, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 3);
|
||||
// IRTEMP_LR_ADDR = addr & 0xfffffffc (for stores, later)
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, out.AddConstant(0xFFFFFFFC));
|
||||
out.Write(IROp::AndConst, IRTEMP_LR_ADDR, IRTEMP_LR_ADDR, 0, 0xFFFFFFFC);
|
||||
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR)
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(0));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, 0);
|
||||
};
|
||||
auto addCommonStore = [&](int off = 0) {
|
||||
// RAM(IRTEMP_LR_ADDR) = IRTEMP_LR_VALUE
|
||||
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(off));
|
||||
out.Write(IROp::Store32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, off);
|
||||
};
|
||||
|
||||
switch (inst.op) {
|
||||
@@ -327,7 +327,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::And, inst.dest, inst.dest, IRTEMP_LR_MASK);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, 24);
|
||||
// IRTEMP_LR_VALUE <<= (24 - shift)
|
||||
out.Write(IROp::Shl, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
|
||||
// dest |= IRTEMP_LR_VALUE
|
||||
@@ -336,7 +336,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
bool src1Dirty = inst.dest == inst.src1;
|
||||
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
|
||||
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, nextOp().constant - inst.constant);
|
||||
|
||||
// dest &= IRTEMP_LR_MASK
|
||||
out.Write(IROp::And, nextOp().dest, nextOp().dest, IRTEMP_LR_MASK);
|
||||
@@ -362,7 +362,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::Shr, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_SHIFT);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
// dest &= (0xffffff00 << (24 - shift))
|
||||
// Alternatively, could shift to a wall and back (but would require two shifts each way.)
|
||||
out.WriteSetConstant(IRTEMP_LR_MASK, 0xffffff00);
|
||||
@@ -377,12 +377,12 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
bool src1Dirty = inst.dest == inst.src1;
|
||||
while (i + 1 < n && !src1Dirty && nextOp().op == inst.op && nextOp().src1 == inst.src1 && (nextOp().constant & 3) == (inst.constant & 3)) {
|
||||
// IRTEMP_LR_VALUE = RAM(IRTEMP_LR_ADDR + offsetDelta)
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, out.AddConstant(nextOp().constant - inst.constant));
|
||||
out.Write(IROp::Load32, IRTEMP_LR_VALUE, IRTEMP_LR_ADDR, 0, (u32)(nextOp().constant - inst.constant));
|
||||
|
||||
if (shiftNeedsReverse) {
|
||||
// IRTEMP_LR_SHIFT = shift again
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
shiftNeedsReverse = false;
|
||||
}
|
||||
// IRTEMP_LR_VALUE >>= IRTEMP_LR_SHIFT
|
||||
@@ -411,7 +411,7 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
// IRTEMP_LR_SHIFT = 24 - shift
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
// IRTEMP_LR_VALUE |= src3 >> (24 - shift)
|
||||
out.Write(IROp::Shr, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
@@ -429,11 +429,11 @@ bool RemoveLoadStoreLeftRight(const IRWriter &in, IRWriter &out, const IROptions
|
||||
// IRTEMP_LR_VALUE &= 0x00ffffff << (24 - shift)
|
||||
out.WriteSetConstant(IRTEMP_LR_MASK, 0x00ffffff);
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
out.Write(IROp::Shr, IRTEMP_LR_MASK, IRTEMP_LR_MASK, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::And, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
out.Write(IROp::Neg, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, out.AddConstant(24));
|
||||
out.Write(IROp::AddConst, IRTEMP_LR_SHIFT, IRTEMP_LR_SHIFT, 0, (u32)24);
|
||||
// IRTEMP_LR_VALUE |= src3 << shift
|
||||
out.Write(IROp::Shl, IRTEMP_LR_MASK, inst.src3, IRTEMP_LR_SHIFT);
|
||||
out.Write(IROp::Or, IRTEMP_LR_VALUE, IRTEMP_LR_VALUE, IRTEMP_LR_MASK);
|
||||
@@ -508,7 +508,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
if (inst.dest != inst.src1)
|
||||
out.Write(IROp::Mov, inst.dest, inst.src1);
|
||||
} else {
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, out.AddConstant(imm2));
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src1, 0, imm2);
|
||||
}
|
||||
} else if (symmetric && gpr.IsImm(inst.src1)) {
|
||||
const u32 imm1 = gpr.GetImm(inst.src1);
|
||||
@@ -518,7 +518,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
if (inst.dest != inst.src2)
|
||||
out.Write(IROp::Mov, inst.dest, inst.src2);
|
||||
} else {
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, out.AddConstant(imm1));
|
||||
out.Write(ArithToArithConst(inst.op), inst.dest, inst.src2, 0, imm1);
|
||||
}
|
||||
} else {
|
||||
gpr.MapDirtyInIn(inst.dest, inst.src1, inst.src2);
|
||||
@@ -632,7 +632,8 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
case IROp::FMovFromGPR:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(IROp::SetConstF, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
|
||||
// NOTE: SetConstantFloat doesn't work here since we actually want the bits.
|
||||
out.Write(IROp::SetConstF, inst.dest, 0, 0, gpr.GetImm(inst.src1));
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -661,7 +662,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::Store32Conditional:
|
||||
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
|
||||
gpr.MapIn(inst.dest);
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapInIn(inst.dest, inst.src1);
|
||||
goto doDefault;
|
||||
@@ -670,7 +671,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::StoreFloat:
|
||||
case IROp::StoreVec4:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -685,7 +686,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::Load32Linked:
|
||||
if (gpr.IsImm(inst.src1) && inst.src1 != inst.dest) {
|
||||
gpr.MapDirty(inst.dest);
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapDirtyIn(inst.dest, inst.src1);
|
||||
goto doDefault;
|
||||
@@ -694,7 +695,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::LoadFloat:
|
||||
case IROp::LoadVec4:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -704,7 +705,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::Load32Right:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
gpr.MapIn(inst.dest);
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapInIn(inst.dest, inst.src1);
|
||||
goto doDefault;
|
||||
@@ -716,7 +717,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
case IROp::ValidateAddress32:
|
||||
case IROp::ValidateAddress128:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(inst.op, inst.dest, 0, out.AddConstant(gpr.GetImm(inst.src1) + inst.constant));
|
||||
out.Write(inst.op, inst.dest, 0, 0, gpr.GetImm(inst.src1) + inst.constant);
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -729,7 +730,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
case IROp::SetPC:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(IROp::SetPCConst, out.AddConstant(gpr.GetImm(inst.src1)));
|
||||
out.Write(IROp::SetPCConst, 0, 0, 0, gpr.GetImm(inst.src1));
|
||||
} else {
|
||||
gpr.MapIn(inst.src1);
|
||||
goto doDefault;
|
||||
@@ -772,7 +773,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
case IROp::SetCtrlVFPUReg:
|
||||
if (gpr.IsImm(inst.src1)) {
|
||||
out.Write(IROp::SetCtrlVFPU, inst.dest, out.AddConstant(gpr.GetImm(inst.src1)));
|
||||
out.Write(IROp::SetCtrlVFPU, inst.dest, 0, 0, gpr.GetImm(inst.src1));
|
||||
} else {
|
||||
gpr.MapDirtyIn(IRREG_VFPU_CTRL_BASE + inst.dest, inst.src1);
|
||||
out.Write(inst);
|
||||
@@ -872,7 +873,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
// Reduce bloat by skipping on fail, and const exit on pass.
|
||||
if (passed) {
|
||||
gpr.FlushAll();
|
||||
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
|
||||
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
|
||||
skipNextExitToConst = true;
|
||||
}
|
||||
break;
|
||||
@@ -896,7 +897,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
|
||||
if (passed) {
|
||||
gpr.FlushAll();
|
||||
out.Write(IROp::ExitToConst, out.AddConstant(inst.constant));
|
||||
out.Write(IROp::ExitToConst, 0, 0, 0, inst.constant);
|
||||
skipNextExitToConst = true;
|
||||
}
|
||||
break;
|
||||
@@ -918,7 +919,7 @@ bool PropagateConstants(const IRWriter &in, IRWriter &out, const IROptions &opts
|
||||
// Prefer ExitToConst to allow block linking.
|
||||
u32 dest = gpr.GetImm(inst.src1);
|
||||
gpr.FlushAll();
|
||||
out.Write(IROp::ExitToConst, out.AddConstant(dest));
|
||||
out.Write(IROp::ExitToConst, 0, 0, 0, dest);
|
||||
break;
|
||||
}
|
||||
gpr.FlushAll();
|
||||
@@ -2040,10 +2041,11 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
|
||||
} else if (inst.constant == 0xBF800000) {
|
||||
out.Write(IROp::Vec4Init, temp, (int)Vec4Init::AllMinusONE);
|
||||
} else {
|
||||
out.Write(IROp::SetConstF, temp, out.AddConstant(inst.constant));
|
||||
// NOTE: WriteSetConstantFloat doesn't work here since we actually want the bits.
|
||||
out.Write(IROp::SetConstF, temp, 0, 0, inst.constant);
|
||||
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
|
||||
}
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
|
||||
isVec4Dirty[inst.dest & ~3] = true;
|
||||
continue;
|
||||
}
|
||||
@@ -2054,7 +2056,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
|
||||
u8 blendMask = 1 << (inst.dest & 3);
|
||||
out.Write(IROp::FMovFromGPR, temp, inst.src1);
|
||||
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
|
||||
isVec4Dirty[inst.dest & ~3] = true;
|
||||
continue;
|
||||
}
|
||||
@@ -2065,7 +2067,7 @@ bool ReduceVec4Flush(const IRWriter &in, IRWriter &out, const IROptions &opts) {
|
||||
u8 blendMask = 1 << (inst.dest & 3);
|
||||
out.Write(inst.op, temp, inst.src1, inst.src2, inst.constant);
|
||||
out.Write(IROp::Vec4Shuffle, temp, temp, 0);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, blendMask);
|
||||
out.Write(IROp::Vec4Blend, inst.dest & ~3, inst.dest & ~3, temp, (u32)blendMask);
|
||||
isVec4Dirty[inst.dest & ~3] = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user