arm64jit: Implement vuc2i.

This commit is contained in:
Unknown W. Brackets committed 2023-09-08 00:02:53 -07:00
1 parent a21a882add
commit e03ae26d20
2 files changed
+23 -8

No files matched your search

+18 -4
View File
@@ -631,11 +631,19 @@ void Arm64JitBackend::CompIR_VecPack(IRInst inst) {
CONDITIONAL_DISABLE;
switch (inst.op) {
case IROp::Vec2Unpack16To31:
case IROp::Vec4Unpack8To32:
case IROp::Vec2Unpack16To32:
case IROp::Vec4DuplicateUpperBitsAndShift1:
CompIR_Generic(inst);
// This operation swizzles the high 8 bits and converts to a signed int.
// It's always after Vec4Unpack8To32.
// 000A000B000C000D -> AAAABBBBCCCCDDDD and then shift right one (to match INT_MAX.)
regs_.Map(inst);
// First, USHR+ORR to get 0A0A0B0B0C0C0D0D.
fp_.USHR(32, EncodeRegToQuad(SCRATCHF1), regs_.FQ(inst.src1), 16);
fp_.ORR(EncodeRegToQuad(SCRATCHF1), EncodeRegToQuad(SCRATCHF1), regs_.FQ(inst.src1));
// Now again, but by 8.
fp_.USHR(32, regs_.FQ(inst.dest), EncodeRegToQuad(SCRATCHF1), 8);
fp_.ORR(regs_.FQ(inst.dest), regs_.FQ(inst.dest), EncodeRegToQuad(SCRATCHF1));
// Finally, shift away the sign. The goal is to saturate 0xFF -> 0x7FFFFFFF.
fp_.USHR(32, regs_.FQ(inst.dest), regs_.FQ(inst.dest), 1);
break;
case IROp::Vec2Pack31To16:
@@ -704,6 +712,12 @@ void Arm64JitBackend::CompIR_VecPack(IRInst inst) {
}
break;
case IROp::Vec2Unpack16To31:
case IROp::Vec4Unpack8To32:
case IROp::Vec2Unpack16To32:
CompIR_Generic(inst);
break;
default:
INVALIDOP;
break;
+5 -4
View File
@@ -926,10 +926,8 @@ namespace MIPSInt
switch ((op >> 16) & 3) {
case 0: // vuc2i
// Quad is the only option.
// This operation is weird. This particular way of working matches hw but does not
// seem quite sane.
// I guess it's used for fixed-point math, and fills more bits to facilitate
// conversion between 8-bit and 16-bit values. But then why not do it in vc2i?
// This converts 8-bit unsigned to 31-bit signed, swizzling to saturate.
// Similar to 5-bit to 8-bit color swizzling, but clamping to INT_MAX.
{
u32 value = s[0];
for (int i = 0; i < 4; i++) {
@@ -942,6 +940,8 @@ namespace MIPSInt
case 1: // vc2i
// Quad is the only option
// Unlike vuc2i, the source and destination are signed so there is no shift.
// It lacks the swizzle because of negative values.
{
u32 value = s[0];
d[0] = (value & 0xFF) << 24;
@@ -953,6 +953,7 @@ namespace MIPSInt
break;
case 2: // vus2i
// Note: for some reason, this skips swizzle such that 0xFFFF -> 0x7FFF8000 unlike vuc2i.
oz = V_Pair;
switch (sz) {
case V_Quad: