mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
arm64jit: Implement vuc2i.
This commit is contained in:
1 parent
a21a882add
commit
e03ae26d20
2 files changed
+23
-8
No files matched your search
@@ -631,11 +631,19 @@ void Arm64JitBackend::CompIR_VecPack(IRInst inst) {
|
||||
CONDITIONAL_DISABLE;
|
||||
|
||||
switch (inst.op) {
|
||||
case IROp::Vec2Unpack16To31:
|
||||
case IROp::Vec4Unpack8To32:
|
||||
case IROp::Vec2Unpack16To32:
|
||||
case IROp::Vec4DuplicateUpperBitsAndShift1:
|
||||
CompIR_Generic(inst);
|
||||
// This operation swizzles the high 8 bits and converts to a signed int.
|
||||
// It's always after Vec4Unpack8To32.
|
||||
// 000A000B000C000D -> AAAABBBBCCCCDDDD and then shift right one (to match INT_MAX.)
|
||||
regs_.Map(inst);
|
||||
// First, USHR+ORR to get 0A0A0B0B0C0C0D0D.
|
||||
fp_.USHR(32, EncodeRegToQuad(SCRATCHF1), regs_.FQ(inst.src1), 16);
|
||||
fp_.ORR(EncodeRegToQuad(SCRATCHF1), EncodeRegToQuad(SCRATCHF1), regs_.FQ(inst.src1));
|
||||
// Now again, but by 8.
|
||||
fp_.USHR(32, regs_.FQ(inst.dest), EncodeRegToQuad(SCRATCHF1), 8);
|
||||
fp_.ORR(regs_.FQ(inst.dest), regs_.FQ(inst.dest), EncodeRegToQuad(SCRATCHF1));
|
||||
// Finally, shift away the sign. The goal is to saturate 0xFF -> 0x7FFFFFFF.
|
||||
fp_.USHR(32, regs_.FQ(inst.dest), regs_.FQ(inst.dest), 1);
|
||||
break;
|
||||
|
||||
case IROp::Vec2Pack31To16:
|
||||
@@ -704,6 +712,12 @@ void Arm64JitBackend::CompIR_VecPack(IRInst inst) {
|
||||
}
|
||||
break;
|
||||
|
||||
case IROp::Vec2Unpack16To31:
|
||||
case IROp::Vec4Unpack8To32:
|
||||
case IROp::Vec2Unpack16To32:
|
||||
CompIR_Generic(inst);
|
||||
break;
|
||||
|
||||
default:
|
||||
INVALIDOP;
|
||||
break;
|
||||
|
||||
@@ -926,10 +926,8 @@ namespace MIPSInt
|
||||
switch ((op >> 16) & 3) {
|
||||
case 0: // vuc2i
|
||||
// Quad is the only option.
|
||||
// This operation is weird. This particular way of working matches hw but does not
|
||||
// seem quite sane.
|
||||
// I guess it's used for fixed-point math, and fills more bits to facilitate
|
||||
// conversion between 8-bit and 16-bit values. But then why not do it in vc2i?
|
||||
// This converts 8-bit unsigned to 31-bit signed, swizzling to saturate.
|
||||
// Similar to 5-bit to 8-bit color swizzling, but clamping to INT_MAX.
|
||||
{
|
||||
u32 value = s[0];
|
||||
for (int i = 0; i < 4; i++) {
|
||||
@@ -942,6 +940,8 @@ namespace MIPSInt
|
||||
|
||||
case 1: // vc2i
|
||||
// Quad is the only option
|
||||
// Unlike vuc2i, the source and destination are signed so there is no shift.
|
||||
// It lacks the swizzle because of negative values.
|
||||
{
|
||||
u32 value = s[0];
|
||||
d[0] = (value & 0xFF) << 24;
|
||||
@@ -953,6 +953,7 @@ namespace MIPSInt
|
||||
break;
|
||||
|
||||
case 2: // vus2i
|
||||
// Note: for some reason, this skips swizzle such that 0xFFFF -> 0x7FFF8000 unlike vuc2i.
|
||||
oz = V_Pair;
|
||||
switch (sz) {
|
||||
case V_Quad:
|
||||
|
||||
Reference in new issue
Block a user