Merge pull request #22314 from hrydgard/cross-build-riscv64

Loongarch64 and RiscV: Try to make these viable by adding cross builds
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-09-19 14:21:25 -06:00
commit f293b10fb2
15 files changed
+756 -151

No files matched your search

+12 -5
View File
@@ -288,6 +288,13 @@ jobs:
args: ./b.sh --loongarch64 PPSSPPHeadless
id: loongarch64
- os: ubuntu-26.04
extra: riscv64
cc: gcc
cxx: g++
args: ./b.sh --riscv64 PPSSPPHeadless
id: riscv64
- os: macos-latest
extra: test
cc: clang
@@ -327,7 +334,7 @@ jobs:
ndk-version: r29
- name: Install Linux dependencies
if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64'
if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64' && matrix.extra != 'riscv64'
run: |
sudo apt-get update -y -qq
sudo apt-get install libsdl3-dev libgl1-mesa-dev libglu1-mesa-dev libsdl3-ttf-dev libfontconfig1-dev libcurl4-openssl-dev
@@ -336,12 +343,12 @@ jobs:
if: runner.os == 'macOS' && matrix.id == 'macos'
run: brew install sdl3 sdl3_ttf
- name: Install loongarch64 cross-compilation dependencies
if: matrix.extra == 'loongarch64'
- name: Install cross-compilation dependencies
if: matrix.extra == 'loongarch64' || matrix.extra == 'riscv64'
run: |
sudo apt-get update -y -qq
sudo apt-get install -y gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu libgl-dev libglu1-mesa-dev
sudo cmake/scripts/setup-loongarch64-cross.sh
sudo apt-get install -y gcc-14-${{ matrix.extra }}-linux-gnu g++-14-${{ matrix.extra }}-linux-gnu binutils-${{ matrix.extra }}-linux-gnu libgl-dev libglu1-mesa-dev
sudo cmake/scripts/setup-cross.sh ${{ matrix.extra }}
- name: Install iOS dependencies
if: matrix.id == 'ios'
+9 -4
View File
@@ -184,6 +184,8 @@ option(USE_VULKAN_DISPLAY_KHR "Enable or disable full screen display of Vulkan"
# :: Frontends
option(MOBILE_DEVICE "Set to ON when targeting a mobile device" ${MOBILE_DEVICE})
option(HEADLESS "Set to OFF to not generate the PPSSPPHeadless target" ${HEADLESS})
# Set by b.sh for the loongarch64/riscv64 cross builds, whose sysroots have no SDL3.
option(HEADLESS_CROSS "Set to ON for a headless-only cross build (no SDL frontend)" ${HEADLESS_CROSS})
option(ATLAS_TOOL "Set to OFF to not generate the ATLAS_TOOL target" ${ATLAS_TOOL})
option(UNITTEST "Set to ON to generate the unittest target" ${UNITTEST})
option(SIMULATOR "Set to ON when targeting an x86 simulator of an ARM platform" ${SIMULATOR})
@@ -331,7 +333,7 @@ set(SDL_LIB_TARGET "")
set(SDL_TTF_LIB_TARGET "")
set(FONTCONFIG_LIB_TARGET "")
if(NOT LIBRETRO AND NOT IOS AND NOT LOONGARCH64_DEVICE AND NOT ANDROID)
if(NOT LIBRETRO AND NOT IOS AND NOT HEADLESS_CROSS AND NOT ANDROID)
find_package(SDL3 QUIET)
if(NOT SDL3_FOUND)
if(CMAKE_SYSTEM_NAME STREQUAL "Linux" AND (EXISTS "/usr/bin/apt" OR EXISTS "/usr/bin/apt-get" OR EXISTS "/etc/debian_version"))
@@ -1001,9 +1003,11 @@ elseif(WIN32)
# Don't care about SDL.
set(TargetBin PPSSPPWindows)
elseif(LIBRETRO)
elseif(LOONGARCH64_DEVICE)
# Cross-compiling for loongarch64 currently only builds PPSSPPHeadless, and there's
# no SDL3 available for that target, so skip the desktop SDL frontend entirely.
elseif(HEADLESS_CROSS)
# The loongarch64/riscv64 cross builds only build PPSSPPHeadless, and there's no
# SDL3 available in those sysroots, so skip the desktop SDL frontend entirely.
# Note this is deliberately not keyed off the target architecture: a native build
# on such a machine can have SDL3 and should still get the normal frontend.
else()
if(GOLD)
set(TargetBin PPSSPPGold)
@@ -1362,6 +1366,7 @@ if(UNITTEST)
unittest/TestShaderGenerators.cpp
unittest/TestArmEmitter.cpp
unittest/TestArm64Emitter.cpp
unittest/TestCrossSIMD.cpp
unittest/TestIRPassSimplify.cpp
unittest/TestX64Emitter.cpp
unittest/TestVertexJit.cpp
+2 -1
View File
@@ -166,7 +166,8 @@ public:
QuickCallFunction((const u8 *)func, scratchreg);
}
void QuickCallFunctionR(const u8 *func, LoongArch64Reg arg, LoongArch64Reg scratchreg = R_RA) {
MOVE(LoongArch64Reg::X4, arg);
// The first integer argument goes in a0, which is R4 - X4 is an LASX vector register.
MOVE(LoongArch64Reg::R4, arg);
QuickJump(scratchreg, R_RA, func);
}
template <typename T>
+54 -10
View File
@@ -6,6 +6,7 @@
#pragma once
#include <cmath>
#include <cstring>
#include "Common/Math/SIMDHeaders.h"
@@ -1187,8 +1188,11 @@ struct Vec4F32 {
}
void StoreConvertToU8(uint8_t *dest) {
__m128i ivalue32 = __lsx_vftintrz_w_s(v);
__m128i ivalue16 = __lsx_vssrlrni_hu_w(ivalue32, ivalue32, 0);
__m128i ivalue8 = __lsx_vssrlrni_bu_h(ivalue16, ivalue16, 0);
// Narrow signed->signed first, then signed->unsigned, matching the packs/packus pair the
// SSE version uses. The logical (unsigned) narrowing shifts turn a negative value into a
// huge unsigned one, which saturates to 255 instead of clamping to 0.
__m128i ivalue16 = __lsx_vssrani_h_w(ivalue32, ivalue32, 0);
__m128i ivalue8 = __lsx_vssrani_bu_h(ivalue16, ivalue16, 0);
uint32_t value = __lsx_vpickve2gr_wu(ivalue8, 0);
memcpy(dest, &value, sizeof(uint32_t));
}
@@ -1198,10 +1202,10 @@ struct Vec4F32 {
// index is a compile-time constant parameter to the intrinsic.
int ival;
switch (index) {
case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0);
case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1);
case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2);
default: ival = __lsx_vpickve2gr_w((__m128i)v, 3);
case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); break;
case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); break;
case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); break;
default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); break;
}
float fval;
memcpy(&fval, &ival, sizeof(float));
@@ -1671,15 +1675,38 @@ struct Vec4F32 {
return temp;
}
static Vec4F32 LoadConvertU8(const uint8_t *src) { // Note: will load 8 bytes, not 4. Only the first 4 bytes will be used.
Vec4F32 temp;
for (int i = 0; i < 4; i++) {
temp.v[i] = (float)src[i];
}
return temp;
}
// Truncates towards zero and saturates to 0-255, like the packing instructions the other
// implementations use.
void StoreConvertToU8(uint8_t *dest) {
for (int i = 0; i < 4; i++) {
int value = (int)v[i];
if (value < 0) value = 0;
if (value > 255) value = 255;
dest[i] = (uint8_t)value;
}
}
static Vec4F32 LoadF24x3_One(const uint32_t *src) {
uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, 0 };
constexpr uint32_t kOneF32Bits = 0x3F800000;
uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, kOneF32Bits };
Vec4F32 temp;
memcpy(temp.v, shifted, sizeof(temp.v));
return temp;
}
static Vec4F32 LoadF24x4(const uint32_t *src) {
return LoadR24x3_One(src);
uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, src[3] << 8 };
Vec4F32 temp;
memcpy(temp.v, shifted, sizeof(temp.v));
return temp;
}
static Vec4F32 FromVec4S32(Vec4S32 src) {
@@ -1695,7 +1722,7 @@ struct Vec4F32 {
Vec4F32 ZeroNaNs() const {
Vec4F32 temp;
for (int i = 0; i < 4; i++) {
temp.v[i] = isnan(v[i]) ? 0.0f : v[i];
temp.v[i] = std::isnan(v[i]) ? 0.0f : v[i];
}
return temp;
}
@@ -1703,7 +1730,7 @@ struct Vec4F32 {
Vec4F32 CleanNaNInfs() {
Vec4F32 temp;
for (int i = 0; i < 4; i++) {
temp.v[i] = (isnan(v[i]) || isinf(v[i])) ? 0.0f : v[i];
temp.v[i] = (std::isnan(v[i]) || std::isinf(v[i])) ? 0.0f : v[i];
}
return temp;
}
@@ -1798,6 +1825,10 @@ struct Vec4F32 {
return Vec4F32{ { v[0], v[1], v[2], 1.0f } };
}
Vec4F32 WithLane3From(Vec4F32 other) const {
return Vec4F32{ { v[0], v[1], v[2], other.v[3] } };
}
Vec4S32 CompareEq(Vec4F32 other) const {
Vec4S32 temp;
for (int i = 0; i < 4; i++) {
@@ -1857,6 +1888,15 @@ struct Vec4F32 {
return Vec4F32{{ v[3], v[3], v[3], v[3] }};
}
// Loads four rows of four floats and transposes them into columns.
static void LoadTranspose(const float *src, Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) {
col0 = Vec4F32::Load(src);
col1 = Vec4F32::Load(src + 4);
col2 = Vec4F32::Load(src + 8);
col3 = Vec4F32::Load(src + 12);
Transpose(col0, col1, col2, col3);
}
// In-place transpose.
static void Transpose(Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) {
float m[16];
@@ -1919,6 +1959,10 @@ inline bool AllCompareBitsSet(Vec4S32 value) {
return true;
}
inline bool AnyCompareBitsSet(Vec4S32 value) {
return value.v[0] != 0 || value.v[1] != 0 || value.v[2] != 0 || value.v[3] != 0;
}
struct Vec4U16 {
uint16_t v[4]; // 64 bits.
+3 -3
View File
@@ -842,7 +842,7 @@ static inline u32 EncodeFVF(RiscVReg vd, RiscVReg rs1, RiscVReg vs2, VUseMask vm
static inline u16 EncodeCR(Opcode16 op, RiscVReg rs2, RiscVReg rd, Funct4 funct4) {
_assert_msg_(SupportsCompressed(), "Compressed instructions unsupported");
return (u16)op | ((u16)rs2 << 2) | ((u16)rd << 7) | ((u16)funct4 << 12);
return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)DecodeReg(rd) << 7) | ((u16)funct4 << 12);
}
static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) {
@@ -850,13 +850,13 @@ static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) {
_assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6);
u16 imm4_0 = ImmBits16(uimm6, 0, 5);
u16 imm5 = ImmBit16(uimm6, 5);
return (u16)op | (imm4_0 << 2) | ((u16)rd << 7) | (imm5 << 12) | ((u16)funct3 << 13);
return (u16)op | (imm4_0 << 2) | ((u16)DecodeReg(rd) << 7) | (imm5 << 12) | ((u16)funct3 << 13);
}
static inline u16 EncodeCSS(Opcode16 op, RiscVReg rs2, u8 uimm6, Funct3 funct3) {
_assert_msg_(SupportsCompressed(), "Compressed instructions unsupported");
_assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6);
return (u16)op | ((u16)rs2 << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13);
return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13);
}
static inline u16 EncodeCIW(Opcode16 op, RiscCReg rd, u8 uimm8, Funct3 funct3) {
+1
View File
@@ -1051,6 +1051,7 @@ ifeq ($(UNITTEST),1)
$(SRC)/unittest/TestX64Emitter.cpp \
$(SRC)/unittest/TestRiscVEmitter.cpp \
$(SRC)/unittest/TestLoongArch64Emitter.cpp \
$(SRC)/unittest/TestCrossSIMD.cpp \
$(SRC)/unittest/TestIRPassSimplify.cpp \
$(SRC)/unittest/TestShaderGenerators.cpp \
$(SRC)/unittest/TestSoftwareGPUJit.cpp \
+16 -10
View File
@@ -31,9 +31,14 @@ do
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/raspberry.armv8.cmake ${CMAKE_ARGS}"
;;
--loongarch64)
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}"
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}"
TARGET_OS=loongarch64
LOONGARCH64_BUILD=1
CROSS_STUB_CC=loongarch64-linux-gnu-gcc-14
;;
--riscv64)
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/riscv64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}"
TARGET_OS=riscv64
CROSS_STUB_CC=riscv64-linux-gnu-gcc-14
;;
--android) CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=android/android.toolchain.cmake ${CMAKE_ARGS}"
TARGET_OS=Android
@@ -122,28 +127,29 @@ echo Building with $CORES_COUNT threads
mkdir -p ${BUILD_DIR}
# For loongarch64 cross-compilation, build a comprehensive GL/GLX stub into
# For the headless cross targets, build a comprehensive GL/GLX stub into
# <build>/stublibs/libGL.so so GLEW's static archive can resolve its symbols
# via PLT entries (a direct B26 branch to address 0 overflows on LoongArch).
# via PLT entries (a direct branch to address 0 overflows the relocation).
# This stub is always (re)generated to pick up any new needed symbols.
if [ ! -z "$LOONGARCH64_BUILD" ]; then
if [ ! -z "$CROSS_STUB_CC" ]; then
STUB_DIR=${BUILD_DIR}/stublibs
STUB_GL=${STUB_DIR}/libGL.so
mkdir -p "${STUB_DIR}"
STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c)
echo "/* Loongarch64 GL/GLX stub - cross-compilation only */" > "$STUB_C"
echo "/* ${TARGET_OS} GL/GLX stub - cross-compilation only */" > "$STUB_C"
# Collect all T (exported) symbols from GL/GLX libs, deduplicate, emit stubs
HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null)
{
for lib in /usr/lib/x86_64-linux-gnu/libGL.so.1 \
/usr/lib/x86_64-linux-gnu/libGLX.so.0 \
/usr/lib/x86_64-linux-gnu/libGLdispatch.so.0; do
for lib in /usr/lib/$HOST_MULTIARCH/libGL.so.1 \
/usr/lib/$HOST_MULTIARCH/libGLX.so.0 \
/usr/lib/$HOST_MULTIARCH/libGLdispatch.so.0; do
[ -f "$lib" ] && nm -D "$lib" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print $3 }'
done
# Always include the minimal GLX symbols GLEW directly references
printf '%s\n' glXGetProcAddressARB glXGetClientString glXQueryVersion \
glBindTexture glGetString glGetIntegerv
} | sort -u | awk '{ print "void "$1"(void){}" }' >> "$STUB_C"
loongarch64-linux-gnu-gcc-14 -shared -fPIC -Wno-implicit-function-declaration \
$CROSS_STUB_CC -shared -fPIC -Wno-implicit-function-declaration \
-o "${STUB_GL}" "$STUB_C"
rm "$STUB_C"
fi
+16 -12
View File
@@ -17,8 +17,8 @@ set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE)
# Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so
# rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have.
# A stub libGL.so + GL headers must be present in the sysroot; they are placed
# there by cmake/scripts/setup-loongarch64-cross.sh (run once with sudo, or
# automatically by the CI install step).
# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically
# by the CI install step).
set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE)
# b.sh also creates a stub in <build-dir>/stublibs/ as a local fallback so the
@@ -27,14 +27,18 @@ set(CMAKE_EXE_LINKER_FLAGS
"${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs"
CACHE STRING "" FORCE)
# Allow CMake's compile-check programs to run via QEMU.
# The /lib64 symlink is created by the setup script; without it, pass -L
# explicitly so QEMU can find the loongarch64 dynamic linker.
if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1")
set(CMAKE_CROSSCOMPILING_EMULATOR "qemu-loongarch64-static"
CACHE STRING "" FORCE)
else()
set(CMAKE_CROSSCOMPILING_EMULATOR
"qemu-loongarch64-static;-L;/usr/loongarch64-linux-gnu"
CACHE STRING "" FORCE)
# Allow CMake's compile-check programs to run via QEMU, if it's installed.
# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic
# ones in qemu-user; either will do, so take whichever is present.
find_program(QEMU_LOONGARCH64 NAMES qemu-loongarch64-static qemu-loongarch64)
if(QEMU_LOONGARCH64)
# The /lib64 symlink is created by the setup script; without it, pass -L
# explicitly so QEMU can find the loongarch64 dynamic linker.
if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1")
set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_LOONGARCH64}" CACHE STRING "" FORCE)
else()
set(CMAKE_CROSSCOMPILING_EMULATOR
"${QEMU_LOONGARCH64};-L;/usr/loongarch64-linux-gnu"
CACHE STRING "" FORCE)
endif()
endif()
+44
View File
@@ -0,0 +1,44 @@
set(CMAKE_SYSTEM_NAME Linux)
set(CMAKE_SYSTEM_PROCESSOR riscv64)
set(CMAKE_C_COMPILER riscv64-linux-gnu-gcc-14)
set(CMAKE_CXX_COMPILER riscv64-linux-gnu-g++-14)
set(CMAKE_ASM_COMPILER riscv64-linux-gnu-gcc-14)
set(CMAKE_FIND_ROOT_PATH /usr/riscv64-linux-gnu)
set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER)
set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY)
set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY)
set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY)
# No X11 in the sysroot; disable the X11 Vulkan path to avoid include-dir errors.
set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE)
# Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so
# rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have.
# A stub libGL.so + GL headers must be present in the sysroot; they are placed
# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically
# by the CI install step).
set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE)
# b.sh also creates a stub in <build-dir>/stublibs/ as a local fallback so the
# linker can resolve GL/GLX calls from GLEW's static archive via PLT entries.
set(CMAKE_EXE_LINKER_FLAGS
"${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs"
CACHE STRING "" FORCE)
# Allow CMake's compile-check programs to run via QEMU, if it's installed.
# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic
# ones in qemu-user; either will do, so take whichever is present.
find_program(QEMU_RISCV64 NAMES qemu-riscv64-static qemu-riscv64)
if(QEMU_RISCV64)
# The /lib64 symlink is created by the setup script; without it, pass -L
# explicitly so QEMU can find the riscv64 dynamic linker.
if(EXISTS "/lib64/ld-linux-riscv64-lp64d.so.1")
set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_RISCV64}" CACHE STRING "" FORCE)
else()
set(CMAKE_CROSSCOMPILING_EMULATOR
"${QEMU_RISCV64};-L;/usr/riscv64-linux-gnu"
CACHE STRING "" FORCE)
endif()
endif()
@@ -1,15 +1,34 @@
#!/bin/bash
# Sets up the loongarch64 cross-compilation environment.
# Run once with sudo. After this, ./b.sh --loongarch64 works on a fresh checkout.
# Sets up a cross-compilation environment for one of the headless-only targets.
# Run once with sudo. After this, ./b.sh --<target> works on a fresh checkout.
#
# Usage: sudo cmake/scripts/setup-cross.sh <loongarch64|riscv64>
set -e
SYSROOT=/usr/loongarch64-linux-gnu
CROSS_GCC=loongarch64-linux-gnu-gcc-14
TARGET="$1"
case "$TARGET" in
loongarch64)
TRIPLE=loongarch64-linux-gnu
LD_SO=ld-linux-loongarch-lp64d.so.1
;;
riscv64)
TRIPLE=riscv64-linux-gnu
LD_SO=ld-linux-riscv64-lp64d.so.1
;;
*)
echo "Usage: sudo $0 <loongarch64|riscv64>"
exit 1
;;
esac
SYSROOT=/usr/$TRIPLE
CROSS_GCC=$TRIPLE-gcc-14
if ! command -v $CROSS_GCC &>/dev/null; then
echo "Error: $CROSS_GCC not found. Install it first:"
echo " sudo apt install gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu"
echo " sudo apt install gcc-14-$TRIPLE g++-14-$TRIPLE binutils-$TRIPLE"
exit 1
fi
@@ -18,6 +37,8 @@ if [ "$(id -u)" -ne 0 ]; then
exit 1
fi
HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null)
# ── OpenGL headers ───────────────────────────────────────────────────────────
# GL headers are platform-independent C headers; copy from the host.
echo "Installing GL headers into sysroot..."
@@ -27,22 +48,22 @@ for h in gl.h glext.h glcorearb.h glu.h; do
done
# ── GL stub library ──────────────────────────────────────────────────────────
# Generate a loongarch64 libGL.so that exports all symbols from the host
# Generate a libGL.so for the target that exports all symbols from the host
# libGL so that GLEW's static archive can be linked via PLT entries (a direct
# B26 branch to address 0 would overflow on LoongArch).
# branch to address 0 would overflow the relocation on these targets).
echo "Generating libGL.so stub..."
HOST_GL=""
for candidate in \
/usr/lib/x86_64-linux-gnu/libGL.so.1 \
/usr/lib/x86_64-linux-gnu/libGL.so \
/usr/lib/$HOST_MULTIARCH/libGL.so.1 \
/usr/lib/$HOST_MULTIARCH/libGL.so \
/usr/lib/libGL.so.1 ; do
[ -f "$candidate" ] && HOST_GL="$candidate" && break
done
STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c)
echo "/* Loongarch64 GL/GLX stub - for cross-compilation only */" > "$STUB_C"
echo "void __loongarch64_gl_placeholder(void) {}" >> "$STUB_C"
echo "/* $TARGET GL/GLX stub - for cross-compilation only */" > "$STUB_C"
echo "void __cross_gl_placeholder(void) {}" >> "$STUB_C"
if [ -n "$HOST_GL" ]; then
nm -D "$HOST_GL" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print "void "$3"(void){}" }' >> "$STUB_C"
@@ -60,16 +81,16 @@ rm "$STUB_C"
echo " Installed to $SYSROOT/lib/libGL.so"
# ── Dynamic linker symlink ───────────────────────────────────────────────────
# Without this, binfmt_misc can't find the loongarch64 dynamic linker.
LD_LINUX=$SYSROOT/lib/ld-linux-loongarch-lp64d.so.1
# Without this, binfmt_misc can't find the target's dynamic linker.
LD_LINUX=$SYSROOT/lib/$LD_SO
if [ -f "$LD_LINUX" ]; then
mkdir -p /lib64
ln -sf "$LD_LINUX" /lib64/ld-linux-loongarch-lp64d.so.1
echo "Created /lib64/ld-linux-loongarch-lp64d.so.1 -> $LD_LINUX"
ln -sf "$LD_LINUX" /lib64/$LD_SO
echo "Created /lib64/$LD_SO -> $LD_LINUX"
fi
echo ""
echo "Setup complete."
echo "Build: ./b.sh --loongarch64"
echo "Run: qemu-loongarch64-static -L $SYSROOT <binary>"
echo " (after setup the /lib64 symlink lets you run loongarch64 binaries directly)"
echo "Build: ./b.sh --$TARGET"
echo "Run: qemu-$TARGET -L $SYSROOT <binary>"
echo " (after setup the /lib64 symlink lets you run $TARGET binaries directly)"
+6 -6
View File
@@ -289,10 +289,10 @@ static GraphicsContext *CreateGraphicsContext(GPUCore gpuCore, std::string **dev
default:
return nullptr;
}
#elif PPSSPP_ARCH(LOONGARCH64)
// The loongarch64 cross-compilation toolchain has no SDL3 packages available (see the
// LOONGARCH64_DEVICE branch in CMakeLists.txt), so this build is compile-tested only and
// never actually needs to create a graphics context at runtime.
#elif PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64)
// The loongarch64 and riscv64 cross-compilation sysroots have no SDL3 (see the HEADLESS_CROSS
// branch in CMakeLists.txt). These builds still run fine under qemu with --graphics=software,
// which needs no graphics context.
*deviceSetting = nullptr;
return nullptr;
#elif PPSSPP_PLATFORM(ANDROID)
@@ -904,7 +904,7 @@ int main(int argc, const char* argv[]) {
// We don't bother with a window.
graphicsContext = new NullGraphicsContext();
} else {
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64)
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64)
fprintf(stderr, "Headless graphics context creation is not supported on this platform.\n");
return 1;
#else
@@ -1134,7 +1134,7 @@ int main(int argc, const char* argv[]) {
ShutdownWebServer();
}
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64)
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64)
// ... see above
#else
if (window) {
+551
View File
@@ -0,0 +1,551 @@
// Copyright (c) 2012- PPSSPP Project.
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU General Public License as published by
// the Free Software Foundation, version 2.0 or later versions.
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU General Public License 2.0 for more details.
// A copy of the GPL 2.0 should have been included with the program.
// If not, see http://www.gnu.org/licenses/
// Official git repository and contact information can be found at
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
// Tests for Common/Math/CrossSIMD.h.
//
// CrossSIMD has four independent implementations - SSE2, NEON, LoongArch LSX and a plain scalar
// fallback - and only one of them is compiled on any given machine. So the point of these tests is
// to check whatever got compiled against results worked out by hand, rather than against each other;
// that way every architecture we build for verifies its own copy. Running the unit tests under
// qemu covers the ones we have no hardware for.
//
// Only functions that all four implementations provide are tested here, otherwise this wouldn't
// build everywhere.
#include "ppsspp_config.h"
#include <cmath>
#include <cstdio>
#include <cstring>
#include "Common/Common.h"
#include "Common/CommonTypes.h"
#include "Common/Math/CrossSIMD.h"
#include "unittest/UnitTest.h"
static bool CompareFloats(const float *values, const float *known_good, int count, int line) {
int wrongCount = 0;
for (int i = 0; i < count; i++) {
if (values[i] != known_good[i]) {
wrongCount++;
}
}
if (wrongCount > 0) {
for (int i = 0; i < count; i++) {
bool wrong = values[i] != known_good[i];
printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : "");
}
printf("At TestCrossSIMD.cpp:%d: %d / %d were wrong\n", line, wrongCount, count);
return false;
}
return true;
}
// For the approximate operations (reciprocals), which are allowed to differ between architectures.
static bool CompareFloatsApprox(const float *values, const float *known_good, int count, float tolerance, int line) {
for (int i = 0; i < count; i++) {
const float diff = fabsf(values[i] - known_good[i]);
const float scale = fabsf(known_good[i]) > 1.0f ? fabsf(known_good[i]) : 1.0f;
if (diff / scale > tolerance) {
printf("At TestCrossSIMD.cpp:%d: lane %d: %0.6f vs %0.6f\n", line, i, values[i], known_good[i]);
return false;
}
}
return true;
}
static bool TestVec4S32() {
const int a_values[4] = { 3, -7, 0x4000, -1 };
const int b_values[4] = { 5, 11, -2, 0x7FFF };
Vec4S32 a = Vec4S32::Load(a_values);
Vec4S32 b = Vec4S32::Load(b_values);
int result[4];
(a + b).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], a_values[i] + b_values[i]);
}
(a - b).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], a_values[i] - b_values[i]);
}
a.Mul(b).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], a_values[i] * b_values[i]);
}
// Mul16 only promises correct results when both sides fit in 16 bits.
const int small_a[4] = { 3, -7, 300, -1 };
const int small_b[4] = { 5, 11, -2, 1000 };
Vec4S32 sa = Vec4S32::Load(small_a);
Vec4S32 sb = Vec4S32::Load(small_b);
sa.Mul16(sb).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], small_a[i] * small_b[i]);
}
Vec4S32::Splat(-5).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], -5);
}
Vec4S32::Zero().Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], 0);
}
a.Shl<2>().Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], a_values[i] << 2);
}
// operator[] should agree with Store.
a.Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(a[i], result[i]);
}
// Min16/Max16 likewise operate on 16-bit lanes.
sa.Min16(sb).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], small_a[i] < small_b[i] ? small_a[i] : small_b[i]);
}
sa.Max16(sb).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_INT(result[i], small_a[i] > small_b[i] ? small_a[i] : small_b[i]);
}
const int sext_values[4] = { 0x0000FFFF, 0x00007FFF, 0x00008000, 0x00000001 };
Vec4S32::Load(sext_values).SignExtend16().Store(result);
EXPECT_EQ_INT(result[0], -1);
EXPECT_EQ_INT(result[1], 32767);
EXPECT_EQ_INT(result[2], -32768);
EXPECT_EQ_INT(result[3], 1);
return true;
}
static bool TestVec4S32Compares() {
const int a_values[4] = { 1, 2, 3, 4 };
const int b_values[4] = { 1, 5, 0, 4 };
Vec4S32 a = Vec4S32::Load(a_values);
Vec4S32 b = Vec4S32::Load(b_values);
int result[4];
// Compares produce all-ones or all-zeros per lane.
a.CompareEq(b).Store(result);
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[1], 0u);
EXPECT_EQ_HEX((u32)result[2], 0u);
EXPECT_EQ_HEX((u32)result[3], 0xFFFFFFFFu);
a.CompareLt(b).Store(result);
EXPECT_EQ_HEX((u32)result[0], 0u);
EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[2], 0u);
EXPECT_EQ_HEX((u32)result[3], 0u);
a.CompareGt(b).Store(result);
EXPECT_EQ_HEX((u32)result[0], 0u);
EXPECT_EQ_HEX((u32)result[1], 0u);
EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[3], 0u);
// AllCompareBitsSet / AnyCompareBitsSet over those masks.
EXPECT_FALSE(AllCompareBitsSet(a.CompareEq(b)));
EXPECT_TRUE(AnyCompareBitsSet(a.CompareEq(b)));
EXPECT_TRUE(AllCompareBitsSet(a.CompareEq(a)));
EXPECT_FALSE(AnyCompareBitsSet(a.CompareLt(a)));
return true;
}
static bool TestVec4F32Arith() {
const float a_values[4] = { 1.0f, -2.0f, 3.5f, 0.25f };
const float b_values[4] = { 4.0f, 8.0f, -0.5f, 2.0f };
Vec4F32 a = Vec4F32::Load(a_values);
Vec4F32 b = Vec4F32::Load(b_values);
float result[4];
float expected[4];
(a + b).Store(result);
for (int i = 0; i < 4; i++) expected[i] = a_values[i] + b_values[i];
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
(a - b).Store(result);
for (int i = 0; i < 4; i++) expected[i] = a_values[i] - b_values[i];
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
(a * b).Store(result);
for (int i = 0; i < 4; i++) expected[i] = a_values[i] * b_values[i];
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
a.Mul(2.0f).Store(result);
for (int i = 0; i < 4; i++) expected[i] = a_values[i] * 2.0f;
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
a.Min(b).Store(result);
for (int i = 0; i < 4; i++) expected[i] = a_values[i] < b_values[i] ? a_values[i] : b_values[i];
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
a.Max(b).Store(result);
for (int i = 0; i < 4; i++) expected[i] = a_values[i] > b_values[i] ? a_values[i] : b_values[i];
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
a.Clamp(0.0f, 3.0f).Store(result);
static const float known_clamp[4] = { 1.0f, 0.0f, 3.0f, 0.25f };
if (!CompareFloats(result, known_clamp, 4, __LINE__)) return false;
// Recip is allowed to be approximate on some architectures, RecipApprox definitely is.
b.Recip().Store(result);
for (int i = 0; i < 4; i++) expected[i] = 1.0f / b_values[i];
if (!CompareFloatsApprox(result, expected, 4, 0.001f, __LINE__)) return false;
b.RecipApprox().Store(result);
if (!CompareFloatsApprox(result, expected, 4, 0.01f, __LINE__)) return false;
Vec4F32::Splat(1.5f).Store(result);
static const float known_splat[4] = { 1.5f, 1.5f, 1.5f, 1.5f };
if (!CompareFloats(result, known_splat, 4, __LINE__)) return false;
Vec4F32::Zero().Store(result);
static const float known_zero[4] = { 0.0f, 0.0f, 0.0f, 0.0f };
if (!CompareFloats(result, known_zero, 4, __LINE__)) return false;
// Dot products. Dot3 ignores lane 3.
EXPECT_EQ_FLOAT(a.Dot3(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2]);
EXPECT_EQ_FLOAT(a.Dot4(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2] + a_values[3] * b_values[3]);
// operator[] and GetLane<> should agree with Store.
a.Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(a[i], result[i]);
}
EXPECT_EQ_FLOAT(a.GetLane<0>(), a_values[0]);
EXPECT_EQ_FLOAT(a.GetLane<3>(), a_values[3]);
return true;
}
static bool TestVec4F32Compares() {
const float a_values[4] = { 1.0f, 2.0f, 3.0f, 4.0f };
const float b_values[4] = { 1.0f, 5.0f, 0.0f, 4.0f };
Vec4F32 a = Vec4F32::Load(a_values);
Vec4F32 b = Vec4F32::Load(b_values);
int result[4];
a.CompareEq(b).Store(result);
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[1], 0u);
a.CompareLt(b).Store(result);
EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[2], 0u);
a.CompareGt(b).Store(result);
EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[1], 0u);
a.CompareLe(b).Store(result);
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[2], 0u);
a.CompareGe(b).Store(result);
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
EXPECT_EQ_HEX((u32)result[1], 0u);
return true;
}
static bool TestVec4F32Lanes() {
const float values[4] = { 1.0f, 2.0f, 3.0f, 4.0f };
const float other_values[4] = { 9.0f, 9.0f, 9.0f, 42.0f };
Vec4F32 v = Vec4F32::Load(values);
Vec4F32 other = Vec4F32::Load(other_values);
float result[4];
v.WithLane3Zero().Store(result);
static const float known_zero3[4] = { 1.0f, 2.0f, 3.0f, 0.0f };
if (!CompareFloats(result, known_zero3, 4, __LINE__)) return false;
v.WithLane3One().Store(result);
static const float known_one3[4] = { 1.0f, 2.0f, 3.0f, 1.0f };
if (!CompareFloats(result, known_one3, 4, __LINE__)) return false;
v.WithLane3From(other).Store(result);
static const float known_from3[4] = { 1.0f, 2.0f, 3.0f, 42.0f };
if (!CompareFloats(result, known_from3, 4, __LINE__)) return false;
v.ShuffleXXXX().Store(result);
static const float known_xxxx[4] = { 1.0f, 1.0f, 1.0f, 1.0f };
if (!CompareFloats(result, known_xxxx, 4, __LINE__)) return false;
v.ShuffleYYYY().Store(result);
static const float known_yyyy[4] = { 2.0f, 2.0f, 2.0f, 2.0f };
if (!CompareFloats(result, known_yyyy, 4, __LINE__)) return false;
v.ShuffleZZZZ().Store(result);
static const float known_zzzz[4] = { 3.0f, 3.0f, 3.0f, 3.0f };
if (!CompareFloats(result, known_zzzz, 4, __LINE__)) return false;
v.ShuffleWWWW().Store(result);
static const float known_wwww[4] = { 4.0f, 4.0f, 4.0f, 4.0f };
if (!CompareFloats(result, known_wwww, 4, __LINE__)) return false;
v.ShuffleXXYY().Store(result);
static const float known_xxyy[4] = { 1.0f, 1.0f, 2.0f, 2.0f };
if (!CompareFloats(result, known_xxyy, 4, __LINE__)) return false;
v.ShuffleZZWW().Store(result);
static const float known_zzww[4] = { 3.0f, 3.0f, 4.0f, 4.0f };
if (!CompareFloats(result, known_zzww, 4, __LINE__)) return false;
// Store2/Store3 must leave the rest of the destination alone.
float partial[4] = { -1.0f, -1.0f, -1.0f, -1.0f };
v.Store2(partial);
static const float known_store2[4] = { 1.0f, 2.0f, -1.0f, -1.0f };
if (!CompareFloats(partial, known_store2, 4, __LINE__)) return false;
float partial3[4] = { -1.0f, -1.0f, -1.0f, -1.0f };
v.Store3(partial3);
static const float known_store3[4] = { 1.0f, 2.0f, 3.0f, -1.0f };
if (!CompareFloats(partial3, known_store3, 4, __LINE__)) return false;
// Transpose of four columns.
static const float col_values[4][4] = {
{ 0.0f, 1.0f, 2.0f, 3.0f },
{ 4.0f, 5.0f, 6.0f, 7.0f },
{ 8.0f, 9.0f, 10.0f, 11.0f },
{ 12.0f, 13.0f, 14.0f, 15.0f },
};
Vec4F32 c0 = Vec4F32::Load(col_values[0]);
Vec4F32 c1 = Vec4F32::Load(col_values[1]);
Vec4F32 c2 = Vec4F32::Load(col_values[2]);
Vec4F32 c3 = Vec4F32::Load(col_values[3]);
Vec4F32::Transpose(c0, c1, c2, c3);
float transposed[16];
c0.Store(transposed);
c1.Store(transposed + 4);
c2.Store(transposed + 8);
c3.Store(transposed + 12);
for (int row = 0; row < 4; row++) {
for (int col = 0; col < 4; col++) {
EXPECT_EQ_FLOAT(transposed[row * 4 + col], col_values[col][row]);
}
}
// LoadTranspose should match a plain Load of each row followed by Transpose.
static const float flat[16] = {
0.0f, 1.0f, 2.0f, 3.0f,
4.0f, 5.0f, 6.0f, 7.0f,
8.0f, 9.0f, 10.0f, 11.0f,
12.0f, 13.0f, 14.0f, 15.0f,
};
Vec4F32 t0, t1, t2, t3;
Vec4F32::LoadTranspose(flat, t0, t1, t2, t3);
float loadTransposed[16];
t0.Store(loadTransposed);
t1.Store(loadTransposed + 4);
t2.Store(loadTransposed + 8);
t3.Store(loadTransposed + 12);
if (!CompareFloats(loadTransposed, transposed, 16, __LINE__)) return false;
return true;
}
static bool TestVec4F32NaNInf() {
const float inf = INFINITY;
const float nan = NAN;
const float values[4] = { 1.0f, nan, inf, -inf };
Vec4F32 v = Vec4F32::Load(values);
float result[4];
v.ZeroNaNs().Store(result);
EXPECT_EQ_FLOAT(result[0], 1.0f);
EXPECT_EQ_FLOAT(result[1], 0.0f);
// ZeroNaNs leaves infinities alone.
EXPECT_TRUE(std::isinf(result[2]));
EXPECT_TRUE(std::isinf(result[3]));
// The contract is just "a value that yields zero when multiplied by zero" - whatever is
// cheapest to get there. So the implementations legitimately differ on what a bad lane
// becomes: SSE2 clamps to +-FLT_MAX, NEON, LSX and the scalar fallback zero it. Check the
// property rather than the value, and that good lanes are left alone.
v.CleanNaNInfs().Store(result);
EXPECT_EQ_FLOAT(result[0], 1.0f);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i] * 0.0f, 0.0f);
}
return true;
}
static bool TestVec4F32Loads() {
float result[4];
// LoadF24x3_One: three 24-bit values shifted up into floats, lane 3 forced to 1.0f.
const uint32_t f24_values[4] = { 0x3F8000 >> 0, 0x400000, 0x3F0000, 0x123456 };
Vec4F32::LoadF24x3_One(f24_values).Store(result);
for (int i = 0; i < 3; i++) {
float expected;
uint32_t bits = f24_values[i] << 8;
memcpy(&expected, &bits, 4);
EXPECT_EQ_FLOAT(result[i], expected);
}
EXPECT_EQ_FLOAT(result[3], 1.0f);
// LoadF24x4: same, but all four lanes come from the source.
Vec4F32::LoadF24x4(f24_values).Store(result);
for (int i = 0; i < 4; i++) {
float expected;
uint32_t bits = f24_values[i] << 8;
memcpy(&expected, &bits, 4);
EXPECT_EQ_FLOAT(result[i], expected);
}
// The normalizing loads. Some of these read 8 bytes, so give them room.
const int8_t s8_values[8] = { -1, -128, 127, 45, 0, 0, 0, 0 };
Vec4F32::LoadS8Norm(s8_values).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i], (float)s8_values[i] / 128.0f);
}
const int16_t s16_values[8] = { -1, -32768, 32767, 1234, 0, 0, 0, 0 };
Vec4F32::LoadS16Norm(s16_values).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i], (float)s16_values[i] / 32768.0f);
}
const uint8_t u8_values[8] = { 0, 255, 128, 7, 0, 0, 0, 0 };
Vec4F32::LoadU8Norm(u8_values).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_APPROX_EQ_FLOAT(result[i], (float)u8_values[i] / 255.0f);
}
// The converting (non-normalizing) loads.
Vec4F32::LoadConvertS16(s16_values).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i], (float)s16_values[i]);
}
Vec4F32::LoadConvertS8(s8_values).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i], (float)s8_values[i]);
}
Vec4F32::LoadConvertU8(u8_values).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i], (float)u8_values[i]);
}
// StoreConvertToU8 truncates towards zero and saturates to 0-255.
const float to_u8_values[4] = { 0.0f, 255.0f, 12.75f, 200.0f };
uint8_t u8_out[4];
Vec4F32::Load(to_u8_values).StoreConvertToU8(u8_out);
EXPECT_EQ_INT(u8_out[0], 0);
EXPECT_EQ_INT(u8_out[1], 255);
EXPECT_EQ_INT(u8_out[2], 12);
EXPECT_EQ_INT(u8_out[3], 200);
const float saturate_values[4] = { -1.0f, 300.0f, -0.5f, 1000.0f };
Vec4F32::Load(saturate_values).StoreConvertToU8(u8_out);
EXPECT_EQ_INT(u8_out[0], 0);
EXPECT_EQ_INT(u8_out[1], 255);
EXPECT_EQ_INT(u8_out[2], 0);
EXPECT_EQ_INT(u8_out[3], 255);
// Int to float.
const int int_values[4] = { 0, -7, 1000, -32768 };
Vec4F32::FromVec4S32(Vec4S32::Load(int_values)).Store(result);
for (int i = 0; i < 4; i++) {
EXPECT_EQ_FLOAT(result[i], (float)int_values[i]);
}
// Load2 only fills the first two lanes.
const float two_values[2] = { 3.0f, 4.0f };
Vec4F32::Load2(two_values).Store(result);
EXPECT_EQ_FLOAT(result[0], 3.0f);
EXPECT_EQ_FLOAT(result[1], 4.0f);
return true;
}
static bool TestMatrices() {
static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f };
static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f };
static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, };
float result[16];
Mat4F32 a(a_values);
Mat4F32 b(b_values);
Mul4x4By4x4(a, b).Store(result);
if (!CompareFloats(result, known_result, 16, __LINE__)) {
return false;
}
Mat4x3F32 d = Mat4x3F32(b_values + 2);
Mul4x3By4x4(d, a).Store(result);
static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, };
if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) {
return false;
}
static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f };
Vec4F32 v = Vec4F32::Load(vec_values);
v.AsVec3ByMatrix44(b).Store3(result);
static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, };
if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) {
return false;
}
Vec4F32 scale = Vec4F32::Load(a_values);
Vec4F32 translate = Vec4F32::Load(b_values);
TranslateAndScaleInplace(a, scale, translate);
a.Store(result);
static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,};
if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) {
return false;
}
return true;
}
bool TestCrossSIMD() {
if (!TestVec4S32()) return false;
if (!TestVec4S32Compares()) return false;
if (!TestVec4F32Arith()) return false;
if (!TestVec4F32Compares()) return false;
if (!TestVec4F32Lanes()) return false;
if (!TestVec4F32NaNInf()) return false;
if (!TestVec4F32Loads()) return false;
if (!TestMatrices()) return false;
return true;
}
+1 -82
View File
@@ -2764,88 +2764,6 @@ bool TestSIMD() {
return true;
}
static void PrintFloats(const float *f, int count) {
for (int i = 0; i < count; i++) {
printf("%.1ff, ", f[i]);
}
printf("\n");
}
static bool CompareFloats(const float *values, const float *known_good, int count, int line) {
int wrongCount = 0;
for (int i = 0; i < count; i++) {
if (values[i] != known_good[i]) {
wrongCount++;
}
}
if (wrongCount > 0) {
for (int i = 0; i < count; i++) {
bool wrong = values[i] != known_good[i];
printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : "");
}
printf("At UnitTest.cpp:%d: %d / %d were wrong\n", line, wrongCount, count);
return false;
} else {
return true;
}
}
bool TestCrossSIMD() {
static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f };
static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f };
static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, };
float result[16];
Mat4F32 a(a_values);
Mat4F32 b(b_values);
Mul4x4By4x4(a, b).Store(result);
if (!CompareFloats(result, known_result, 16, __LINE__)) {
return false;
}
Mat4x3F32 d = Mat4x3F32(b_values + 2);
Mul4x3By4x4(d, a).Store(result);
static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, };
if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) {
return false;
}
static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f };
Vec4F32 v = Vec4F32::Load(vec_values);
v.AsVec3ByMatrix44(b).Store3(result);
static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, };
if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) {
return false;
}
Vec4F32 scale = Vec4F32::Load(a_values);
Vec4F32 translate = Vec4F32::Load(b_values);
TranslateAndScaleInplace(a, scale, translate);
a.Store(result);
static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,};
if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) {
return false;
}
s8 values[4] = {-1, -128, 127, 45};
float fvalues[4];
Vec4F32::LoadS8Norm(values).Store(fvalues);
static const float known_s8norm_result[4] = {(float)values[0]/128.0f, (float)values[1]/128.0f, (float)values[2]/128.0f, (float)values[3]/128.0f,};
if (!CompareFloats(fvalues, known_s8norm_result, ARRAY_SIZE(known_s8norm_result), __LINE__)) {
return false;
}
// PrintFloats(result, 16);
return true;
}
bool TestVolumeFunc() {
for (int i = 0; i <= 20; i++) {
float mul = Volume10ToMultiplier(i);
@@ -3044,6 +2962,7 @@ struct TestItem {
bool TestArmEmitter();
bool TestArm64Emitter();
bool TestCrossSIMD();
bool TestX64Emitter();
bool TestRiscVEmitter();
bool TestLoongArch64Emitter();
+1
View File
@@ -287,6 +287,7 @@
<ClCompile Include="..\Windows\CaptureDevice.cpp" />
<ClCompile Include="JitHarness.cpp" />
<ClCompile Include="TestArm64Emitter.cpp" />
<ClCompile Include="TestCrossSIMD.cpp" />
<ClCompile Include="TestIRPassSimplify.cpp" />
<ClCompile Include="TestLoongArch64Emitter.cpp" />
<ClCompile Include="TestRiscVEmitter.cpp" />
+1
View File
@@ -14,6 +14,7 @@
<ClCompile Include="TestShaderGenerators.cpp" />
<ClCompile Include="TestThreadManager.cpp" />
<ClCompile Include="TestSoftwareGPUJit.cpp" />
<ClCompile Include="TestCrossSIMD.cpp" />
<ClCompile Include="TestIRPassSimplify.cpp" />
<ClCompile Include="TestDemangle.cpp" />
<ClCompile Include="TestLzrc.cpp" />