From ba06b454f38d3ad3a8d8801f21c9dc9891584c82 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 10:45:03 -0600 Subject: [PATCH 1/9] loongarch64 cross build: don't assume an x86-64 host The GL stub generation in b.sh and setup-loongarch64-cross.sh harvested symbols from hardcoded /usr/lib/x86_64-linux-gnu paths. On any other host those simply don't exist, and neither script treats that as an error - so the stub silently fell back to the handful of GLX symbols listed inline, which isn't enough for GLEW's static archive to link against. Ask the host compiler for its multiarch triplet instead. Co-Authored-By: Claude Opus 5 (1M context) --- b.sh | 7 ++++--- cmake/scripts/setup-loongarch64-cross.sh | 6 ++++-- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/b.sh b/b.sh index a7683d0fb0..2cae01608b 100755 --- a/b.sh +++ b/b.sh @@ -133,10 +133,11 @@ if [ ! -z "$LOONGARCH64_BUILD" ]; then STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c) echo "/* Loongarch64 GL/GLX stub - cross-compilation only */" > "$STUB_C" # Collect all T (exported) symbols from GL/GLX libs, deduplicate, emit stubs + HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) { - for lib in /usr/lib/x86_64-linux-gnu/libGL.so.1 \ - /usr/lib/x86_64-linux-gnu/libGLX.so.0 \ - /usr/lib/x86_64-linux-gnu/libGLdispatch.so.0; do + for lib in /usr/lib/$HOST_MULTIARCH/libGL.so.1 \ + /usr/lib/$HOST_MULTIARCH/libGLX.so.0 \ + /usr/lib/$HOST_MULTIARCH/libGLdispatch.so.0; do [ -f "$lib" ] && nm -D "$lib" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print $3 }' done # Always include the minimal GLX symbols GLEW directly references diff --git a/cmake/scripts/setup-loongarch64-cross.sh b/cmake/scripts/setup-loongarch64-cross.sh index 978dbcebc1..e6a6323dd6 100755 --- a/cmake/scripts/setup-loongarch64-cross.sh +++ b/cmake/scripts/setup-loongarch64-cross.sh @@ -32,10 +32,12 @@ done # B26 branch to address 0 would overflow on LoongArch). echo "Generating libGL.so stub..." +HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) + HOST_GL="" for candidate in \ - /usr/lib/x86_64-linux-gnu/libGL.so.1 \ - /usr/lib/x86_64-linux-gnu/libGL.so \ + /usr/lib/$HOST_MULTIARCH/libGL.so.1 \ + /usr/lib/$HOST_MULTIARCH/libGL.so \ /usr/lib/libGL.so.1 ; do [ -f "$candidate" ] && HOST_GL="$candidate" && break done From 9a29bf84d86dfb9335c62af6b27513f151ccfc4e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 11:01:40 -0600 Subject: [PATCH 2/9] Add a riscv64 cross build, generalizing the loongarch64 one The pieces were nearly all in place already - CMakeLists has detected riscv64 and set RISCV64 since the backend landed - so this is mostly a matter of not hardcoding loongarch64 in the cross-build plumbing: - setup-loongarch64-cross.sh becomes setup-cross.sh , taking the triple and dynamic linker name from the argument. - b.sh gains --riscv64, and the GL stub block is keyed off a compiler variable rather than a loongarch64-only flag. - New cmake/Toolchains/riscv64-linux-gnu.cmake, mirroring the existing one. The SDL carve-out moves from LOONGARCH64_DEVICE to a new HEADLESS_CROSS option set by b.sh. That flag was keyed off the target architecture, which is also true when building natively on such a machine - harmless so far, but riscv64 hardware that people actually build PPSSPP on exists, and it should still get the normal SDL frontend. Both toolchain files now find qemu rather than assuming the static build: Debian ships the static binaries in qemu-user-static and the dynamic ones in qemu-user, and only the latter is available on some releases. Neither being present is fine too - it just means no compile-check programs run. Verified by a clean loongarch64 build; riscv64 is untested so far, the toolchain isn't installed here yet. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/build.yml | 17 ++++-- CMakeLists.txt | 12 ++-- b.sh | 19 ++++--- cmake/Toolchains/loongarch64-linux-gnu.cmake | 28 ++++++---- cmake/Toolchains/riscv64-linux-gnu.cmake | 44 +++++++++++++++ ...up-loongarch64-cross.sh => setup-cross.sh} | 55 +++++++++++++------ 6 files changed, 129 insertions(+), 46 deletions(-) create mode 100644 cmake/Toolchains/riscv64-linux-gnu.cmake rename cmake/scripts/{setup-loongarch64-cross.sh => setup-cross.sh} (63%) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 82d6f160f1..7ae9fdcc8f 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -279,6 +279,13 @@ jobs: args: ./b.sh --loongarch64 PPSSPPHeadless id: loongarch64 + - os: ubuntu-26.04 + extra: riscv64 + cc: gcc + cxx: g++ + args: ./b.sh --riscv64 PPSSPPHeadless + id: riscv64 + - os: macos-latest extra: test cc: clang @@ -318,7 +325,7 @@ jobs: ndk-version: r29 - name: Install Linux dependencies - if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64' + if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64' && matrix.extra != 'riscv64' run: | sudo apt-get update -y -qq sudo apt-get install libsdl3-dev libgl1-mesa-dev libglu1-mesa-dev libsdl3-ttf-dev libfontconfig1-dev libcurl4-openssl-dev @@ -327,12 +334,12 @@ jobs: if: runner.os == 'macOS' && matrix.id == 'macos' run: brew install sdl3 sdl3_ttf - - name: Install loongarch64 cross-compilation dependencies - if: matrix.extra == 'loongarch64' + - name: Install cross-compilation dependencies + if: matrix.extra == 'loongarch64' || matrix.extra == 'riscv64' run: | sudo apt-get update -y -qq - sudo apt-get install -y gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu libgl-dev libglu1-mesa-dev - sudo cmake/scripts/setup-loongarch64-cross.sh + sudo apt-get install -y gcc-14-${{ matrix.extra }}-linux-gnu g++-14-${{ matrix.extra }}-linux-gnu binutils-${{ matrix.extra }}-linux-gnu libgl-dev libglu1-mesa-dev + sudo cmake/scripts/setup-cross.sh ${{ matrix.extra }} - name: Install iOS dependencies if: matrix.id == 'ios' diff --git a/CMakeLists.txt b/CMakeLists.txt index c81d415524..8a13d80564 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -184,6 +184,8 @@ option(USE_VULKAN_DISPLAY_KHR "Enable or disable full screen display of Vulkan" # :: Frontends option(MOBILE_DEVICE "Set to ON when targeting a mobile device" ${MOBILE_DEVICE}) option(HEADLESS "Set to OFF to not generate the PPSSPPHeadless target" ${HEADLESS}) +# Set by b.sh for the loongarch64/riscv64 cross builds, whose sysroots have no SDL3. +option(HEADLESS_CROSS "Set to ON for a headless-only cross build (no SDL frontend)" ${HEADLESS_CROSS}) option(ATLAS_TOOL "Set to OFF to not generate the ATLAS_TOOL target" ${ATLAS_TOOL}) option(UNITTEST "Set to ON to generate the unittest target" ${UNITTEST}) option(SIMULATOR "Set to ON when targeting an x86 simulator of an ARM platform" ${SIMULATOR}) @@ -331,7 +333,7 @@ set(SDL_LIB_TARGET "") set(SDL_TTF_LIB_TARGET "") set(FONTCONFIG_LIB_TARGET "") -if(NOT LIBRETRO AND NOT IOS AND NOT LOONGARCH64_DEVICE AND NOT ANDROID) +if(NOT LIBRETRO AND NOT IOS AND NOT HEADLESS_CROSS AND NOT ANDROID) find_package(SDL3 QUIET) if(NOT SDL3_FOUND) if(CMAKE_SYSTEM_NAME STREQUAL "Linux" AND (EXISTS "/usr/bin/apt" OR EXISTS "/usr/bin/apt-get" OR EXISTS "/etc/debian_version")) @@ -1001,9 +1003,11 @@ elseif(WIN32) # Don't care about SDL. set(TargetBin PPSSPPWindows) elseif(LIBRETRO) -elseif(LOONGARCH64_DEVICE) - # Cross-compiling for loongarch64 currently only builds PPSSPPHeadless, and there's - # no SDL3 available for that target, so skip the desktop SDL frontend entirely. +elseif(HEADLESS_CROSS) + # The loongarch64/riscv64 cross builds only build PPSSPPHeadless, and there's no + # SDL3 available in those sysroots, so skip the desktop SDL frontend entirely. + # Note this is deliberately not keyed off the target architecture: a native build + # on such a machine can have SDL3 and should still get the normal frontend. else() if(GOLD) set(TargetBin PPSSPPGold) diff --git a/b.sh b/b.sh index 2cae01608b..ff80e9bdd1 100755 --- a/b.sh +++ b/b.sh @@ -31,9 +31,14 @@ do CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/raspberry.armv8.cmake ${CMAKE_ARGS}" ;; --loongarch64) - CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}" + CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}" TARGET_OS=loongarch64 - LOONGARCH64_BUILD=1 + CROSS_STUB_CC=loongarch64-linux-gnu-gcc-14 + ;; + --riscv64) + CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/riscv64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}" + TARGET_OS=riscv64 + CROSS_STUB_CC=riscv64-linux-gnu-gcc-14 ;; --android) CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=android/android.toolchain.cmake ${CMAKE_ARGS}" TARGET_OS=Android @@ -122,16 +127,16 @@ echo Building with $CORES_COUNT threads mkdir -p ${BUILD_DIR} -# For loongarch64 cross-compilation, build a comprehensive GL/GLX stub into +# For the headless cross targets, build a comprehensive GL/GLX stub into # /stublibs/libGL.so so GLEW's static archive can resolve its symbols -# via PLT entries (a direct B26 branch to address 0 overflows on LoongArch). +# via PLT entries (a direct branch to address 0 overflows the relocation). # This stub is always (re)generated to pick up any new needed symbols. -if [ ! -z "$LOONGARCH64_BUILD" ]; then +if [ ! -z "$CROSS_STUB_CC" ]; then STUB_DIR=${BUILD_DIR}/stublibs STUB_GL=${STUB_DIR}/libGL.so mkdir -p "${STUB_DIR}" STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c) - echo "/* Loongarch64 GL/GLX stub - cross-compilation only */" > "$STUB_C" + echo "/* ${TARGET_OS} GL/GLX stub - cross-compilation only */" > "$STUB_C" # Collect all T (exported) symbols from GL/GLX libs, deduplicate, emit stubs HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) { @@ -144,7 +149,7 @@ if [ ! -z "$LOONGARCH64_BUILD" ]; then printf '%s\n' glXGetProcAddressARB glXGetClientString glXQueryVersion \ glBindTexture glGetString glGetIntegerv } | sort -u | awk '{ print "void "$1"(void){}" }' >> "$STUB_C" - loongarch64-linux-gnu-gcc-14 -shared -fPIC -Wno-implicit-function-declaration \ + $CROSS_STUB_CC -shared -fPIC -Wno-implicit-function-declaration \ -o "${STUB_GL}" "$STUB_C" rm "$STUB_C" fi diff --git a/cmake/Toolchains/loongarch64-linux-gnu.cmake b/cmake/Toolchains/loongarch64-linux-gnu.cmake index b39ed83905..014630439c 100644 --- a/cmake/Toolchains/loongarch64-linux-gnu.cmake +++ b/cmake/Toolchains/loongarch64-linux-gnu.cmake @@ -17,8 +17,8 @@ set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE) # Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so # rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have. # A stub libGL.so + GL headers must be present in the sysroot; they are placed -# there by cmake/scripts/setup-loongarch64-cross.sh (run once with sudo, or -# automatically by the CI install step). +# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically +# by the CI install step). set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE) # b.sh also creates a stub in /stublibs/ as a local fallback so the @@ -27,14 +27,18 @@ set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs" CACHE STRING "" FORCE) -# Allow CMake's compile-check programs to run via QEMU. -# The /lib64 symlink is created by the setup script; without it, pass -L -# explicitly so QEMU can find the loongarch64 dynamic linker. -if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1") - set(CMAKE_CROSSCOMPILING_EMULATOR "qemu-loongarch64-static" - CACHE STRING "" FORCE) -else() - set(CMAKE_CROSSCOMPILING_EMULATOR - "qemu-loongarch64-static;-L;/usr/loongarch64-linux-gnu" - CACHE STRING "" FORCE) +# Allow CMake's compile-check programs to run via QEMU, if it's installed. +# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic +# ones in qemu-user; either will do, so take whichever is present. +find_program(QEMU_LOONGARCH64 NAMES qemu-loongarch64-static qemu-loongarch64) +if(QEMU_LOONGARCH64) + # The /lib64 symlink is created by the setup script; without it, pass -L + # explicitly so QEMU can find the loongarch64 dynamic linker. + if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1") + set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_LOONGARCH64}" CACHE STRING "" FORCE) + else() + set(CMAKE_CROSSCOMPILING_EMULATOR + "${QEMU_LOONGARCH64};-L;/usr/loongarch64-linux-gnu" + CACHE STRING "" FORCE) + endif() endif() diff --git a/cmake/Toolchains/riscv64-linux-gnu.cmake b/cmake/Toolchains/riscv64-linux-gnu.cmake new file mode 100644 index 0000000000..a80150c6e8 --- /dev/null +++ b/cmake/Toolchains/riscv64-linux-gnu.cmake @@ -0,0 +1,44 @@ +set(CMAKE_SYSTEM_NAME Linux) +set(CMAKE_SYSTEM_PROCESSOR riscv64) + +set(CMAKE_C_COMPILER riscv64-linux-gnu-gcc-14) +set(CMAKE_CXX_COMPILER riscv64-linux-gnu-g++-14) +set(CMAKE_ASM_COMPILER riscv64-linux-gnu-gcc-14) + +set(CMAKE_FIND_ROOT_PATH /usr/riscv64-linux-gnu) +set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER) +set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY) + +# No X11 in the sysroot; disable the X11 Vulkan path to avoid include-dir errors. +set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE) + +# Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so +# rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have. +# A stub libGL.so + GL headers must be present in the sysroot; they are placed +# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically +# by the CI install step). +set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE) + +# b.sh also creates a stub in /stublibs/ as a local fallback so the +# linker can resolve GL/GLX calls from GLEW's static archive via PLT entries. +set(CMAKE_EXE_LINKER_FLAGS + "${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs" + CACHE STRING "" FORCE) + +# Allow CMake's compile-check programs to run via QEMU, if it's installed. +# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic +# ones in qemu-user; either will do, so take whichever is present. +find_program(QEMU_RISCV64 NAMES qemu-riscv64-static qemu-riscv64) +if(QEMU_RISCV64) + # The /lib64 symlink is created by the setup script; without it, pass -L + # explicitly so QEMU can find the riscv64 dynamic linker. + if(EXISTS "/lib64/ld-linux-riscv64-lp64d.so.1") + set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_RISCV64}" CACHE STRING "" FORCE) + else() + set(CMAKE_CROSSCOMPILING_EMULATOR + "${QEMU_RISCV64};-L;/usr/riscv64-linux-gnu" + CACHE STRING "" FORCE) + endif() +endif() diff --git a/cmake/scripts/setup-loongarch64-cross.sh b/cmake/scripts/setup-cross.sh similarity index 63% rename from cmake/scripts/setup-loongarch64-cross.sh rename to cmake/scripts/setup-cross.sh index e6a6323dd6..ea9b8df86b 100755 --- a/cmake/scripts/setup-loongarch64-cross.sh +++ b/cmake/scripts/setup-cross.sh @@ -1,15 +1,34 @@ #!/bin/bash -# Sets up the loongarch64 cross-compilation environment. -# Run once with sudo. After this, ./b.sh --loongarch64 works on a fresh checkout. +# Sets up a cross-compilation environment for one of the headless-only targets. +# Run once with sudo. After this, ./b.sh -- works on a fresh checkout. +# +# Usage: sudo cmake/scripts/setup-cross.sh set -e -SYSROOT=/usr/loongarch64-linux-gnu -CROSS_GCC=loongarch64-linux-gnu-gcc-14 +TARGET="$1" + +case "$TARGET" in + loongarch64) + TRIPLE=loongarch64-linux-gnu + LD_SO=ld-linux-loongarch-lp64d.so.1 + ;; + riscv64) + TRIPLE=riscv64-linux-gnu + LD_SO=ld-linux-riscv64-lp64d.so.1 + ;; + *) + echo "Usage: sudo $0 " + exit 1 + ;; +esac + +SYSROOT=/usr/$TRIPLE +CROSS_GCC=$TRIPLE-gcc-14 if ! command -v $CROSS_GCC &>/dev/null; then echo "Error: $CROSS_GCC not found. Install it first:" - echo " sudo apt install gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu" + echo " sudo apt install gcc-14-$TRIPLE g++-14-$TRIPLE binutils-$TRIPLE" exit 1 fi @@ -18,6 +37,8 @@ if [ "$(id -u)" -ne 0 ]; then exit 1 fi +HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) + # ── OpenGL headers ─────────────────────────────────────────────────────────── # GL headers are platform-independent C headers; copy from the host. echo "Installing GL headers into sysroot..." @@ -27,13 +48,11 @@ for h in gl.h glext.h glcorearb.h glu.h; do done # ── GL stub library ────────────────────────────────────────────────────────── -# Generate a loongarch64 libGL.so that exports all symbols from the host +# Generate a libGL.so for the target that exports all symbols from the host # libGL so that GLEW's static archive can be linked via PLT entries (a direct -# B26 branch to address 0 would overflow on LoongArch). +# branch to address 0 would overflow the relocation on these targets). echo "Generating libGL.so stub..." -HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) - HOST_GL="" for candidate in \ /usr/lib/$HOST_MULTIARCH/libGL.so.1 \ @@ -43,8 +62,8 @@ for candidate in \ done STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c) -echo "/* Loongarch64 GL/GLX stub - for cross-compilation only */" > "$STUB_C" -echo "void __loongarch64_gl_placeholder(void) {}" >> "$STUB_C" +echo "/* $TARGET GL/GLX stub - for cross-compilation only */" > "$STUB_C" +echo "void __cross_gl_placeholder(void) {}" >> "$STUB_C" if [ -n "$HOST_GL" ]; then nm -D "$HOST_GL" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print "void "$3"(void){}" }' >> "$STUB_C" @@ -62,16 +81,16 @@ rm "$STUB_C" echo " Installed to $SYSROOT/lib/libGL.so" # ── Dynamic linker symlink ─────────────────────────────────────────────────── -# Without this, binfmt_misc can't find the loongarch64 dynamic linker. -LD_LINUX=$SYSROOT/lib/ld-linux-loongarch-lp64d.so.1 +# Without this, binfmt_misc can't find the target's dynamic linker. +LD_LINUX=$SYSROOT/lib/$LD_SO if [ -f "$LD_LINUX" ]; then mkdir -p /lib64 - ln -sf "$LD_LINUX" /lib64/ld-linux-loongarch-lp64d.so.1 - echo "Created /lib64/ld-linux-loongarch-lp64d.so.1 -> $LD_LINUX" + ln -sf "$LD_LINUX" /lib64/$LD_SO + echo "Created /lib64/$LD_SO -> $LD_LINUX" fi echo "" echo "Setup complete." -echo "Build: ./b.sh --loongarch64" -echo "Run: qemu-loongarch64-static -L $SYSROOT " -echo " (after setup the /lib64 symlink lets you run loongarch64 binaries directly)" +echo "Build: ./b.sh --$TARGET" +echo "Run: qemu-$TARGET -L $SYSROOT " +echo " (after setup the /lib64 symlink lets you run $TARGET binaries directly)" From 8e3b4245c27995de0c15fb1658b7a6d512aee46a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 12:19:24 -0600 Subject: [PATCH 3/9] CrossSIMD: fix the scalar fallback, which nothing compiled until now The riscv64 cross build is the first target to take the non-SIMD path, and it didn't compile: - LoadF24x4 called LoadR24x3_One, which doesn't exist. It should shift all four lanes, like the SSE/NEON/LSX versions do. - isnan/isinf were unqualified, and wasn't included. - WithLane3From and AnyCompareBitsSet were missing entirely. LoadF24x3_One also left lane 3 as zero, where all three SIMD versions set it to 1.0f - a behavioural bug that would only have shown up once someone ran this path. Verified by flipping TEST_FALLBACK in the header: the whole tree builds, and with the scalar path in use the unit tests and pspautotests both pass in full (342/342, software renderer, which leans on this code heavily). Co-Authored-By: Claude Opus 5 (1M context) --- Common/Math/CrossSIMD.h | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/Common/Math/CrossSIMD.h b/Common/Math/CrossSIMD.h index af30c25b06..2d5f62da37 100644 --- a/Common/Math/CrossSIMD.h +++ b/Common/Math/CrossSIMD.h @@ -6,6 +6,7 @@ #pragma once +#include #include #include "Common/Math/SIMDHeaders.h" @@ -1672,14 +1673,18 @@ struct Vec4F32 { } static Vec4F32 LoadF24x3_One(const uint32_t *src) { - uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, 0 }; + constexpr uint32_t kOneF32Bits = 0x3F800000; + uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, kOneF32Bits }; Vec4F32 temp; memcpy(temp.v, shifted, sizeof(temp.v)); return temp; } static Vec4F32 LoadF24x4(const uint32_t *src) { - return LoadR24x3_One(src); + uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, src[3] << 8 }; + Vec4F32 temp; + memcpy(temp.v, shifted, sizeof(temp.v)); + return temp; } static Vec4F32 FromVec4S32(Vec4S32 src) { @@ -1695,7 +1700,7 @@ struct Vec4F32 { Vec4F32 ZeroNaNs() const { Vec4F32 temp; for (int i = 0; i < 4; i++) { - temp.v[i] = isnan(v[i]) ? 0.0f : v[i]; + temp.v[i] = std::isnan(v[i]) ? 0.0f : v[i]; } return temp; } @@ -1703,7 +1708,7 @@ struct Vec4F32 { Vec4F32 CleanNaNInfs() { Vec4F32 temp; for (int i = 0; i < 4; i++) { - temp.v[i] = (isnan(v[i]) || isinf(v[i])) ? 0.0f : v[i]; + temp.v[i] = (std::isnan(v[i]) || std::isinf(v[i])) ? 0.0f : v[i]; } return temp; } @@ -1798,6 +1803,10 @@ struct Vec4F32 { return Vec4F32{ { v[0], v[1], v[2], 1.0f } }; } + Vec4F32 WithLane3From(Vec4F32 other) const { + return Vec4F32{ { v[0], v[1], v[2], other.v[3] } }; + } + Vec4S32 CompareEq(Vec4F32 other) const { Vec4S32 temp; for (int i = 0; i < 4; i++) { @@ -1919,6 +1928,10 @@ inline bool AllCompareBitsSet(Vec4S32 value) { return true; } +inline bool AnyCompareBitsSet(Vec4S32 value) { + return value.v[0] != 0 || value.v[1] != 0 || value.v[2] != 0 || value.v[3] != 0; +} + struct Vec4U16 { uint16_t v[4]; // 64 bits. From 91dff7cb3f8de43d032f24ae6b573832254b5e4e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 11:14:08 -0600 Subject: [PATCH 4/9] LoongArch64: fix QuickCallFunctionR passing the argument in a vector register MOVE(LoongArch64Reg::X4, arg) targets X4, which is an LASX 256-bit vector register (0x64), not a0. The LP64D ABI puts the first integer argument in a0, which is R4. This is not a corner case: GenerateFixedCode uses QuickCallFunctionR for CoreTiming::Advance, so the emitter asserted ("DJK instruction rd must be GPR") while building the dispatcher, before a single block was compiled. The LoongArch native JIT could not start at all. Found by running the loongarch64 cross build under qemu-user. Co-Authored-By: Claude Opus 5 (1M context) --- Common/LoongArch64Emitter.h | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Common/LoongArch64Emitter.h b/Common/LoongArch64Emitter.h index 134b5de61d..d5abda3809 100644 --- a/Common/LoongArch64Emitter.h +++ b/Common/LoongArch64Emitter.h @@ -166,7 +166,8 @@ public: QuickCallFunction((const u8 *)func, scratchreg); } void QuickCallFunctionR(const u8 *func, LoongArch64Reg arg, LoongArch64Reg scratchreg = R_RA) { - MOVE(LoongArch64Reg::X4, arg); + // The first integer argument goes in a0, which is R4 - X4 is an LASX vector register. + MOVE(LoongArch64Reg::R4, arg); QuickJump(scratchreg, R_RA, func); } template From 320cfd77414b59b1be33ebc5d57ff3b4b38c87b3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 12:35:09 -0600 Subject: [PATCH 5/9] CrossSIMD: fill in the scalar fallback, fix two LSX bugs LoadConvertU8, StoreConvertToU8 and LoadTranspose existed in the SSE2, NEON and LSX implementations but not in the scalar one, so anything using them wouldn't build on a target without SIMD. The two LSX bugs were found by the new CrossSIMD unit test, run under qemu-loongarch64: - Vec4F32::operator[] had a switch with no breaks, so every index fell through to the default and returned lane 3. - StoreConvertToU8 narrowed with the logical (unsigned) saturating shifts, which turn a negative value into a huge unsigned one and saturate it to 255. It should clamp to 0, as the packs/packus pair in the SSE version does. Narrow signed->signed and then signed->unsigned instead. Co-Authored-By: Claude Opus 5 (1M context) --- Common/Math/CrossSIMD.h | 43 +++++++++++++++++++++++++++++++++++------ 1 file changed, 37 insertions(+), 6 deletions(-) diff --git a/Common/Math/CrossSIMD.h b/Common/Math/CrossSIMD.h index 2d5f62da37..8c9316b671 100644 --- a/Common/Math/CrossSIMD.h +++ b/Common/Math/CrossSIMD.h @@ -1188,8 +1188,11 @@ struct Vec4F32 { } void StoreConvertToU8(uint8_t *dest) { __m128i ivalue32 = __lsx_vftintrz_w_s(v); - __m128i ivalue16 = __lsx_vssrlrni_hu_w(ivalue32, ivalue32, 0); - __m128i ivalue8 = __lsx_vssrlrni_bu_h(ivalue16, ivalue16, 0); + // Narrow signed->signed first, then signed->unsigned, matching the packs/packus pair the + // SSE version uses. The logical (unsigned) narrowing shifts turn a negative value into a + // huge unsigned one, which saturates to 255 instead of clamping to 0. + __m128i ivalue16 = __lsx_vssrani_h_w(ivalue32, ivalue32, 0); + __m128i ivalue8 = __lsx_vssrani_bu_h(ivalue16, ivalue16, 0); uint32_t value = __lsx_vpickve2gr_wu(ivalue8, 0); memcpy(dest, &value, sizeof(uint32_t)); } @@ -1199,10 +1202,10 @@ struct Vec4F32 { // index is a compile-time constant parameter to the intrinsic. int ival; switch (index) { - case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); - case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); - case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); - default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); + case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); break; + case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); break; + case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); break; + default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); break; } float fval; memcpy(&fval, &ival, sizeof(float)); @@ -1672,6 +1675,25 @@ struct Vec4F32 { return temp; } + static Vec4F32 LoadConvertU8(const uint8_t *src) { // Note: will load 8 bytes, not 4. Only the first 4 bytes will be used. + Vec4F32 temp; + for (int i = 0; i < 4; i++) { + temp.v[i] = (float)src[i]; + } + return temp; + } + + // Truncates towards zero and saturates to 0-255, like the packing instructions the other + // implementations use. + void StoreConvertToU8(uint8_t *dest) { + for (int i = 0; i < 4; i++) { + int value = (int)v[i]; + if (value < 0) value = 0; + if (value > 255) value = 255; + dest[i] = (uint8_t)value; + } + } + static Vec4F32 LoadF24x3_One(const uint32_t *src) { constexpr uint32_t kOneF32Bits = 0x3F800000; uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, kOneF32Bits }; @@ -1866,6 +1888,15 @@ struct Vec4F32 { return Vec4F32{{ v[3], v[3], v[3], v[3] }}; } + // Loads four rows of four floats and transposes them into columns. + static void LoadTranspose(const float *src, Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) { + col0 = Vec4F32::Load(src); + col1 = Vec4F32::Load(src + 4); + col2 = Vec4F32::Load(src + 8); + col3 = Vec4F32::Load(src + 12); + Transpose(col0, col1, col2, col3); + } + // In-place transpose. static void Transpose(Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) { float m[16]; From 5d703b014e361eda2a2a805dd8170f6457485193 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 12:35:09 -0600 Subject: [PATCH 6/9] unittest: move the CrossSIMD test to its own file and expand it CrossSIMD has four independent implementations and only one is compiled per machine, so each architecture has to check its own copy against hand-worked results. Running this under qemu gives us coverage of the targets we have no hardware for - it's already caught two LSX bugs. Covers Vec4S32 and Vec4F32 arithmetic, comparisons and the mask helpers, the lane accessors and shuffles, transpose, the various loads (including the 24-bit and normalizing ones), NaN/Inf handling, and the matrix routines the old test already had. Verified on three of the four implementations: NEON natively, the scalar fallback via TEST_FALLBACK, and LSX under qemu-loongarch64. SSE2 is left to CI. Co-Authored-By: Claude Opus 5 (1M context) --- CMakeLists.txt | 1 + android/jni/Android.mk | 1 + unittest/TestCrossSIMD.cpp | 545 +++++++++++++++++++++++++++++ unittest/UnitTest.cpp | 83 +---- unittest/UnitTests.vcxproj | 1 + unittest/UnitTests.vcxproj.filters | 1 + 6 files changed, 550 insertions(+), 82 deletions(-) create mode 100644 unittest/TestCrossSIMD.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 8a13d80564..65e2b49ca7 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1366,6 +1366,7 @@ if(UNITTEST) unittest/TestShaderGenerators.cpp unittest/TestArmEmitter.cpp unittest/TestArm64Emitter.cpp + unittest/TestCrossSIMD.cpp unittest/TestIRPassSimplify.cpp unittest/TestX64Emitter.cpp unittest/TestVertexJit.cpp diff --git a/android/jni/Android.mk b/android/jni/Android.mk index 2e0a7ab5a9..b308bc058d 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -1051,6 +1051,7 @@ ifeq ($(UNITTEST),1) $(SRC)/unittest/TestX64Emitter.cpp \ $(SRC)/unittest/TestRiscVEmitter.cpp \ $(SRC)/unittest/TestLoongArch64Emitter.cpp \ + $(SRC)/unittest/TestCrossSIMD.cpp \ $(SRC)/unittest/TestIRPassSimplify.cpp \ $(SRC)/unittest/TestShaderGenerators.cpp \ $(SRC)/unittest/TestSoftwareGPUJit.cpp \ diff --git a/unittest/TestCrossSIMD.cpp b/unittest/TestCrossSIMD.cpp new file mode 100644 index 0000000000..e584bf762b --- /dev/null +++ b/unittest/TestCrossSIMD.cpp @@ -0,0 +1,545 @@ +// Copyright (c) 2012- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +// Tests for Common/Math/CrossSIMD.h. +// +// CrossSIMD has four independent implementations - SSE2, NEON, LoongArch LSX and a plain scalar +// fallback - and only one of them is compiled on any given machine. So the point of these tests is +// to check whatever got compiled against results worked out by hand, rather than against each other; +// that way every architecture we build for verifies its own copy. Running the unit tests under +// qemu covers the ones we have no hardware for. +// +// Only functions that all four implementations provide are tested here, otherwise this wouldn't +// build everywhere. + +#include "ppsspp_config.h" + +#include +#include +#include + +#include "Common/Common.h" +#include "Common/CommonTypes.h" +#include "Common/Math/CrossSIMD.h" +#include "unittest/UnitTest.h" + +static bool CompareFloats(const float *values, const float *known_good, int count, int line) { + int wrongCount = 0; + for (int i = 0; i < count; i++) { + if (values[i] != known_good[i]) { + wrongCount++; + } + } + if (wrongCount > 0) { + for (int i = 0; i < count; i++) { + bool wrong = values[i] != known_good[i]; + printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : ""); + } + printf("At TestCrossSIMD.cpp:%d: %d / %d were wrong\n", line, wrongCount, count); + return false; + } + return true; +} + +// For the approximate operations (reciprocals), which are allowed to differ between architectures. +static bool CompareFloatsApprox(const float *values, const float *known_good, int count, float tolerance, int line) { + for (int i = 0; i < count; i++) { + const float diff = fabsf(values[i] - known_good[i]); + const float scale = fabsf(known_good[i]) > 1.0f ? fabsf(known_good[i]) : 1.0f; + if (diff / scale > tolerance) { + printf("At TestCrossSIMD.cpp:%d: lane %d: %0.6f vs %0.6f\n", line, i, values[i], known_good[i]); + return false; + } + } + return true; +} + +static bool TestVec4S32() { + const int a_values[4] = { 3, -7, 0x4000, -1 }; + const int b_values[4] = { 5, 11, -2, 0x7FFF }; + + Vec4S32 a = Vec4S32::Load(a_values); + Vec4S32 b = Vec4S32::Load(b_values); + + int result[4]; + + (a + b).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] + b_values[i]); + } + + (a - b).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] - b_values[i]); + } + + a.Mul(b).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] * b_values[i]); + } + + // Mul16 only promises correct results when both sides fit in 16 bits. + const int small_a[4] = { 3, -7, 300, -1 }; + const int small_b[4] = { 5, 11, -2, 1000 }; + Vec4S32 sa = Vec4S32::Load(small_a); + Vec4S32 sb = Vec4S32::Load(small_b); + sa.Mul16(sb).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], small_a[i] * small_b[i]); + } + + Vec4S32::Splat(-5).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], -5); + } + + Vec4S32::Zero().Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], 0); + } + + a.Shl<2>().Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] << 2); + } + + // operator[] should agree with Store. + a.Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(a[i], result[i]); + } + + // Min16/Max16 likewise operate on 16-bit lanes. + sa.Min16(sb).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], small_a[i] < small_b[i] ? small_a[i] : small_b[i]); + } + sa.Max16(sb).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], small_a[i] > small_b[i] ? small_a[i] : small_b[i]); + } + + const int sext_values[4] = { 0x0000FFFF, 0x00007FFF, 0x00008000, 0x00000001 }; + Vec4S32::Load(sext_values).SignExtend16().Store(result); + EXPECT_EQ_INT(result[0], -1); + EXPECT_EQ_INT(result[1], 32767); + EXPECT_EQ_INT(result[2], -32768); + EXPECT_EQ_INT(result[3], 1); + + return true; +} + +static bool TestVec4S32Compares() { + const int a_values[4] = { 1, 2, 3, 4 }; + const int b_values[4] = { 1, 5, 0, 4 }; + Vec4S32 a = Vec4S32::Load(a_values); + Vec4S32 b = Vec4S32::Load(b_values); + + int result[4]; + + // Compares produce all-ones or all-zeros per lane. + a.CompareEq(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + EXPECT_EQ_HEX((u32)result[2], 0u); + EXPECT_EQ_HEX((u32)result[3], 0xFFFFFFFFu); + + a.CompareLt(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0u); + EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[2], 0u); + EXPECT_EQ_HEX((u32)result[3], 0u); + + a.CompareGt(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0u); + EXPECT_EQ_HEX((u32)result[1], 0u); + EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[3], 0u); + + // AllCompareBitsSet / AnyCompareBitsSet over those masks. + EXPECT_FALSE(AllCompareBitsSet(a.CompareEq(b))); + EXPECT_TRUE(AnyCompareBitsSet(a.CompareEq(b))); + EXPECT_TRUE(AllCompareBitsSet(a.CompareEq(a))); + EXPECT_FALSE(AnyCompareBitsSet(a.CompareLt(a))); + + return true; +} + +static bool TestVec4F32Arith() { + const float a_values[4] = { 1.0f, -2.0f, 3.5f, 0.25f }; + const float b_values[4] = { 4.0f, 8.0f, -0.5f, 2.0f }; + + Vec4F32 a = Vec4F32::Load(a_values); + Vec4F32 b = Vec4F32::Load(b_values); + + float result[4]; + float expected[4]; + + (a + b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] + b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + (a - b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] - b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + (a * b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] * b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Mul(2.0f).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] * 2.0f; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Min(b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] < b_values[i] ? a_values[i] : b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Max(b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] > b_values[i] ? a_values[i] : b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Clamp(0.0f, 3.0f).Store(result); + static const float known_clamp[4] = { 1.0f, 0.0f, 3.0f, 0.25f }; + if (!CompareFloats(result, known_clamp, 4, __LINE__)) return false; + + // Recip is allowed to be approximate on some architectures, RecipApprox definitely is. + b.Recip().Store(result); + for (int i = 0; i < 4; i++) expected[i] = 1.0f / b_values[i]; + if (!CompareFloatsApprox(result, expected, 4, 0.001f, __LINE__)) return false; + + b.RecipApprox().Store(result); + if (!CompareFloatsApprox(result, expected, 4, 0.01f, __LINE__)) return false; + + Vec4F32::Splat(1.5f).Store(result); + static const float known_splat[4] = { 1.5f, 1.5f, 1.5f, 1.5f }; + if (!CompareFloats(result, known_splat, 4, __LINE__)) return false; + + Vec4F32::Zero().Store(result); + static const float known_zero[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; + if (!CompareFloats(result, known_zero, 4, __LINE__)) return false; + + // Dot products. Dot3 ignores lane 3. + EXPECT_EQ_FLOAT(a.Dot3(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2]); + EXPECT_EQ_FLOAT(a.Dot4(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2] + a_values[3] * b_values[3]); + + // operator[] and GetLane<> should agree with Store. + a.Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(a[i], result[i]); + } + EXPECT_EQ_FLOAT(a.GetLane<0>(), a_values[0]); + EXPECT_EQ_FLOAT(a.GetLane<3>(), a_values[3]); + + return true; +} + +static bool TestVec4F32Compares() { + const float a_values[4] = { 1.0f, 2.0f, 3.0f, 4.0f }; + const float b_values[4] = { 1.0f, 5.0f, 0.0f, 4.0f }; + Vec4F32 a = Vec4F32::Load(a_values); + Vec4F32 b = Vec4F32::Load(b_values); + + int result[4]; + + a.CompareEq(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + + a.CompareLt(b).Store(result); + EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[2], 0u); + + a.CompareGt(b).Store(result); + EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + + a.CompareLe(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[2], 0u); + + a.CompareGe(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + + return true; +} + +static bool TestVec4F32Lanes() { + const float values[4] = { 1.0f, 2.0f, 3.0f, 4.0f }; + const float other_values[4] = { 9.0f, 9.0f, 9.0f, 42.0f }; + Vec4F32 v = Vec4F32::Load(values); + Vec4F32 other = Vec4F32::Load(other_values); + + float result[4]; + + v.WithLane3Zero().Store(result); + static const float known_zero3[4] = { 1.0f, 2.0f, 3.0f, 0.0f }; + if (!CompareFloats(result, known_zero3, 4, __LINE__)) return false; + + v.WithLane3One().Store(result); + static const float known_one3[4] = { 1.0f, 2.0f, 3.0f, 1.0f }; + if (!CompareFloats(result, known_one3, 4, __LINE__)) return false; + + v.WithLane3From(other).Store(result); + static const float known_from3[4] = { 1.0f, 2.0f, 3.0f, 42.0f }; + if (!CompareFloats(result, known_from3, 4, __LINE__)) return false; + + v.ShuffleXXXX().Store(result); + static const float known_xxxx[4] = { 1.0f, 1.0f, 1.0f, 1.0f }; + if (!CompareFloats(result, known_xxxx, 4, __LINE__)) return false; + + v.ShuffleYYYY().Store(result); + static const float known_yyyy[4] = { 2.0f, 2.0f, 2.0f, 2.0f }; + if (!CompareFloats(result, known_yyyy, 4, __LINE__)) return false; + + v.ShuffleZZZZ().Store(result); + static const float known_zzzz[4] = { 3.0f, 3.0f, 3.0f, 3.0f }; + if (!CompareFloats(result, known_zzzz, 4, __LINE__)) return false; + + v.ShuffleWWWW().Store(result); + static const float known_wwww[4] = { 4.0f, 4.0f, 4.0f, 4.0f }; + if (!CompareFloats(result, known_wwww, 4, __LINE__)) return false; + + v.ShuffleXXYY().Store(result); + static const float known_xxyy[4] = { 1.0f, 1.0f, 2.0f, 2.0f }; + if (!CompareFloats(result, known_xxyy, 4, __LINE__)) return false; + + v.ShuffleZZWW().Store(result); + static const float known_zzww[4] = { 3.0f, 3.0f, 4.0f, 4.0f }; + if (!CompareFloats(result, known_zzww, 4, __LINE__)) return false; + + // Store2/Store3 must leave the rest of the destination alone. + float partial[4] = { -1.0f, -1.0f, -1.0f, -1.0f }; + v.Store2(partial); + static const float known_store2[4] = { 1.0f, 2.0f, -1.0f, -1.0f }; + if (!CompareFloats(partial, known_store2, 4, __LINE__)) return false; + + float partial3[4] = { -1.0f, -1.0f, -1.0f, -1.0f }; + v.Store3(partial3); + static const float known_store3[4] = { 1.0f, 2.0f, 3.0f, -1.0f }; + if (!CompareFloats(partial3, known_store3, 4, __LINE__)) return false; + + // Transpose of four columns. + static const float col_values[4][4] = { + { 0.0f, 1.0f, 2.0f, 3.0f }, + { 4.0f, 5.0f, 6.0f, 7.0f }, + { 8.0f, 9.0f, 10.0f, 11.0f }, + { 12.0f, 13.0f, 14.0f, 15.0f }, + }; + Vec4F32 c0 = Vec4F32::Load(col_values[0]); + Vec4F32 c1 = Vec4F32::Load(col_values[1]); + Vec4F32 c2 = Vec4F32::Load(col_values[2]); + Vec4F32 c3 = Vec4F32::Load(col_values[3]); + Vec4F32::Transpose(c0, c1, c2, c3); + float transposed[16]; + c0.Store(transposed); + c1.Store(transposed + 4); + c2.Store(transposed + 8); + c3.Store(transposed + 12); + for (int row = 0; row < 4; row++) { + for (int col = 0; col < 4; col++) { + EXPECT_EQ_FLOAT(transposed[row * 4 + col], col_values[col][row]); + } + } + + // LoadTranspose should match a plain Load of each row followed by Transpose. + static const float flat[16] = { + 0.0f, 1.0f, 2.0f, 3.0f, + 4.0f, 5.0f, 6.0f, 7.0f, + 8.0f, 9.0f, 10.0f, 11.0f, + 12.0f, 13.0f, 14.0f, 15.0f, + }; + Vec4F32 t0, t1, t2, t3; + Vec4F32::LoadTranspose(flat, t0, t1, t2, t3); + float loadTransposed[16]; + t0.Store(loadTransposed); + t1.Store(loadTransposed + 4); + t2.Store(loadTransposed + 8); + t3.Store(loadTransposed + 12); + if (!CompareFloats(loadTransposed, transposed, 16, __LINE__)) return false; + + return true; +} + +static bool TestVec4F32NaNInf() { + const float inf = INFINITY; + const float nan = NAN; + const float values[4] = { 1.0f, nan, inf, -inf }; + Vec4F32 v = Vec4F32::Load(values); + + float result[4]; + + v.ZeroNaNs().Store(result); + EXPECT_EQ_FLOAT(result[0], 1.0f); + EXPECT_EQ_FLOAT(result[1], 0.0f); + // ZeroNaNs leaves infinities alone. + EXPECT_TRUE(std::isinf(result[2])); + EXPECT_TRUE(std::isinf(result[3])); + + v.CleanNaNInfs().Store(result); + static const float known_clean[4] = { 1.0f, 0.0f, 0.0f, 0.0f }; + if (!CompareFloats(result, known_clean, 4, __LINE__)) return false; + + return true; +} + +static bool TestVec4F32Loads() { + float result[4]; + + // LoadF24x3_One: three 24-bit values shifted up into floats, lane 3 forced to 1.0f. + const uint32_t f24_values[4] = { 0x3F8000 >> 0, 0x400000, 0x3F0000, 0x123456 }; + Vec4F32::LoadF24x3_One(f24_values).Store(result); + for (int i = 0; i < 3; i++) { + float expected; + uint32_t bits = f24_values[i] << 8; + memcpy(&expected, &bits, 4); + EXPECT_EQ_FLOAT(result[i], expected); + } + EXPECT_EQ_FLOAT(result[3], 1.0f); + + // LoadF24x4: same, but all four lanes come from the source. + Vec4F32::LoadF24x4(f24_values).Store(result); + for (int i = 0; i < 4; i++) { + float expected; + uint32_t bits = f24_values[i] << 8; + memcpy(&expected, &bits, 4); + EXPECT_EQ_FLOAT(result[i], expected); + } + + // The normalizing loads. Some of these read 8 bytes, so give them room. + const int8_t s8_values[8] = { -1, -128, 127, 45, 0, 0, 0, 0 }; + Vec4F32::LoadS8Norm(s8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s8_values[i] / 128.0f); + } + + const int16_t s16_values[8] = { -1, -32768, 32767, 1234, 0, 0, 0, 0 }; + Vec4F32::LoadS16Norm(s16_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s16_values[i] / 32768.0f); + } + + const uint8_t u8_values[8] = { 0, 255, 128, 7, 0, 0, 0, 0 }; + Vec4F32::LoadU8Norm(u8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_APPROX_EQ_FLOAT(result[i], (float)u8_values[i] / 255.0f); + } + + // The converting (non-normalizing) loads. + Vec4F32::LoadConvertS16(s16_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s16_values[i]); + } + + Vec4F32::LoadConvertS8(s8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s8_values[i]); + } + + Vec4F32::LoadConvertU8(u8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)u8_values[i]); + } + + // StoreConvertToU8 truncates towards zero and saturates to 0-255. + const float to_u8_values[4] = { 0.0f, 255.0f, 12.75f, 200.0f }; + uint8_t u8_out[4]; + Vec4F32::Load(to_u8_values).StoreConvertToU8(u8_out); + EXPECT_EQ_INT(u8_out[0], 0); + EXPECT_EQ_INT(u8_out[1], 255); + EXPECT_EQ_INT(u8_out[2], 12); + EXPECT_EQ_INT(u8_out[3], 200); + + const float saturate_values[4] = { -1.0f, 300.0f, -0.5f, 1000.0f }; + Vec4F32::Load(saturate_values).StoreConvertToU8(u8_out); + EXPECT_EQ_INT(u8_out[0], 0); + EXPECT_EQ_INT(u8_out[1], 255); + EXPECT_EQ_INT(u8_out[2], 0); + EXPECT_EQ_INT(u8_out[3], 255); + + // Int to float. + const int int_values[4] = { 0, -7, 1000, -32768 }; + Vec4F32::FromVec4S32(Vec4S32::Load(int_values)).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)int_values[i]); + } + + // Load2 only fills the first two lanes. + const float two_values[2] = { 3.0f, 4.0f }; + Vec4F32::Load2(two_values).Store(result); + EXPECT_EQ_FLOAT(result[0], 3.0f); + EXPECT_EQ_FLOAT(result[1], 4.0f); + + return true; +} + +static bool TestMatrices() { + static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f }; + static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f }; + static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, }; + float result[16]; + Mat4F32 a(a_values); + Mat4F32 b(b_values); + + Mul4x4By4x4(a, b).Store(result); + if (!CompareFloats(result, known_result, 16, __LINE__)) { + return false; + } + + Mat4x3F32 d = Mat4x3F32(b_values + 2); + Mul4x3By4x4(d, a).Store(result); + + static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, }; + if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) { + return false; + } + + static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f }; + Vec4F32 v = Vec4F32::Load(vec_values); + + v.AsVec3ByMatrix44(b).Store3(result); + + static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, }; + if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) { + return false; + } + Vec4F32 scale = Vec4F32::Load(a_values); + Vec4F32 translate = Vec4F32::Load(b_values); + + TranslateAndScaleInplace(a, scale, translate); + a.Store(result); + + static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,}; + if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) { + return false; + } + + return true; +} + +bool TestCrossSIMD() { + if (!TestVec4S32()) return false; + if (!TestVec4S32Compares()) return false; + if (!TestVec4F32Arith()) return false; + if (!TestVec4F32Compares()) return false; + if (!TestVec4F32Lanes()) return false; + if (!TestVec4F32NaNInf()) return false; + if (!TestVec4F32Loads()) return false; + if (!TestMatrices()) return false; + return true; +} diff --git a/unittest/UnitTest.cpp b/unittest/UnitTest.cpp index ebb9dc7cd7..383fe445b0 100644 --- a/unittest/UnitTest.cpp +++ b/unittest/UnitTest.cpp @@ -2764,88 +2764,6 @@ bool TestSIMD() { return true; } -static void PrintFloats(const float *f, int count) { - for (int i = 0; i < count; i++) { - printf("%.1ff, ", f[i]); - } - printf("\n"); -} - -static bool CompareFloats(const float *values, const float *known_good, int count, int line) { - int wrongCount = 0; - - for (int i = 0; i < count; i++) { - if (values[i] != known_good[i]) { - wrongCount++; - } - } - - if (wrongCount > 0) { - for (int i = 0; i < count; i++) { - bool wrong = values[i] != known_good[i]; - printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : ""); - } - printf("At UnitTest.cpp:%d: %d / %d were wrong\n", line, wrongCount, count); - return false; - } else { - return true; - } -} - -bool TestCrossSIMD() { - static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f }; - static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f }; - static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, }; - float result[16]; - Mat4F32 a(a_values); - Mat4F32 b(b_values); - - Mul4x4By4x4(a, b).Store(result); - if (!CompareFloats(result, known_result, 16, __LINE__)) { - return false; - } - - Mat4x3F32 d = Mat4x3F32(b_values + 2); - Mul4x3By4x4(d, a).Store(result); - - static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, }; - if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) { - return false; - } - - static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f }; - Vec4F32 v = Vec4F32::Load(vec_values); - - v.AsVec3ByMatrix44(b).Store3(result); - - static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, }; - if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) { - return false; - } - Vec4F32 scale = Vec4F32::Load(a_values); - Vec4F32 translate = Vec4F32::Load(b_values); - - TranslateAndScaleInplace(a, scale, translate); - a.Store(result); - - static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,}; - if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) { - return false; - } - - s8 values[4] = {-1, -128, 127, 45}; - float fvalues[4]; - Vec4F32::LoadS8Norm(values).Store(fvalues); - static const float known_s8norm_result[4] = {(float)values[0]/128.0f, (float)values[1]/128.0f, (float)values[2]/128.0f, (float)values[3]/128.0f,}; - if (!CompareFloats(fvalues, known_s8norm_result, ARRAY_SIZE(known_s8norm_result), __LINE__)) { - return false; - } - - // PrintFloats(result, 16); - - return true; -} - bool TestVolumeFunc() { for (int i = 0; i <= 20; i++) { float mul = Volume10ToMultiplier(i); @@ -3044,6 +2962,7 @@ struct TestItem { bool TestArmEmitter(); bool TestArm64Emitter(); +bool TestCrossSIMD(); bool TestX64Emitter(); bool TestRiscVEmitter(); bool TestLoongArch64Emitter(); diff --git a/unittest/UnitTests.vcxproj b/unittest/UnitTests.vcxproj index 5793f24e1e..e77fb4ba3f 100644 --- a/unittest/UnitTests.vcxproj +++ b/unittest/UnitTests.vcxproj @@ -287,6 +287,7 @@ + diff --git a/unittest/UnitTests.vcxproj.filters b/unittest/UnitTests.vcxproj.filters index 37af8831d9..9d249d43dd 100644 --- a/unittest/UnitTests.vcxproj.filters +++ b/unittest/UnitTests.vcxproj.filters @@ -14,6 +14,7 @@ + From e86b2dcc8119fa6a706a4499b39712c756f3dc84 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 12:41:13 -0600 Subject: [PATCH 7/9] headless: allow the riscv64 cross build Same treatment the loongarch64 cross build already gets - its sysroot has no SDL3, so there's no windowed graphics context to create. Both run fine under qemu with --graphics=software, which needs no context at all, so correct the comment that called this compile-tested only. Co-Authored-By: Claude Opus 5 (1M context) --- headless/Headless.cpp | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/headless/Headless.cpp b/headless/Headless.cpp index 904ad8b1ff..a2785c48ec 100644 --- a/headless/Headless.cpp +++ b/headless/Headless.cpp @@ -289,10 +289,10 @@ static GraphicsContext *CreateGraphicsContext(GPUCore gpuCore, std::string **dev default: return nullptr; } -#elif PPSSPP_ARCH(LOONGARCH64) - // The loongarch64 cross-compilation toolchain has no SDL3 packages available (see the - // LOONGARCH64_DEVICE branch in CMakeLists.txt), so this build is compile-tested only and - // never actually needs to create a graphics context at runtime. +#elif PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64) + // The loongarch64 and riscv64 cross-compilation sysroots have no SDL3 (see the HEADLESS_CROSS + // branch in CMakeLists.txt). These builds still run fine under qemu with --graphics=software, + // which needs no graphics context. *deviceSetting = nullptr; return nullptr; #elif PPSSPP_PLATFORM(ANDROID) @@ -900,7 +900,7 @@ int main(int argc, const char* argv[]) { // We don't bother with a window. graphicsContext = new NullGraphicsContext(); } else { -#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) +#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64) fprintf(stderr, "Headless graphics context creation is not supported on this platform.\n"); return 1; #else @@ -1130,7 +1130,7 @@ int main(int argc, const char* argv[]) { ShutdownWebServer(); } -#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) +#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64) // ... see above #else if (window) { From f3d63f3c7ef636cba3fef175e75d00f47ffab9d4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 13:21:35 -0600 Subject: [PATCH 8/9] RISC-V: mask the register number in the compressed encoders EncodeCR, EncodeCI and EncodeCSS shifted the RiscVReg enum value straight into the instruction without DecodeReg(), which every 32-bit encoder in the file uses. FPRs are 0x20..0x3F in that enum, so bit 5 is set for all of them and spilled into a neighbouring field - for the CI format, into bit 12, which holds imm[5]. The visible effect: the dispatcher's epilogue restores the callee-saved FP registers with c.fldsp, and every offset whose bit 5 was clear came out 32 bytes too high. fs3-fs6 were restored from the wrong slots and fs11 from 224(sp), past the 208-byte frame. So the native JIT corrupted whatever the C++ caller had in those registers - which, in the headless test runner, was the wall-clock deadline, making every test report an instant TIMEOUT. pspautotests on riscv64 under qemu goes from 0/342 to 338/342 with this, matching the IR interpreter exactly. Found with the disassembly the enableDisasm flag in RiscVAsm.cpp emits - the save offsets and the restore offsets simply didn't match. Co-Authored-By: Claude Opus 5 (1M context) --- Common/RiscVEmitter.cpp | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/Common/RiscVEmitter.cpp b/Common/RiscVEmitter.cpp index 17dda6ffed..8232eb1d51 100644 --- a/Common/RiscVEmitter.cpp +++ b/Common/RiscVEmitter.cpp @@ -842,7 +842,7 @@ static inline u32 EncodeFVF(RiscVReg vd, RiscVReg rs1, RiscVReg vs2, VUseMask vm static inline u16 EncodeCR(Opcode16 op, RiscVReg rs2, RiscVReg rd, Funct4 funct4) { _assert_msg_(SupportsCompressed(), "Compressed instructions unsupported"); - return (u16)op | ((u16)rs2 << 2) | ((u16)rd << 7) | ((u16)funct4 << 12); + return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)DecodeReg(rd) << 7) | ((u16)funct4 << 12); } static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) { @@ -850,13 +850,13 @@ static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) { _assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6); u16 imm4_0 = ImmBits16(uimm6, 0, 5); u16 imm5 = ImmBit16(uimm6, 5); - return (u16)op | (imm4_0 << 2) | ((u16)rd << 7) | (imm5 << 12) | ((u16)funct3 << 13); + return (u16)op | (imm4_0 << 2) | ((u16)DecodeReg(rd) << 7) | (imm5 << 12) | ((u16)funct3 << 13); } static inline u16 EncodeCSS(Opcode16 op, RiscVReg rs2, u8 uimm6, Funct3 funct3) { _assert_msg_(SupportsCompressed(), "Compressed instructions unsupported"); _assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6); - return (u16)op | ((u16)rs2 << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13); + return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13); } static inline u16 EncodeCIW(Opcode16 op, RiscCReg rd, u8 uimm8, Funct3 funct3) { From cd9bb569a3400eb833c77ed5dc1f34d7522bbea9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 19 Sep 2026 13:21:36 -0600 Subject: [PATCH 9/9] unittest: don't pin CleanNaNInfs to one implementation's output The contract is only that a bad lane becomes something that yields zero when multiplied by zero, by whatever route is cheapest. SSE2 clamps to +-FLT_MAX, NEON, LSX and the scalar fallback zero the lane - both fine. Check that property, and that good lanes are untouched. Co-Authored-By: Claude Opus 5 (1M context) --- unittest/TestCrossSIMD.cpp | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/unittest/TestCrossSIMD.cpp b/unittest/TestCrossSIMD.cpp index e584bf762b..9ad066a315 100644 --- a/unittest/TestCrossSIMD.cpp +++ b/unittest/TestCrossSIMD.cpp @@ -391,9 +391,15 @@ static bool TestVec4F32NaNInf() { EXPECT_TRUE(std::isinf(result[2])); EXPECT_TRUE(std::isinf(result[3])); + // The contract is just "a value that yields zero when multiplied by zero" - whatever is + // cheapest to get there. So the implementations legitimately differ on what a bad lane + // becomes: SSE2 clamps to +-FLT_MAX, NEON, LSX and the scalar fallback zero it. Check the + // property rather than the value, and that good lanes are left alone. v.CleanNaNInfs().Store(result); - static const float known_clean[4] = { 1.0f, 0.0f, 0.0f, 0.0f }; - if (!CompareFloats(result, known_clean, 4, __LINE__)) return false; + EXPECT_EQ_FLOAT(result[0], 1.0f); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i] * 0.0f, 0.0f); + } return true; }