diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 02a3de140e..0f16faf1f6 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -288,6 +288,13 @@ jobs: args: ./b.sh --loongarch64 PPSSPPHeadless id: loongarch64 + - os: ubuntu-26.04 + extra: riscv64 + cc: gcc + cxx: g++ + args: ./b.sh --riscv64 PPSSPPHeadless + id: riscv64 + - os: macos-latest extra: test cc: clang @@ -327,7 +334,7 @@ jobs: ndk-version: r29 - name: Install Linux dependencies - if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64' + if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64' && matrix.extra != 'riscv64' run: | sudo apt-get update -y -qq sudo apt-get install libsdl3-dev libgl1-mesa-dev libglu1-mesa-dev libsdl3-ttf-dev libfontconfig1-dev libcurl4-openssl-dev @@ -336,12 +343,12 @@ jobs: if: runner.os == 'macOS' && matrix.id == 'macos' run: brew install sdl3 sdl3_ttf - - name: Install loongarch64 cross-compilation dependencies - if: matrix.extra == 'loongarch64' + - name: Install cross-compilation dependencies + if: matrix.extra == 'loongarch64' || matrix.extra == 'riscv64' run: | sudo apt-get update -y -qq - sudo apt-get install -y gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu libgl-dev libglu1-mesa-dev - sudo cmake/scripts/setup-loongarch64-cross.sh + sudo apt-get install -y gcc-14-${{ matrix.extra }}-linux-gnu g++-14-${{ matrix.extra }}-linux-gnu binutils-${{ matrix.extra }}-linux-gnu libgl-dev libglu1-mesa-dev + sudo cmake/scripts/setup-cross.sh ${{ matrix.extra }} - name: Install iOS dependencies if: matrix.id == 'ios' diff --git a/CMakeLists.txt b/CMakeLists.txt index c81d415524..65e2b49ca7 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -184,6 +184,8 @@ option(USE_VULKAN_DISPLAY_KHR "Enable or disable full screen display of Vulkan" # :: Frontends option(MOBILE_DEVICE "Set to ON when targeting a mobile device" ${MOBILE_DEVICE}) option(HEADLESS "Set to OFF to not generate the PPSSPPHeadless target" ${HEADLESS}) +# Set by b.sh for the loongarch64/riscv64 cross builds, whose sysroots have no SDL3. +option(HEADLESS_CROSS "Set to ON for a headless-only cross build (no SDL frontend)" ${HEADLESS_CROSS}) option(ATLAS_TOOL "Set to OFF to not generate the ATLAS_TOOL target" ${ATLAS_TOOL}) option(UNITTEST "Set to ON to generate the unittest target" ${UNITTEST}) option(SIMULATOR "Set to ON when targeting an x86 simulator of an ARM platform" ${SIMULATOR}) @@ -331,7 +333,7 @@ set(SDL_LIB_TARGET "") set(SDL_TTF_LIB_TARGET "") set(FONTCONFIG_LIB_TARGET "") -if(NOT LIBRETRO AND NOT IOS AND NOT LOONGARCH64_DEVICE AND NOT ANDROID) +if(NOT LIBRETRO AND NOT IOS AND NOT HEADLESS_CROSS AND NOT ANDROID) find_package(SDL3 QUIET) if(NOT SDL3_FOUND) if(CMAKE_SYSTEM_NAME STREQUAL "Linux" AND (EXISTS "/usr/bin/apt" OR EXISTS "/usr/bin/apt-get" OR EXISTS "/etc/debian_version")) @@ -1001,9 +1003,11 @@ elseif(WIN32) # Don't care about SDL. set(TargetBin PPSSPPWindows) elseif(LIBRETRO) -elseif(LOONGARCH64_DEVICE) - # Cross-compiling for loongarch64 currently only builds PPSSPPHeadless, and there's - # no SDL3 available for that target, so skip the desktop SDL frontend entirely. +elseif(HEADLESS_CROSS) + # The loongarch64/riscv64 cross builds only build PPSSPPHeadless, and there's no + # SDL3 available in those sysroots, so skip the desktop SDL frontend entirely. + # Note this is deliberately not keyed off the target architecture: a native build + # on such a machine can have SDL3 and should still get the normal frontend. else() if(GOLD) set(TargetBin PPSSPPGold) @@ -1362,6 +1366,7 @@ if(UNITTEST) unittest/TestShaderGenerators.cpp unittest/TestArmEmitter.cpp unittest/TestArm64Emitter.cpp + unittest/TestCrossSIMD.cpp unittest/TestIRPassSimplify.cpp unittest/TestX64Emitter.cpp unittest/TestVertexJit.cpp diff --git a/Common/LoongArch64Emitter.h b/Common/LoongArch64Emitter.h index 134b5de61d..d5abda3809 100644 --- a/Common/LoongArch64Emitter.h +++ b/Common/LoongArch64Emitter.h @@ -166,7 +166,8 @@ public: QuickCallFunction((const u8 *)func, scratchreg); } void QuickCallFunctionR(const u8 *func, LoongArch64Reg arg, LoongArch64Reg scratchreg = R_RA) { - MOVE(LoongArch64Reg::X4, arg); + // The first integer argument goes in a0, which is R4 - X4 is an LASX vector register. + MOVE(LoongArch64Reg::R4, arg); QuickJump(scratchreg, R_RA, func); } template diff --git a/Common/Math/CrossSIMD.h b/Common/Math/CrossSIMD.h index af30c25b06..8c9316b671 100644 --- a/Common/Math/CrossSIMD.h +++ b/Common/Math/CrossSIMD.h @@ -6,6 +6,7 @@ #pragma once +#include #include #include "Common/Math/SIMDHeaders.h" @@ -1187,8 +1188,11 @@ struct Vec4F32 { } void StoreConvertToU8(uint8_t *dest) { __m128i ivalue32 = __lsx_vftintrz_w_s(v); - __m128i ivalue16 = __lsx_vssrlrni_hu_w(ivalue32, ivalue32, 0); - __m128i ivalue8 = __lsx_vssrlrni_bu_h(ivalue16, ivalue16, 0); + // Narrow signed->signed first, then signed->unsigned, matching the packs/packus pair the + // SSE version uses. The logical (unsigned) narrowing shifts turn a negative value into a + // huge unsigned one, which saturates to 255 instead of clamping to 0. + __m128i ivalue16 = __lsx_vssrani_h_w(ivalue32, ivalue32, 0); + __m128i ivalue8 = __lsx_vssrani_bu_h(ivalue16, ivalue16, 0); uint32_t value = __lsx_vpickve2gr_wu(ivalue8, 0); memcpy(dest, &value, sizeof(uint32_t)); } @@ -1198,10 +1202,10 @@ struct Vec4F32 { // index is a compile-time constant parameter to the intrinsic. int ival; switch (index) { - case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); - case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); - case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); - default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); + case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); break; + case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); break; + case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); break; + default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); break; } float fval; memcpy(&fval, &ival, sizeof(float)); @@ -1671,15 +1675,38 @@ struct Vec4F32 { return temp; } + static Vec4F32 LoadConvertU8(const uint8_t *src) { // Note: will load 8 bytes, not 4. Only the first 4 bytes will be used. + Vec4F32 temp; + for (int i = 0; i < 4; i++) { + temp.v[i] = (float)src[i]; + } + return temp; + } + + // Truncates towards zero and saturates to 0-255, like the packing instructions the other + // implementations use. + void StoreConvertToU8(uint8_t *dest) { + for (int i = 0; i < 4; i++) { + int value = (int)v[i]; + if (value < 0) value = 0; + if (value > 255) value = 255; + dest[i] = (uint8_t)value; + } + } + static Vec4F32 LoadF24x3_One(const uint32_t *src) { - uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, 0 }; + constexpr uint32_t kOneF32Bits = 0x3F800000; + uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, kOneF32Bits }; Vec4F32 temp; memcpy(temp.v, shifted, sizeof(temp.v)); return temp; } static Vec4F32 LoadF24x4(const uint32_t *src) { - return LoadR24x3_One(src); + uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, src[3] << 8 }; + Vec4F32 temp; + memcpy(temp.v, shifted, sizeof(temp.v)); + return temp; } static Vec4F32 FromVec4S32(Vec4S32 src) { @@ -1695,7 +1722,7 @@ struct Vec4F32 { Vec4F32 ZeroNaNs() const { Vec4F32 temp; for (int i = 0; i < 4; i++) { - temp.v[i] = isnan(v[i]) ? 0.0f : v[i]; + temp.v[i] = std::isnan(v[i]) ? 0.0f : v[i]; } return temp; } @@ -1703,7 +1730,7 @@ struct Vec4F32 { Vec4F32 CleanNaNInfs() { Vec4F32 temp; for (int i = 0; i < 4; i++) { - temp.v[i] = (isnan(v[i]) || isinf(v[i])) ? 0.0f : v[i]; + temp.v[i] = (std::isnan(v[i]) || std::isinf(v[i])) ? 0.0f : v[i]; } return temp; } @@ -1798,6 +1825,10 @@ struct Vec4F32 { return Vec4F32{ { v[0], v[1], v[2], 1.0f } }; } + Vec4F32 WithLane3From(Vec4F32 other) const { + return Vec4F32{ { v[0], v[1], v[2], other.v[3] } }; + } + Vec4S32 CompareEq(Vec4F32 other) const { Vec4S32 temp; for (int i = 0; i < 4; i++) { @@ -1857,6 +1888,15 @@ struct Vec4F32 { return Vec4F32{{ v[3], v[3], v[3], v[3] }}; } + // Loads four rows of four floats and transposes them into columns. + static void LoadTranspose(const float *src, Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) { + col0 = Vec4F32::Load(src); + col1 = Vec4F32::Load(src + 4); + col2 = Vec4F32::Load(src + 8); + col3 = Vec4F32::Load(src + 12); + Transpose(col0, col1, col2, col3); + } + // In-place transpose. static void Transpose(Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) { float m[16]; @@ -1919,6 +1959,10 @@ inline bool AllCompareBitsSet(Vec4S32 value) { return true; } +inline bool AnyCompareBitsSet(Vec4S32 value) { + return value.v[0] != 0 || value.v[1] != 0 || value.v[2] != 0 || value.v[3] != 0; +} + struct Vec4U16 { uint16_t v[4]; // 64 bits. diff --git a/Common/RiscVEmitter.cpp b/Common/RiscVEmitter.cpp index 17dda6ffed..8232eb1d51 100644 --- a/Common/RiscVEmitter.cpp +++ b/Common/RiscVEmitter.cpp @@ -842,7 +842,7 @@ static inline u32 EncodeFVF(RiscVReg vd, RiscVReg rs1, RiscVReg vs2, VUseMask vm static inline u16 EncodeCR(Opcode16 op, RiscVReg rs2, RiscVReg rd, Funct4 funct4) { _assert_msg_(SupportsCompressed(), "Compressed instructions unsupported"); - return (u16)op | ((u16)rs2 << 2) | ((u16)rd << 7) | ((u16)funct4 << 12); + return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)DecodeReg(rd) << 7) | ((u16)funct4 << 12); } static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) { @@ -850,13 +850,13 @@ static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) { _assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6); u16 imm4_0 = ImmBits16(uimm6, 0, 5); u16 imm5 = ImmBit16(uimm6, 5); - return (u16)op | (imm4_0 << 2) | ((u16)rd << 7) | (imm5 << 12) | ((u16)funct3 << 13); + return (u16)op | (imm4_0 << 2) | ((u16)DecodeReg(rd) << 7) | (imm5 << 12) | ((u16)funct3 << 13); } static inline u16 EncodeCSS(Opcode16 op, RiscVReg rs2, u8 uimm6, Funct3 funct3) { _assert_msg_(SupportsCompressed(), "Compressed instructions unsupported"); _assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6); - return (u16)op | ((u16)rs2 << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13); + return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13); } static inline u16 EncodeCIW(Opcode16 op, RiscCReg rd, u8 uimm8, Funct3 funct3) { diff --git a/android/jni/Android.mk b/android/jni/Android.mk index 2e0a7ab5a9..b308bc058d 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -1051,6 +1051,7 @@ ifeq ($(UNITTEST),1) $(SRC)/unittest/TestX64Emitter.cpp \ $(SRC)/unittest/TestRiscVEmitter.cpp \ $(SRC)/unittest/TestLoongArch64Emitter.cpp \ + $(SRC)/unittest/TestCrossSIMD.cpp \ $(SRC)/unittest/TestIRPassSimplify.cpp \ $(SRC)/unittest/TestShaderGenerators.cpp \ $(SRC)/unittest/TestSoftwareGPUJit.cpp \ diff --git a/b.sh b/b.sh index a7683d0fb0..ff80e9bdd1 100755 --- a/b.sh +++ b/b.sh @@ -31,9 +31,14 @@ do CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/raspberry.armv8.cmake ${CMAKE_ARGS}" ;; --loongarch64) - CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}" + CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}" TARGET_OS=loongarch64 - LOONGARCH64_BUILD=1 + CROSS_STUB_CC=loongarch64-linux-gnu-gcc-14 + ;; + --riscv64) + CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/riscv64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}" + TARGET_OS=riscv64 + CROSS_STUB_CC=riscv64-linux-gnu-gcc-14 ;; --android) CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=android/android.toolchain.cmake ${CMAKE_ARGS}" TARGET_OS=Android @@ -122,28 +127,29 @@ echo Building with $CORES_COUNT threads mkdir -p ${BUILD_DIR} -# For loongarch64 cross-compilation, build a comprehensive GL/GLX stub into +# For the headless cross targets, build a comprehensive GL/GLX stub into # /stublibs/libGL.so so GLEW's static archive can resolve its symbols -# via PLT entries (a direct B26 branch to address 0 overflows on LoongArch). +# via PLT entries (a direct branch to address 0 overflows the relocation). # This stub is always (re)generated to pick up any new needed symbols. -if [ ! -z "$LOONGARCH64_BUILD" ]; then +if [ ! -z "$CROSS_STUB_CC" ]; then STUB_DIR=${BUILD_DIR}/stublibs STUB_GL=${STUB_DIR}/libGL.so mkdir -p "${STUB_DIR}" STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c) - echo "/* Loongarch64 GL/GLX stub - cross-compilation only */" > "$STUB_C" + echo "/* ${TARGET_OS} GL/GLX stub - cross-compilation only */" > "$STUB_C" # Collect all T (exported) symbols from GL/GLX libs, deduplicate, emit stubs + HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) { - for lib in /usr/lib/x86_64-linux-gnu/libGL.so.1 \ - /usr/lib/x86_64-linux-gnu/libGLX.so.0 \ - /usr/lib/x86_64-linux-gnu/libGLdispatch.so.0; do + for lib in /usr/lib/$HOST_MULTIARCH/libGL.so.1 \ + /usr/lib/$HOST_MULTIARCH/libGLX.so.0 \ + /usr/lib/$HOST_MULTIARCH/libGLdispatch.so.0; do [ -f "$lib" ] && nm -D "$lib" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print $3 }' done # Always include the minimal GLX symbols GLEW directly references printf '%s\n' glXGetProcAddressARB glXGetClientString glXQueryVersion \ glBindTexture glGetString glGetIntegerv } | sort -u | awk '{ print "void "$1"(void){}" }' >> "$STUB_C" - loongarch64-linux-gnu-gcc-14 -shared -fPIC -Wno-implicit-function-declaration \ + $CROSS_STUB_CC -shared -fPIC -Wno-implicit-function-declaration \ -o "${STUB_GL}" "$STUB_C" rm "$STUB_C" fi diff --git a/cmake/Toolchains/loongarch64-linux-gnu.cmake b/cmake/Toolchains/loongarch64-linux-gnu.cmake index b39ed83905..014630439c 100644 --- a/cmake/Toolchains/loongarch64-linux-gnu.cmake +++ b/cmake/Toolchains/loongarch64-linux-gnu.cmake @@ -17,8 +17,8 @@ set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE) # Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so # rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have. # A stub libGL.so + GL headers must be present in the sysroot; they are placed -# there by cmake/scripts/setup-loongarch64-cross.sh (run once with sudo, or -# automatically by the CI install step). +# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically +# by the CI install step). set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE) # b.sh also creates a stub in /stublibs/ as a local fallback so the @@ -27,14 +27,18 @@ set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs" CACHE STRING "" FORCE) -# Allow CMake's compile-check programs to run via QEMU. -# The /lib64 symlink is created by the setup script; without it, pass -L -# explicitly so QEMU can find the loongarch64 dynamic linker. -if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1") - set(CMAKE_CROSSCOMPILING_EMULATOR "qemu-loongarch64-static" - CACHE STRING "" FORCE) -else() - set(CMAKE_CROSSCOMPILING_EMULATOR - "qemu-loongarch64-static;-L;/usr/loongarch64-linux-gnu" - CACHE STRING "" FORCE) +# Allow CMake's compile-check programs to run via QEMU, if it's installed. +# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic +# ones in qemu-user; either will do, so take whichever is present. +find_program(QEMU_LOONGARCH64 NAMES qemu-loongarch64-static qemu-loongarch64) +if(QEMU_LOONGARCH64) + # The /lib64 symlink is created by the setup script; without it, pass -L + # explicitly so QEMU can find the loongarch64 dynamic linker. + if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1") + set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_LOONGARCH64}" CACHE STRING "" FORCE) + else() + set(CMAKE_CROSSCOMPILING_EMULATOR + "${QEMU_LOONGARCH64};-L;/usr/loongarch64-linux-gnu" + CACHE STRING "" FORCE) + endif() endif() diff --git a/cmake/Toolchains/riscv64-linux-gnu.cmake b/cmake/Toolchains/riscv64-linux-gnu.cmake new file mode 100644 index 0000000000..a80150c6e8 --- /dev/null +++ b/cmake/Toolchains/riscv64-linux-gnu.cmake @@ -0,0 +1,44 @@ +set(CMAKE_SYSTEM_NAME Linux) +set(CMAKE_SYSTEM_PROCESSOR riscv64) + +set(CMAKE_C_COMPILER riscv64-linux-gnu-gcc-14) +set(CMAKE_CXX_COMPILER riscv64-linux-gnu-g++-14) +set(CMAKE_ASM_COMPILER riscv64-linux-gnu-gcc-14) + +set(CMAKE_FIND_ROOT_PATH /usr/riscv64-linux-gnu) +set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER) +set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY) +set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY) + +# No X11 in the sysroot; disable the X11 Vulkan path to avoid include-dir errors. +set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE) + +# Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so +# rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have. +# A stub libGL.so + GL headers must be present in the sysroot; they are placed +# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically +# by the CI install step). +set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE) + +# b.sh also creates a stub in /stublibs/ as a local fallback so the +# linker can resolve GL/GLX calls from GLEW's static archive via PLT entries. +set(CMAKE_EXE_LINKER_FLAGS + "${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs" + CACHE STRING "" FORCE) + +# Allow CMake's compile-check programs to run via QEMU, if it's installed. +# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic +# ones in qemu-user; either will do, so take whichever is present. +find_program(QEMU_RISCV64 NAMES qemu-riscv64-static qemu-riscv64) +if(QEMU_RISCV64) + # The /lib64 symlink is created by the setup script; without it, pass -L + # explicitly so QEMU can find the riscv64 dynamic linker. + if(EXISTS "/lib64/ld-linux-riscv64-lp64d.so.1") + set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_RISCV64}" CACHE STRING "" FORCE) + else() + set(CMAKE_CROSSCOMPILING_EMULATOR + "${QEMU_RISCV64};-L;/usr/riscv64-linux-gnu" + CACHE STRING "" FORCE) + endif() +endif() diff --git a/cmake/scripts/setup-loongarch64-cross.sh b/cmake/scripts/setup-cross.sh similarity index 58% rename from cmake/scripts/setup-loongarch64-cross.sh rename to cmake/scripts/setup-cross.sh index 978dbcebc1..ea9b8df86b 100755 --- a/cmake/scripts/setup-loongarch64-cross.sh +++ b/cmake/scripts/setup-cross.sh @@ -1,15 +1,34 @@ #!/bin/bash -# Sets up the loongarch64 cross-compilation environment. -# Run once with sudo. After this, ./b.sh --loongarch64 works on a fresh checkout. +# Sets up a cross-compilation environment for one of the headless-only targets. +# Run once with sudo. After this, ./b.sh -- works on a fresh checkout. +# +# Usage: sudo cmake/scripts/setup-cross.sh set -e -SYSROOT=/usr/loongarch64-linux-gnu -CROSS_GCC=loongarch64-linux-gnu-gcc-14 +TARGET="$1" + +case "$TARGET" in + loongarch64) + TRIPLE=loongarch64-linux-gnu + LD_SO=ld-linux-loongarch-lp64d.so.1 + ;; + riscv64) + TRIPLE=riscv64-linux-gnu + LD_SO=ld-linux-riscv64-lp64d.so.1 + ;; + *) + echo "Usage: sudo $0 " + exit 1 + ;; +esac + +SYSROOT=/usr/$TRIPLE +CROSS_GCC=$TRIPLE-gcc-14 if ! command -v $CROSS_GCC &>/dev/null; then echo "Error: $CROSS_GCC not found. Install it first:" - echo " sudo apt install gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu" + echo " sudo apt install gcc-14-$TRIPLE g++-14-$TRIPLE binutils-$TRIPLE" exit 1 fi @@ -18,6 +37,8 @@ if [ "$(id -u)" -ne 0 ]; then exit 1 fi +HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null) + # ── OpenGL headers ─────────────────────────────────────────────────────────── # GL headers are platform-independent C headers; copy from the host. echo "Installing GL headers into sysroot..." @@ -27,22 +48,22 @@ for h in gl.h glext.h glcorearb.h glu.h; do done # ── GL stub library ────────────────────────────────────────────────────────── -# Generate a loongarch64 libGL.so that exports all symbols from the host +# Generate a libGL.so for the target that exports all symbols from the host # libGL so that GLEW's static archive can be linked via PLT entries (a direct -# B26 branch to address 0 would overflow on LoongArch). +# branch to address 0 would overflow the relocation on these targets). echo "Generating libGL.so stub..." HOST_GL="" for candidate in \ - /usr/lib/x86_64-linux-gnu/libGL.so.1 \ - /usr/lib/x86_64-linux-gnu/libGL.so \ + /usr/lib/$HOST_MULTIARCH/libGL.so.1 \ + /usr/lib/$HOST_MULTIARCH/libGL.so \ /usr/lib/libGL.so.1 ; do [ -f "$candidate" ] && HOST_GL="$candidate" && break done STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c) -echo "/* Loongarch64 GL/GLX stub - for cross-compilation only */" > "$STUB_C" -echo "void __loongarch64_gl_placeholder(void) {}" >> "$STUB_C" +echo "/* $TARGET GL/GLX stub - for cross-compilation only */" > "$STUB_C" +echo "void __cross_gl_placeholder(void) {}" >> "$STUB_C" if [ -n "$HOST_GL" ]; then nm -D "$HOST_GL" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print "void "$3"(void){}" }' >> "$STUB_C" @@ -60,16 +81,16 @@ rm "$STUB_C" echo " Installed to $SYSROOT/lib/libGL.so" # ── Dynamic linker symlink ─────────────────────────────────────────────────── -# Without this, binfmt_misc can't find the loongarch64 dynamic linker. -LD_LINUX=$SYSROOT/lib/ld-linux-loongarch-lp64d.so.1 +# Without this, binfmt_misc can't find the target's dynamic linker. +LD_LINUX=$SYSROOT/lib/$LD_SO if [ -f "$LD_LINUX" ]; then mkdir -p /lib64 - ln -sf "$LD_LINUX" /lib64/ld-linux-loongarch-lp64d.so.1 - echo "Created /lib64/ld-linux-loongarch-lp64d.so.1 -> $LD_LINUX" + ln -sf "$LD_LINUX" /lib64/$LD_SO + echo "Created /lib64/$LD_SO -> $LD_LINUX" fi echo "" echo "Setup complete." -echo "Build: ./b.sh --loongarch64" -echo "Run: qemu-loongarch64-static -L $SYSROOT " -echo " (after setup the /lib64 symlink lets you run loongarch64 binaries directly)" +echo "Build: ./b.sh --$TARGET" +echo "Run: qemu-$TARGET -L $SYSROOT " +echo " (after setup the /lib64 symlink lets you run $TARGET binaries directly)" diff --git a/headless/Headless.cpp b/headless/Headless.cpp index dfc5b25268..0bdd823435 100644 --- a/headless/Headless.cpp +++ b/headless/Headless.cpp @@ -289,10 +289,10 @@ static GraphicsContext *CreateGraphicsContext(GPUCore gpuCore, std::string **dev default: return nullptr; } -#elif PPSSPP_ARCH(LOONGARCH64) - // The loongarch64 cross-compilation toolchain has no SDL3 packages available (see the - // LOONGARCH64_DEVICE branch in CMakeLists.txt), so this build is compile-tested only and - // never actually needs to create a graphics context at runtime. +#elif PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64) + // The loongarch64 and riscv64 cross-compilation sysroots have no SDL3 (see the HEADLESS_CROSS + // branch in CMakeLists.txt). These builds still run fine under qemu with --graphics=software, + // which needs no graphics context. *deviceSetting = nullptr; return nullptr; #elif PPSSPP_PLATFORM(ANDROID) @@ -904,7 +904,7 @@ int main(int argc, const char* argv[]) { // We don't bother with a window. graphicsContext = new NullGraphicsContext(); } else { -#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) +#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64) fprintf(stderr, "Headless graphics context creation is not supported on this platform.\n"); return 1; #else @@ -1134,7 +1134,7 @@ int main(int argc, const char* argv[]) { ShutdownWebServer(); } -#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) +#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64) // ... see above #else if (window) { diff --git a/unittest/TestCrossSIMD.cpp b/unittest/TestCrossSIMD.cpp new file mode 100644 index 0000000000..9ad066a315 --- /dev/null +++ b/unittest/TestCrossSIMD.cpp @@ -0,0 +1,551 @@ +// Copyright (c) 2012- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +// Tests for Common/Math/CrossSIMD.h. +// +// CrossSIMD has four independent implementations - SSE2, NEON, LoongArch LSX and a plain scalar +// fallback - and only one of them is compiled on any given machine. So the point of these tests is +// to check whatever got compiled against results worked out by hand, rather than against each other; +// that way every architecture we build for verifies its own copy. Running the unit tests under +// qemu covers the ones we have no hardware for. +// +// Only functions that all four implementations provide are tested here, otherwise this wouldn't +// build everywhere. + +#include "ppsspp_config.h" + +#include +#include +#include + +#include "Common/Common.h" +#include "Common/CommonTypes.h" +#include "Common/Math/CrossSIMD.h" +#include "unittest/UnitTest.h" + +static bool CompareFloats(const float *values, const float *known_good, int count, int line) { + int wrongCount = 0; + for (int i = 0; i < count; i++) { + if (values[i] != known_good[i]) { + wrongCount++; + } + } + if (wrongCount > 0) { + for (int i = 0; i < count; i++) { + bool wrong = values[i] != known_good[i]; + printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : ""); + } + printf("At TestCrossSIMD.cpp:%d: %d / %d were wrong\n", line, wrongCount, count); + return false; + } + return true; +} + +// For the approximate operations (reciprocals), which are allowed to differ between architectures. +static bool CompareFloatsApprox(const float *values, const float *known_good, int count, float tolerance, int line) { + for (int i = 0; i < count; i++) { + const float diff = fabsf(values[i] - known_good[i]); + const float scale = fabsf(known_good[i]) > 1.0f ? fabsf(known_good[i]) : 1.0f; + if (diff / scale > tolerance) { + printf("At TestCrossSIMD.cpp:%d: lane %d: %0.6f vs %0.6f\n", line, i, values[i], known_good[i]); + return false; + } + } + return true; +} + +static bool TestVec4S32() { + const int a_values[4] = { 3, -7, 0x4000, -1 }; + const int b_values[4] = { 5, 11, -2, 0x7FFF }; + + Vec4S32 a = Vec4S32::Load(a_values); + Vec4S32 b = Vec4S32::Load(b_values); + + int result[4]; + + (a + b).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] + b_values[i]); + } + + (a - b).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] - b_values[i]); + } + + a.Mul(b).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] * b_values[i]); + } + + // Mul16 only promises correct results when both sides fit in 16 bits. + const int small_a[4] = { 3, -7, 300, -1 }; + const int small_b[4] = { 5, 11, -2, 1000 }; + Vec4S32 sa = Vec4S32::Load(small_a); + Vec4S32 sb = Vec4S32::Load(small_b); + sa.Mul16(sb).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], small_a[i] * small_b[i]); + } + + Vec4S32::Splat(-5).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], -5); + } + + Vec4S32::Zero().Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], 0); + } + + a.Shl<2>().Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], a_values[i] << 2); + } + + // operator[] should agree with Store. + a.Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(a[i], result[i]); + } + + // Min16/Max16 likewise operate on 16-bit lanes. + sa.Min16(sb).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], small_a[i] < small_b[i] ? small_a[i] : small_b[i]); + } + sa.Max16(sb).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_INT(result[i], small_a[i] > small_b[i] ? small_a[i] : small_b[i]); + } + + const int sext_values[4] = { 0x0000FFFF, 0x00007FFF, 0x00008000, 0x00000001 }; + Vec4S32::Load(sext_values).SignExtend16().Store(result); + EXPECT_EQ_INT(result[0], -1); + EXPECT_EQ_INT(result[1], 32767); + EXPECT_EQ_INT(result[2], -32768); + EXPECT_EQ_INT(result[3], 1); + + return true; +} + +static bool TestVec4S32Compares() { + const int a_values[4] = { 1, 2, 3, 4 }; + const int b_values[4] = { 1, 5, 0, 4 }; + Vec4S32 a = Vec4S32::Load(a_values); + Vec4S32 b = Vec4S32::Load(b_values); + + int result[4]; + + // Compares produce all-ones or all-zeros per lane. + a.CompareEq(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + EXPECT_EQ_HEX((u32)result[2], 0u); + EXPECT_EQ_HEX((u32)result[3], 0xFFFFFFFFu); + + a.CompareLt(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0u); + EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[2], 0u); + EXPECT_EQ_HEX((u32)result[3], 0u); + + a.CompareGt(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0u); + EXPECT_EQ_HEX((u32)result[1], 0u); + EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[3], 0u); + + // AllCompareBitsSet / AnyCompareBitsSet over those masks. + EXPECT_FALSE(AllCompareBitsSet(a.CompareEq(b))); + EXPECT_TRUE(AnyCompareBitsSet(a.CompareEq(b))); + EXPECT_TRUE(AllCompareBitsSet(a.CompareEq(a))); + EXPECT_FALSE(AnyCompareBitsSet(a.CompareLt(a))); + + return true; +} + +static bool TestVec4F32Arith() { + const float a_values[4] = { 1.0f, -2.0f, 3.5f, 0.25f }; + const float b_values[4] = { 4.0f, 8.0f, -0.5f, 2.0f }; + + Vec4F32 a = Vec4F32::Load(a_values); + Vec4F32 b = Vec4F32::Load(b_values); + + float result[4]; + float expected[4]; + + (a + b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] + b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + (a - b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] - b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + (a * b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] * b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Mul(2.0f).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] * 2.0f; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Min(b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] < b_values[i] ? a_values[i] : b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Max(b).Store(result); + for (int i = 0; i < 4; i++) expected[i] = a_values[i] > b_values[i] ? a_values[i] : b_values[i]; + if (!CompareFloats(result, expected, 4, __LINE__)) return false; + + a.Clamp(0.0f, 3.0f).Store(result); + static const float known_clamp[4] = { 1.0f, 0.0f, 3.0f, 0.25f }; + if (!CompareFloats(result, known_clamp, 4, __LINE__)) return false; + + // Recip is allowed to be approximate on some architectures, RecipApprox definitely is. + b.Recip().Store(result); + for (int i = 0; i < 4; i++) expected[i] = 1.0f / b_values[i]; + if (!CompareFloatsApprox(result, expected, 4, 0.001f, __LINE__)) return false; + + b.RecipApprox().Store(result); + if (!CompareFloatsApprox(result, expected, 4, 0.01f, __LINE__)) return false; + + Vec4F32::Splat(1.5f).Store(result); + static const float known_splat[4] = { 1.5f, 1.5f, 1.5f, 1.5f }; + if (!CompareFloats(result, known_splat, 4, __LINE__)) return false; + + Vec4F32::Zero().Store(result); + static const float known_zero[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; + if (!CompareFloats(result, known_zero, 4, __LINE__)) return false; + + // Dot products. Dot3 ignores lane 3. + EXPECT_EQ_FLOAT(a.Dot3(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2]); + EXPECT_EQ_FLOAT(a.Dot4(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2] + a_values[3] * b_values[3]); + + // operator[] and GetLane<> should agree with Store. + a.Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(a[i], result[i]); + } + EXPECT_EQ_FLOAT(a.GetLane<0>(), a_values[0]); + EXPECT_EQ_FLOAT(a.GetLane<3>(), a_values[3]); + + return true; +} + +static bool TestVec4F32Compares() { + const float a_values[4] = { 1.0f, 2.0f, 3.0f, 4.0f }; + const float b_values[4] = { 1.0f, 5.0f, 0.0f, 4.0f }; + Vec4F32 a = Vec4F32::Load(a_values); + Vec4F32 b = Vec4F32::Load(b_values); + + int result[4]; + + a.CompareEq(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + + a.CompareLt(b).Store(result); + EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[2], 0u); + + a.CompareGt(b).Store(result); + EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + + a.CompareLe(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[2], 0u); + + a.CompareGe(b).Store(result); + EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu); + EXPECT_EQ_HEX((u32)result[1], 0u); + + return true; +} + +static bool TestVec4F32Lanes() { + const float values[4] = { 1.0f, 2.0f, 3.0f, 4.0f }; + const float other_values[4] = { 9.0f, 9.0f, 9.0f, 42.0f }; + Vec4F32 v = Vec4F32::Load(values); + Vec4F32 other = Vec4F32::Load(other_values); + + float result[4]; + + v.WithLane3Zero().Store(result); + static const float known_zero3[4] = { 1.0f, 2.0f, 3.0f, 0.0f }; + if (!CompareFloats(result, known_zero3, 4, __LINE__)) return false; + + v.WithLane3One().Store(result); + static const float known_one3[4] = { 1.0f, 2.0f, 3.0f, 1.0f }; + if (!CompareFloats(result, known_one3, 4, __LINE__)) return false; + + v.WithLane3From(other).Store(result); + static const float known_from3[4] = { 1.0f, 2.0f, 3.0f, 42.0f }; + if (!CompareFloats(result, known_from3, 4, __LINE__)) return false; + + v.ShuffleXXXX().Store(result); + static const float known_xxxx[4] = { 1.0f, 1.0f, 1.0f, 1.0f }; + if (!CompareFloats(result, known_xxxx, 4, __LINE__)) return false; + + v.ShuffleYYYY().Store(result); + static const float known_yyyy[4] = { 2.0f, 2.0f, 2.0f, 2.0f }; + if (!CompareFloats(result, known_yyyy, 4, __LINE__)) return false; + + v.ShuffleZZZZ().Store(result); + static const float known_zzzz[4] = { 3.0f, 3.0f, 3.0f, 3.0f }; + if (!CompareFloats(result, known_zzzz, 4, __LINE__)) return false; + + v.ShuffleWWWW().Store(result); + static const float known_wwww[4] = { 4.0f, 4.0f, 4.0f, 4.0f }; + if (!CompareFloats(result, known_wwww, 4, __LINE__)) return false; + + v.ShuffleXXYY().Store(result); + static const float known_xxyy[4] = { 1.0f, 1.0f, 2.0f, 2.0f }; + if (!CompareFloats(result, known_xxyy, 4, __LINE__)) return false; + + v.ShuffleZZWW().Store(result); + static const float known_zzww[4] = { 3.0f, 3.0f, 4.0f, 4.0f }; + if (!CompareFloats(result, known_zzww, 4, __LINE__)) return false; + + // Store2/Store3 must leave the rest of the destination alone. + float partial[4] = { -1.0f, -1.0f, -1.0f, -1.0f }; + v.Store2(partial); + static const float known_store2[4] = { 1.0f, 2.0f, -1.0f, -1.0f }; + if (!CompareFloats(partial, known_store2, 4, __LINE__)) return false; + + float partial3[4] = { -1.0f, -1.0f, -1.0f, -1.0f }; + v.Store3(partial3); + static const float known_store3[4] = { 1.0f, 2.0f, 3.0f, -1.0f }; + if (!CompareFloats(partial3, known_store3, 4, __LINE__)) return false; + + // Transpose of four columns. + static const float col_values[4][4] = { + { 0.0f, 1.0f, 2.0f, 3.0f }, + { 4.0f, 5.0f, 6.0f, 7.0f }, + { 8.0f, 9.0f, 10.0f, 11.0f }, + { 12.0f, 13.0f, 14.0f, 15.0f }, + }; + Vec4F32 c0 = Vec4F32::Load(col_values[0]); + Vec4F32 c1 = Vec4F32::Load(col_values[1]); + Vec4F32 c2 = Vec4F32::Load(col_values[2]); + Vec4F32 c3 = Vec4F32::Load(col_values[3]); + Vec4F32::Transpose(c0, c1, c2, c3); + float transposed[16]; + c0.Store(transposed); + c1.Store(transposed + 4); + c2.Store(transposed + 8); + c3.Store(transposed + 12); + for (int row = 0; row < 4; row++) { + for (int col = 0; col < 4; col++) { + EXPECT_EQ_FLOAT(transposed[row * 4 + col], col_values[col][row]); + } + } + + // LoadTranspose should match a plain Load of each row followed by Transpose. + static const float flat[16] = { + 0.0f, 1.0f, 2.0f, 3.0f, + 4.0f, 5.0f, 6.0f, 7.0f, + 8.0f, 9.0f, 10.0f, 11.0f, + 12.0f, 13.0f, 14.0f, 15.0f, + }; + Vec4F32 t0, t1, t2, t3; + Vec4F32::LoadTranspose(flat, t0, t1, t2, t3); + float loadTransposed[16]; + t0.Store(loadTransposed); + t1.Store(loadTransposed + 4); + t2.Store(loadTransposed + 8); + t3.Store(loadTransposed + 12); + if (!CompareFloats(loadTransposed, transposed, 16, __LINE__)) return false; + + return true; +} + +static bool TestVec4F32NaNInf() { + const float inf = INFINITY; + const float nan = NAN; + const float values[4] = { 1.0f, nan, inf, -inf }; + Vec4F32 v = Vec4F32::Load(values); + + float result[4]; + + v.ZeroNaNs().Store(result); + EXPECT_EQ_FLOAT(result[0], 1.0f); + EXPECT_EQ_FLOAT(result[1], 0.0f); + // ZeroNaNs leaves infinities alone. + EXPECT_TRUE(std::isinf(result[2])); + EXPECT_TRUE(std::isinf(result[3])); + + // The contract is just "a value that yields zero when multiplied by zero" - whatever is + // cheapest to get there. So the implementations legitimately differ on what a bad lane + // becomes: SSE2 clamps to +-FLT_MAX, NEON, LSX and the scalar fallback zero it. Check the + // property rather than the value, and that good lanes are left alone. + v.CleanNaNInfs().Store(result); + EXPECT_EQ_FLOAT(result[0], 1.0f); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i] * 0.0f, 0.0f); + } + + return true; +} + +static bool TestVec4F32Loads() { + float result[4]; + + // LoadF24x3_One: three 24-bit values shifted up into floats, lane 3 forced to 1.0f. + const uint32_t f24_values[4] = { 0x3F8000 >> 0, 0x400000, 0x3F0000, 0x123456 }; + Vec4F32::LoadF24x3_One(f24_values).Store(result); + for (int i = 0; i < 3; i++) { + float expected; + uint32_t bits = f24_values[i] << 8; + memcpy(&expected, &bits, 4); + EXPECT_EQ_FLOAT(result[i], expected); + } + EXPECT_EQ_FLOAT(result[3], 1.0f); + + // LoadF24x4: same, but all four lanes come from the source. + Vec4F32::LoadF24x4(f24_values).Store(result); + for (int i = 0; i < 4; i++) { + float expected; + uint32_t bits = f24_values[i] << 8; + memcpy(&expected, &bits, 4); + EXPECT_EQ_FLOAT(result[i], expected); + } + + // The normalizing loads. Some of these read 8 bytes, so give them room. + const int8_t s8_values[8] = { -1, -128, 127, 45, 0, 0, 0, 0 }; + Vec4F32::LoadS8Norm(s8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s8_values[i] / 128.0f); + } + + const int16_t s16_values[8] = { -1, -32768, 32767, 1234, 0, 0, 0, 0 }; + Vec4F32::LoadS16Norm(s16_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s16_values[i] / 32768.0f); + } + + const uint8_t u8_values[8] = { 0, 255, 128, 7, 0, 0, 0, 0 }; + Vec4F32::LoadU8Norm(u8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_APPROX_EQ_FLOAT(result[i], (float)u8_values[i] / 255.0f); + } + + // The converting (non-normalizing) loads. + Vec4F32::LoadConvertS16(s16_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s16_values[i]); + } + + Vec4F32::LoadConvertS8(s8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)s8_values[i]); + } + + Vec4F32::LoadConvertU8(u8_values).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)u8_values[i]); + } + + // StoreConvertToU8 truncates towards zero and saturates to 0-255. + const float to_u8_values[4] = { 0.0f, 255.0f, 12.75f, 200.0f }; + uint8_t u8_out[4]; + Vec4F32::Load(to_u8_values).StoreConvertToU8(u8_out); + EXPECT_EQ_INT(u8_out[0], 0); + EXPECT_EQ_INT(u8_out[1], 255); + EXPECT_EQ_INT(u8_out[2], 12); + EXPECT_EQ_INT(u8_out[3], 200); + + const float saturate_values[4] = { -1.0f, 300.0f, -0.5f, 1000.0f }; + Vec4F32::Load(saturate_values).StoreConvertToU8(u8_out); + EXPECT_EQ_INT(u8_out[0], 0); + EXPECT_EQ_INT(u8_out[1], 255); + EXPECT_EQ_INT(u8_out[2], 0); + EXPECT_EQ_INT(u8_out[3], 255); + + // Int to float. + const int int_values[4] = { 0, -7, 1000, -32768 }; + Vec4F32::FromVec4S32(Vec4S32::Load(int_values)).Store(result); + for (int i = 0; i < 4; i++) { + EXPECT_EQ_FLOAT(result[i], (float)int_values[i]); + } + + // Load2 only fills the first two lanes. + const float two_values[2] = { 3.0f, 4.0f }; + Vec4F32::Load2(two_values).Store(result); + EXPECT_EQ_FLOAT(result[0], 3.0f); + EXPECT_EQ_FLOAT(result[1], 4.0f); + + return true; +} + +static bool TestMatrices() { + static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f }; + static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f }; + static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, }; + float result[16]; + Mat4F32 a(a_values); + Mat4F32 b(b_values); + + Mul4x4By4x4(a, b).Store(result); + if (!CompareFloats(result, known_result, 16, __LINE__)) { + return false; + } + + Mat4x3F32 d = Mat4x3F32(b_values + 2); + Mul4x3By4x4(d, a).Store(result); + + static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, }; + if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) { + return false; + } + + static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f }; + Vec4F32 v = Vec4F32::Load(vec_values); + + v.AsVec3ByMatrix44(b).Store3(result); + + static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, }; + if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) { + return false; + } + Vec4F32 scale = Vec4F32::Load(a_values); + Vec4F32 translate = Vec4F32::Load(b_values); + + TranslateAndScaleInplace(a, scale, translate); + a.Store(result); + + static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,}; + if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) { + return false; + } + + return true; +} + +bool TestCrossSIMD() { + if (!TestVec4S32()) return false; + if (!TestVec4S32Compares()) return false; + if (!TestVec4F32Arith()) return false; + if (!TestVec4F32Compares()) return false; + if (!TestVec4F32Lanes()) return false; + if (!TestVec4F32NaNInf()) return false; + if (!TestVec4F32Loads()) return false; + if (!TestMatrices()) return false; + return true; +} diff --git a/unittest/UnitTest.cpp b/unittest/UnitTest.cpp index ebb9dc7cd7..383fe445b0 100644 --- a/unittest/UnitTest.cpp +++ b/unittest/UnitTest.cpp @@ -2764,88 +2764,6 @@ bool TestSIMD() { return true; } -static void PrintFloats(const float *f, int count) { - for (int i = 0; i < count; i++) { - printf("%.1ff, ", f[i]); - } - printf("\n"); -} - -static bool CompareFloats(const float *values, const float *known_good, int count, int line) { - int wrongCount = 0; - - for (int i = 0; i < count; i++) { - if (values[i] != known_good[i]) { - wrongCount++; - } - } - - if (wrongCount > 0) { - for (int i = 0; i < count; i++) { - bool wrong = values[i] != known_good[i]; - printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : ""); - } - printf("At UnitTest.cpp:%d: %d / %d were wrong\n", line, wrongCount, count); - return false; - } else { - return true; - } -} - -bool TestCrossSIMD() { - static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f }; - static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f }; - static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, }; - float result[16]; - Mat4F32 a(a_values); - Mat4F32 b(b_values); - - Mul4x4By4x4(a, b).Store(result); - if (!CompareFloats(result, known_result, 16, __LINE__)) { - return false; - } - - Mat4x3F32 d = Mat4x3F32(b_values + 2); - Mul4x3By4x4(d, a).Store(result); - - static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, }; - if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) { - return false; - } - - static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f }; - Vec4F32 v = Vec4F32::Load(vec_values); - - v.AsVec3ByMatrix44(b).Store3(result); - - static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, }; - if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) { - return false; - } - Vec4F32 scale = Vec4F32::Load(a_values); - Vec4F32 translate = Vec4F32::Load(b_values); - - TranslateAndScaleInplace(a, scale, translate); - a.Store(result); - - static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,}; - if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) { - return false; - } - - s8 values[4] = {-1, -128, 127, 45}; - float fvalues[4]; - Vec4F32::LoadS8Norm(values).Store(fvalues); - static const float known_s8norm_result[4] = {(float)values[0]/128.0f, (float)values[1]/128.0f, (float)values[2]/128.0f, (float)values[3]/128.0f,}; - if (!CompareFloats(fvalues, known_s8norm_result, ARRAY_SIZE(known_s8norm_result), __LINE__)) { - return false; - } - - // PrintFloats(result, 16); - - return true; -} - bool TestVolumeFunc() { for (int i = 0; i <= 20; i++) { float mul = Volume10ToMultiplier(i); @@ -3044,6 +2962,7 @@ struct TestItem { bool TestArmEmitter(); bool TestArm64Emitter(); +bool TestCrossSIMD(); bool TestX64Emitter(); bool TestRiscVEmitter(); bool TestLoongArch64Emitter(); diff --git a/unittest/UnitTests.vcxproj b/unittest/UnitTests.vcxproj index 5793f24e1e..e77fb4ba3f 100644 --- a/unittest/UnitTests.vcxproj +++ b/unittest/UnitTests.vcxproj @@ -287,6 +287,7 @@ + diff --git a/unittest/UnitTests.vcxproj.filters b/unittest/UnitTests.vcxproj.filters index 37af8831d9..9d249d43dd 100644 --- a/unittest/UnitTests.vcxproj.filters +++ b/unittest/UnitTests.vcxproj.filters @@ -14,6 +14,7 @@ +