mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #22314 from hrydgard/cross-build-riscv64
Loongarch64 and RiscV: Try to make these viable by adding cross builds
This commit is contained in:
15 files changed
+756
-151
No files matched your search
@@ -288,6 +288,13 @@ jobs:
|
||||
args: ./b.sh --loongarch64 PPSSPPHeadless
|
||||
id: loongarch64
|
||||
|
||||
- os: ubuntu-26.04
|
||||
extra: riscv64
|
||||
cc: gcc
|
||||
cxx: g++
|
||||
args: ./b.sh --riscv64 PPSSPPHeadless
|
||||
id: riscv64
|
||||
|
||||
- os: macos-latest
|
||||
extra: test
|
||||
cc: clang
|
||||
@@ -327,7 +334,7 @@ jobs:
|
||||
ndk-version: r29
|
||||
|
||||
- name: Install Linux dependencies
|
||||
if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64'
|
||||
if: runner.os == 'Linux' && matrix.extra != 'android' && matrix.extra != 'loongarch64' && matrix.extra != 'riscv64'
|
||||
run: |
|
||||
sudo apt-get update -y -qq
|
||||
sudo apt-get install libsdl3-dev libgl1-mesa-dev libglu1-mesa-dev libsdl3-ttf-dev libfontconfig1-dev libcurl4-openssl-dev
|
||||
@@ -336,12 +343,12 @@ jobs:
|
||||
if: runner.os == 'macOS' && matrix.id == 'macos'
|
||||
run: brew install sdl3 sdl3_ttf
|
||||
|
||||
- name: Install loongarch64 cross-compilation dependencies
|
||||
if: matrix.extra == 'loongarch64'
|
||||
- name: Install cross-compilation dependencies
|
||||
if: matrix.extra == 'loongarch64' || matrix.extra == 'riscv64'
|
||||
run: |
|
||||
sudo apt-get update -y -qq
|
||||
sudo apt-get install -y gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu libgl-dev libglu1-mesa-dev
|
||||
sudo cmake/scripts/setup-loongarch64-cross.sh
|
||||
sudo apt-get install -y gcc-14-${{ matrix.extra }}-linux-gnu g++-14-${{ matrix.extra }}-linux-gnu binutils-${{ matrix.extra }}-linux-gnu libgl-dev libglu1-mesa-dev
|
||||
sudo cmake/scripts/setup-cross.sh ${{ matrix.extra }}
|
||||
|
||||
- name: Install iOS dependencies
|
||||
if: matrix.id == 'ios'
|
||||
|
||||
+9
-4
@@ -184,6 +184,8 @@ option(USE_VULKAN_DISPLAY_KHR "Enable or disable full screen display of Vulkan"
|
||||
# :: Frontends
|
||||
option(MOBILE_DEVICE "Set to ON when targeting a mobile device" ${MOBILE_DEVICE})
|
||||
option(HEADLESS "Set to OFF to not generate the PPSSPPHeadless target" ${HEADLESS})
|
||||
# Set by b.sh for the loongarch64/riscv64 cross builds, whose sysroots have no SDL3.
|
||||
option(HEADLESS_CROSS "Set to ON for a headless-only cross build (no SDL frontend)" ${HEADLESS_CROSS})
|
||||
option(ATLAS_TOOL "Set to OFF to not generate the ATLAS_TOOL target" ${ATLAS_TOOL})
|
||||
option(UNITTEST "Set to ON to generate the unittest target" ${UNITTEST})
|
||||
option(SIMULATOR "Set to ON when targeting an x86 simulator of an ARM platform" ${SIMULATOR})
|
||||
@@ -331,7 +333,7 @@ set(SDL_LIB_TARGET "")
|
||||
set(SDL_TTF_LIB_TARGET "")
|
||||
set(FONTCONFIG_LIB_TARGET "")
|
||||
|
||||
if(NOT LIBRETRO AND NOT IOS AND NOT LOONGARCH64_DEVICE AND NOT ANDROID)
|
||||
if(NOT LIBRETRO AND NOT IOS AND NOT HEADLESS_CROSS AND NOT ANDROID)
|
||||
find_package(SDL3 QUIET)
|
||||
if(NOT SDL3_FOUND)
|
||||
if(CMAKE_SYSTEM_NAME STREQUAL "Linux" AND (EXISTS "/usr/bin/apt" OR EXISTS "/usr/bin/apt-get" OR EXISTS "/etc/debian_version"))
|
||||
@@ -1001,9 +1003,11 @@ elseif(WIN32)
|
||||
# Don't care about SDL.
|
||||
set(TargetBin PPSSPPWindows)
|
||||
elseif(LIBRETRO)
|
||||
elseif(LOONGARCH64_DEVICE)
|
||||
# Cross-compiling for loongarch64 currently only builds PPSSPPHeadless, and there's
|
||||
# no SDL3 available for that target, so skip the desktop SDL frontend entirely.
|
||||
elseif(HEADLESS_CROSS)
|
||||
# The loongarch64/riscv64 cross builds only build PPSSPPHeadless, and there's no
|
||||
# SDL3 available in those sysroots, so skip the desktop SDL frontend entirely.
|
||||
# Note this is deliberately not keyed off the target architecture: a native build
|
||||
# on such a machine can have SDL3 and should still get the normal frontend.
|
||||
else()
|
||||
if(GOLD)
|
||||
set(TargetBin PPSSPPGold)
|
||||
@@ -1362,6 +1366,7 @@ if(UNITTEST)
|
||||
unittest/TestShaderGenerators.cpp
|
||||
unittest/TestArmEmitter.cpp
|
||||
unittest/TestArm64Emitter.cpp
|
||||
unittest/TestCrossSIMD.cpp
|
||||
unittest/TestIRPassSimplify.cpp
|
||||
unittest/TestX64Emitter.cpp
|
||||
unittest/TestVertexJit.cpp
|
||||
|
||||
@@ -166,7 +166,8 @@ public:
|
||||
QuickCallFunction((const u8 *)func, scratchreg);
|
||||
}
|
||||
void QuickCallFunctionR(const u8 *func, LoongArch64Reg arg, LoongArch64Reg scratchreg = R_RA) {
|
||||
MOVE(LoongArch64Reg::X4, arg);
|
||||
// The first integer argument goes in a0, which is R4 - X4 is an LASX vector register.
|
||||
MOVE(LoongArch64Reg::R4, arg);
|
||||
QuickJump(scratchreg, R_RA, func);
|
||||
}
|
||||
template <typename T>
|
||||
|
||||
+54
-10
@@ -6,6 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include "Common/Math/SIMDHeaders.h"
|
||||
|
||||
@@ -1187,8 +1188,11 @@ struct Vec4F32 {
|
||||
}
|
||||
void StoreConvertToU8(uint8_t *dest) {
|
||||
__m128i ivalue32 = __lsx_vftintrz_w_s(v);
|
||||
__m128i ivalue16 = __lsx_vssrlrni_hu_w(ivalue32, ivalue32, 0);
|
||||
__m128i ivalue8 = __lsx_vssrlrni_bu_h(ivalue16, ivalue16, 0);
|
||||
// Narrow signed->signed first, then signed->unsigned, matching the packs/packus pair the
|
||||
// SSE version uses. The logical (unsigned) narrowing shifts turn a negative value into a
|
||||
// huge unsigned one, which saturates to 255 instead of clamping to 0.
|
||||
__m128i ivalue16 = __lsx_vssrani_h_w(ivalue32, ivalue32, 0);
|
||||
__m128i ivalue8 = __lsx_vssrani_bu_h(ivalue16, ivalue16, 0);
|
||||
uint32_t value = __lsx_vpickve2gr_wu(ivalue8, 0);
|
||||
memcpy(dest, &value, sizeof(uint32_t));
|
||||
}
|
||||
@@ -1198,10 +1202,10 @@ struct Vec4F32 {
|
||||
// index is a compile-time constant parameter to the intrinsic.
|
||||
int ival;
|
||||
switch (index) {
|
||||
case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0);
|
||||
case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1);
|
||||
case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2);
|
||||
default: ival = __lsx_vpickve2gr_w((__m128i)v, 3);
|
||||
case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); break;
|
||||
case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); break;
|
||||
case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); break;
|
||||
default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); break;
|
||||
}
|
||||
float fval;
|
||||
memcpy(&fval, &ival, sizeof(float));
|
||||
@@ -1671,15 +1675,38 @@ struct Vec4F32 {
|
||||
return temp;
|
||||
}
|
||||
|
||||
static Vec4F32 LoadConvertU8(const uint8_t *src) { // Note: will load 8 bytes, not 4. Only the first 4 bytes will be used.
|
||||
Vec4F32 temp;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
temp.v[i] = (float)src[i];
|
||||
}
|
||||
return temp;
|
||||
}
|
||||
|
||||
// Truncates towards zero and saturates to 0-255, like the packing instructions the other
|
||||
// implementations use.
|
||||
void StoreConvertToU8(uint8_t *dest) {
|
||||
for (int i = 0; i < 4; i++) {
|
||||
int value = (int)v[i];
|
||||
if (value < 0) value = 0;
|
||||
if (value > 255) value = 255;
|
||||
dest[i] = (uint8_t)value;
|
||||
}
|
||||
}
|
||||
|
||||
static Vec4F32 LoadF24x3_One(const uint32_t *src) {
|
||||
uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, 0 };
|
||||
constexpr uint32_t kOneF32Bits = 0x3F800000;
|
||||
uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, kOneF32Bits };
|
||||
Vec4F32 temp;
|
||||
memcpy(temp.v, shifted, sizeof(temp.v));
|
||||
return temp;
|
||||
}
|
||||
|
||||
static Vec4F32 LoadF24x4(const uint32_t *src) {
|
||||
return LoadR24x3_One(src);
|
||||
uint32_t shifted[4] = { src[0] << 8, src[1] << 8, src[2] << 8, src[3] << 8 };
|
||||
Vec4F32 temp;
|
||||
memcpy(temp.v, shifted, sizeof(temp.v));
|
||||
return temp;
|
||||
}
|
||||
|
||||
static Vec4F32 FromVec4S32(Vec4S32 src) {
|
||||
@@ -1695,7 +1722,7 @@ struct Vec4F32 {
|
||||
Vec4F32 ZeroNaNs() const {
|
||||
Vec4F32 temp;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
temp.v[i] = isnan(v[i]) ? 0.0f : v[i];
|
||||
temp.v[i] = std::isnan(v[i]) ? 0.0f : v[i];
|
||||
}
|
||||
return temp;
|
||||
}
|
||||
@@ -1703,7 +1730,7 @@ struct Vec4F32 {
|
||||
Vec4F32 CleanNaNInfs() {
|
||||
Vec4F32 temp;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
temp.v[i] = (isnan(v[i]) || isinf(v[i])) ? 0.0f : v[i];
|
||||
temp.v[i] = (std::isnan(v[i]) || std::isinf(v[i])) ? 0.0f : v[i];
|
||||
}
|
||||
return temp;
|
||||
}
|
||||
@@ -1798,6 +1825,10 @@ struct Vec4F32 {
|
||||
return Vec4F32{ { v[0], v[1], v[2], 1.0f } };
|
||||
}
|
||||
|
||||
Vec4F32 WithLane3From(Vec4F32 other) const {
|
||||
return Vec4F32{ { v[0], v[1], v[2], other.v[3] } };
|
||||
}
|
||||
|
||||
Vec4S32 CompareEq(Vec4F32 other) const {
|
||||
Vec4S32 temp;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
@@ -1857,6 +1888,15 @@ struct Vec4F32 {
|
||||
return Vec4F32{{ v[3], v[3], v[3], v[3] }};
|
||||
}
|
||||
|
||||
// Loads four rows of four floats and transposes them into columns.
|
||||
static void LoadTranspose(const float *src, Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) {
|
||||
col0 = Vec4F32::Load(src);
|
||||
col1 = Vec4F32::Load(src + 4);
|
||||
col2 = Vec4F32::Load(src + 8);
|
||||
col3 = Vec4F32::Load(src + 12);
|
||||
Transpose(col0, col1, col2, col3);
|
||||
}
|
||||
|
||||
// In-place transpose.
|
||||
static void Transpose(Vec4F32 &col0, Vec4F32 &col1, Vec4F32 &col2, Vec4F32 &col3) {
|
||||
float m[16];
|
||||
@@ -1919,6 +1959,10 @@ inline bool AllCompareBitsSet(Vec4S32 value) {
|
||||
return true;
|
||||
}
|
||||
|
||||
inline bool AnyCompareBitsSet(Vec4S32 value) {
|
||||
return value.v[0] != 0 || value.v[1] != 0 || value.v[2] != 0 || value.v[3] != 0;
|
||||
}
|
||||
|
||||
struct Vec4U16 {
|
||||
uint16_t v[4]; // 64 bits.
|
||||
|
||||
|
||||
@@ -842,7 +842,7 @@ static inline u32 EncodeFVF(RiscVReg vd, RiscVReg rs1, RiscVReg vs2, VUseMask vm
|
||||
|
||||
static inline u16 EncodeCR(Opcode16 op, RiscVReg rs2, RiscVReg rd, Funct4 funct4) {
|
||||
_assert_msg_(SupportsCompressed(), "Compressed instructions unsupported");
|
||||
return (u16)op | ((u16)rs2 << 2) | ((u16)rd << 7) | ((u16)funct4 << 12);
|
||||
return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)DecodeReg(rd) << 7) | ((u16)funct4 << 12);
|
||||
}
|
||||
|
||||
static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) {
|
||||
@@ -850,13 +850,13 @@ static inline u16 EncodeCI(Opcode16 op, u8 uimm6, RiscVReg rd, Funct3 funct3) {
|
||||
_assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6);
|
||||
u16 imm4_0 = ImmBits16(uimm6, 0, 5);
|
||||
u16 imm5 = ImmBit16(uimm6, 5);
|
||||
return (u16)op | (imm4_0 << 2) | ((u16)rd << 7) | (imm5 << 12) | ((u16)funct3 << 13);
|
||||
return (u16)op | (imm4_0 << 2) | ((u16)DecodeReg(rd) << 7) | (imm5 << 12) | ((u16)funct3 << 13);
|
||||
}
|
||||
|
||||
static inline u16 EncodeCSS(Opcode16 op, RiscVReg rs2, u8 uimm6, Funct3 funct3) {
|
||||
_assert_msg_(SupportsCompressed(), "Compressed instructions unsupported");
|
||||
_assert_msg_(uimm6 <= 0x3F, "CI immediate overflow: %04x", uimm6);
|
||||
return (u16)op | ((u16)rs2 << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13);
|
||||
return (u16)op | ((u16)DecodeReg(rs2) << 2) | ((u16)uimm6 << 7) | ((u16)funct3 << 13);
|
||||
}
|
||||
|
||||
static inline u16 EncodeCIW(Opcode16 op, RiscCReg rd, u8 uimm8, Funct3 funct3) {
|
||||
|
||||
@@ -1051,6 +1051,7 @@ ifeq ($(UNITTEST),1)
|
||||
$(SRC)/unittest/TestX64Emitter.cpp \
|
||||
$(SRC)/unittest/TestRiscVEmitter.cpp \
|
||||
$(SRC)/unittest/TestLoongArch64Emitter.cpp \
|
||||
$(SRC)/unittest/TestCrossSIMD.cpp \
|
||||
$(SRC)/unittest/TestIRPassSimplify.cpp \
|
||||
$(SRC)/unittest/TestShaderGenerators.cpp \
|
||||
$(SRC)/unittest/TestSoftwareGPUJit.cpp \
|
||||
|
||||
@@ -31,9 +31,14 @@ do
|
||||
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/raspberry.armv8.cmake ${CMAKE_ARGS}"
|
||||
;;
|
||||
--loongarch64)
|
||||
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}"
|
||||
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/loongarch64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}"
|
||||
TARGET_OS=loongarch64
|
||||
LOONGARCH64_BUILD=1
|
||||
CROSS_STUB_CC=loongarch64-linux-gnu-gcc-14
|
||||
;;
|
||||
--riscv64)
|
||||
CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=cmake/Toolchains/riscv64-linux-gnu.cmake -DHEADLESS=ON -DHEADLESS_CROSS=ON -DUSE_SYSTEM_LIBPNG=OFF -DUSE_SYSTEM_LIBSDL2=OFF ${CMAKE_ARGS}"
|
||||
TARGET_OS=riscv64
|
||||
CROSS_STUB_CC=riscv64-linux-gnu-gcc-14
|
||||
;;
|
||||
--android) CMAKE_ARGS="-DCMAKE_TOOLCHAIN_FILE=android/android.toolchain.cmake ${CMAKE_ARGS}"
|
||||
TARGET_OS=Android
|
||||
@@ -122,28 +127,29 @@ echo Building with $CORES_COUNT threads
|
||||
|
||||
mkdir -p ${BUILD_DIR}
|
||||
|
||||
# For loongarch64 cross-compilation, build a comprehensive GL/GLX stub into
|
||||
# For the headless cross targets, build a comprehensive GL/GLX stub into
|
||||
# <build>/stublibs/libGL.so so GLEW's static archive can resolve its symbols
|
||||
# via PLT entries (a direct B26 branch to address 0 overflows on LoongArch).
|
||||
# via PLT entries (a direct branch to address 0 overflows the relocation).
|
||||
# This stub is always (re)generated to pick up any new needed symbols.
|
||||
if [ ! -z "$LOONGARCH64_BUILD" ]; then
|
||||
if [ ! -z "$CROSS_STUB_CC" ]; then
|
||||
STUB_DIR=${BUILD_DIR}/stublibs
|
||||
STUB_GL=${STUB_DIR}/libGL.so
|
||||
mkdir -p "${STUB_DIR}"
|
||||
STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c)
|
||||
echo "/* Loongarch64 GL/GLX stub - cross-compilation only */" > "$STUB_C"
|
||||
echo "/* ${TARGET_OS} GL/GLX stub - cross-compilation only */" > "$STUB_C"
|
||||
# Collect all T (exported) symbols from GL/GLX libs, deduplicate, emit stubs
|
||||
HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null)
|
||||
{
|
||||
for lib in /usr/lib/x86_64-linux-gnu/libGL.so.1 \
|
||||
/usr/lib/x86_64-linux-gnu/libGLX.so.0 \
|
||||
/usr/lib/x86_64-linux-gnu/libGLdispatch.so.0; do
|
||||
for lib in /usr/lib/$HOST_MULTIARCH/libGL.so.1 \
|
||||
/usr/lib/$HOST_MULTIARCH/libGLX.so.0 \
|
||||
/usr/lib/$HOST_MULTIARCH/libGLdispatch.so.0; do
|
||||
[ -f "$lib" ] && nm -D "$lib" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print $3 }'
|
||||
done
|
||||
# Always include the minimal GLX symbols GLEW directly references
|
||||
printf '%s\n' glXGetProcAddressARB glXGetClientString glXQueryVersion \
|
||||
glBindTexture glGetString glGetIntegerv
|
||||
} | sort -u | awk '{ print "void "$1"(void){}" }' >> "$STUB_C"
|
||||
loongarch64-linux-gnu-gcc-14 -shared -fPIC -Wno-implicit-function-declaration \
|
||||
$CROSS_STUB_CC -shared -fPIC -Wno-implicit-function-declaration \
|
||||
-o "${STUB_GL}" "$STUB_C"
|
||||
rm "$STUB_C"
|
||||
fi
|
||||
|
||||
@@ -17,8 +17,8 @@ set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE)
|
||||
# Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so
|
||||
# rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have.
|
||||
# A stub libGL.so + GL headers must be present in the sysroot; they are placed
|
||||
# there by cmake/scripts/setup-loongarch64-cross.sh (run once with sudo, or
|
||||
# automatically by the CI install step).
|
||||
# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically
|
||||
# by the CI install step).
|
||||
set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE)
|
||||
|
||||
# b.sh also creates a stub in <build-dir>/stublibs/ as a local fallback so the
|
||||
@@ -27,14 +27,18 @@ set(CMAKE_EXE_LINKER_FLAGS
|
||||
"${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs"
|
||||
CACHE STRING "" FORCE)
|
||||
|
||||
# Allow CMake's compile-check programs to run via QEMU.
|
||||
# The /lib64 symlink is created by the setup script; without it, pass -L
|
||||
# explicitly so QEMU can find the loongarch64 dynamic linker.
|
||||
if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1")
|
||||
set(CMAKE_CROSSCOMPILING_EMULATOR "qemu-loongarch64-static"
|
||||
CACHE STRING "" FORCE)
|
||||
else()
|
||||
set(CMAKE_CROSSCOMPILING_EMULATOR
|
||||
"qemu-loongarch64-static;-L;/usr/loongarch64-linux-gnu"
|
||||
CACHE STRING "" FORCE)
|
||||
# Allow CMake's compile-check programs to run via QEMU, if it's installed.
|
||||
# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic
|
||||
# ones in qemu-user; either will do, so take whichever is present.
|
||||
find_program(QEMU_LOONGARCH64 NAMES qemu-loongarch64-static qemu-loongarch64)
|
||||
if(QEMU_LOONGARCH64)
|
||||
# The /lib64 symlink is created by the setup script; without it, pass -L
|
||||
# explicitly so QEMU can find the loongarch64 dynamic linker.
|
||||
if(EXISTS "/lib64/ld-linux-loongarch-lp64d.so.1")
|
||||
set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_LOONGARCH64}" CACHE STRING "" FORCE)
|
||||
else()
|
||||
set(CMAKE_CROSSCOMPILING_EMULATOR
|
||||
"${QEMU_LOONGARCH64};-L;/usr/loongarch64-linux-gnu"
|
||||
CACHE STRING "" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
@@ -0,0 +1,44 @@
|
||||
set(CMAKE_SYSTEM_NAME Linux)
|
||||
set(CMAKE_SYSTEM_PROCESSOR riscv64)
|
||||
|
||||
set(CMAKE_C_COMPILER riscv64-linux-gnu-gcc-14)
|
||||
set(CMAKE_CXX_COMPILER riscv64-linux-gnu-g++-14)
|
||||
set(CMAKE_ASM_COMPILER riscv64-linux-gnu-gcc-14)
|
||||
|
||||
set(CMAKE_FIND_ROOT_PATH /usr/riscv64-linux-gnu)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PROGRAM NEVER)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_LIBRARY ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_INCLUDE ONLY)
|
||||
set(CMAKE_FIND_ROOT_PATH_MODE_PACKAGE ONLY)
|
||||
|
||||
# No X11 in the sysroot; disable the X11 Vulkan path to avoid include-dir errors.
|
||||
set(USING_X11_VULKAN OFF CACHE BOOL "" FORCE)
|
||||
|
||||
# Use legacy GL preference so find_package(OpenGL) looks for a single libGL.so
|
||||
# rather than the GLVND split (libOpenGL.so + libGLX.so), which we don't have.
|
||||
# A stub libGL.so + GL headers must be present in the sysroot; they are placed
|
||||
# there by cmake/scripts/setup-cross.sh (run once with sudo, or automatically
|
||||
# by the CI install step).
|
||||
set(OpenGL_GL_PREFERENCE LEGACY CACHE STRING "" FORCE)
|
||||
|
||||
# b.sh also creates a stub in <build-dir>/stublibs/ as a local fallback so the
|
||||
# linker can resolve GL/GLX calls from GLEW's static archive via PLT entries.
|
||||
set(CMAKE_EXE_LINKER_FLAGS
|
||||
"${CMAKE_EXE_LINKER_FLAGS} -L${CMAKE_BINARY_DIR}/stublibs"
|
||||
CACHE STRING "" FORCE)
|
||||
|
||||
# Allow CMake's compile-check programs to run via QEMU, if it's installed.
|
||||
# Debian/Ubuntu ship the static binaries in qemu-user-static and the dynamic
|
||||
# ones in qemu-user; either will do, so take whichever is present.
|
||||
find_program(QEMU_RISCV64 NAMES qemu-riscv64-static qemu-riscv64)
|
||||
if(QEMU_RISCV64)
|
||||
# The /lib64 symlink is created by the setup script; without it, pass -L
|
||||
# explicitly so QEMU can find the riscv64 dynamic linker.
|
||||
if(EXISTS "/lib64/ld-linux-riscv64-lp64d.so.1")
|
||||
set(CMAKE_CROSSCOMPILING_EMULATOR "${QEMU_RISCV64}" CACHE STRING "" FORCE)
|
||||
else()
|
||||
set(CMAKE_CROSSCOMPILING_EMULATOR
|
||||
"${QEMU_RISCV64};-L;/usr/riscv64-linux-gnu"
|
||||
CACHE STRING "" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
@@ -1,15 +1,34 @@
|
||||
#!/bin/bash
|
||||
# Sets up the loongarch64 cross-compilation environment.
|
||||
# Run once with sudo. After this, ./b.sh --loongarch64 works on a fresh checkout.
|
||||
# Sets up a cross-compilation environment for one of the headless-only targets.
|
||||
# Run once with sudo. After this, ./b.sh --<target> works on a fresh checkout.
|
||||
#
|
||||
# Usage: sudo cmake/scripts/setup-cross.sh <loongarch64|riscv64>
|
||||
|
||||
set -e
|
||||
|
||||
SYSROOT=/usr/loongarch64-linux-gnu
|
||||
CROSS_GCC=loongarch64-linux-gnu-gcc-14
|
||||
TARGET="$1"
|
||||
|
||||
case "$TARGET" in
|
||||
loongarch64)
|
||||
TRIPLE=loongarch64-linux-gnu
|
||||
LD_SO=ld-linux-loongarch-lp64d.so.1
|
||||
;;
|
||||
riscv64)
|
||||
TRIPLE=riscv64-linux-gnu
|
||||
LD_SO=ld-linux-riscv64-lp64d.so.1
|
||||
;;
|
||||
*)
|
||||
echo "Usage: sudo $0 <loongarch64|riscv64>"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
SYSROOT=/usr/$TRIPLE
|
||||
CROSS_GCC=$TRIPLE-gcc-14
|
||||
|
||||
if ! command -v $CROSS_GCC &>/dev/null; then
|
||||
echo "Error: $CROSS_GCC not found. Install it first:"
|
||||
echo " sudo apt install gcc-14-loongarch64-linux-gnu g++-14-loongarch64-linux-gnu binutils-loongarch64-linux-gnu"
|
||||
echo " sudo apt install gcc-14-$TRIPLE g++-14-$TRIPLE binutils-$TRIPLE"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
@@ -18,6 +37,8 @@ if [ "$(id -u)" -ne 0 ]; then
|
||||
exit 1
|
||||
fi
|
||||
|
||||
HOST_MULTIARCH=$(gcc -print-multiarch 2>/dev/null || dpkg-architecture -qDEB_HOST_MULTIARCH 2>/dev/null)
|
||||
|
||||
# ── OpenGL headers ───────────────────────────────────────────────────────────
|
||||
# GL headers are platform-independent C headers; copy from the host.
|
||||
echo "Installing GL headers into sysroot..."
|
||||
@@ -27,22 +48,22 @@ for h in gl.h glext.h glcorearb.h glu.h; do
|
||||
done
|
||||
|
||||
# ── GL stub library ──────────────────────────────────────────────────────────
|
||||
# Generate a loongarch64 libGL.so that exports all symbols from the host
|
||||
# Generate a libGL.so for the target that exports all symbols from the host
|
||||
# libGL so that GLEW's static archive can be linked via PLT entries (a direct
|
||||
# B26 branch to address 0 would overflow on LoongArch).
|
||||
# branch to address 0 would overflow the relocation on these targets).
|
||||
echo "Generating libGL.so stub..."
|
||||
|
||||
HOST_GL=""
|
||||
for candidate in \
|
||||
/usr/lib/x86_64-linux-gnu/libGL.so.1 \
|
||||
/usr/lib/x86_64-linux-gnu/libGL.so \
|
||||
/usr/lib/$HOST_MULTIARCH/libGL.so.1 \
|
||||
/usr/lib/$HOST_MULTIARCH/libGL.so \
|
||||
/usr/lib/libGL.so.1 ; do
|
||||
[ -f "$candidate" ] && HOST_GL="$candidate" && break
|
||||
done
|
||||
|
||||
STUB_C=$(mktemp /tmp/gl_stub_XXXXXX.c)
|
||||
echo "/* Loongarch64 GL/GLX stub - for cross-compilation only */" > "$STUB_C"
|
||||
echo "void __loongarch64_gl_placeholder(void) {}" >> "$STUB_C"
|
||||
echo "/* $TARGET GL/GLX stub - for cross-compilation only */" > "$STUB_C"
|
||||
echo "void __cross_gl_placeholder(void) {}" >> "$STUB_C"
|
||||
|
||||
if [ -n "$HOST_GL" ]; then
|
||||
nm -D "$HOST_GL" 2>/dev/null | awk '/^[0-9a-f]+ T /{ print "void "$3"(void){}" }' >> "$STUB_C"
|
||||
@@ -60,16 +81,16 @@ rm "$STUB_C"
|
||||
echo " Installed to $SYSROOT/lib/libGL.so"
|
||||
|
||||
# ── Dynamic linker symlink ───────────────────────────────────────────────────
|
||||
# Without this, binfmt_misc can't find the loongarch64 dynamic linker.
|
||||
LD_LINUX=$SYSROOT/lib/ld-linux-loongarch-lp64d.so.1
|
||||
# Without this, binfmt_misc can't find the target's dynamic linker.
|
||||
LD_LINUX=$SYSROOT/lib/$LD_SO
|
||||
if [ -f "$LD_LINUX" ]; then
|
||||
mkdir -p /lib64
|
||||
ln -sf "$LD_LINUX" /lib64/ld-linux-loongarch-lp64d.so.1
|
||||
echo "Created /lib64/ld-linux-loongarch-lp64d.so.1 -> $LD_LINUX"
|
||||
ln -sf "$LD_LINUX" /lib64/$LD_SO
|
||||
echo "Created /lib64/$LD_SO -> $LD_LINUX"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Setup complete."
|
||||
echo "Build: ./b.sh --loongarch64"
|
||||
echo "Run: qemu-loongarch64-static -L $SYSROOT <binary>"
|
||||
echo " (after setup the /lib64 symlink lets you run loongarch64 binaries directly)"
|
||||
echo "Build: ./b.sh --$TARGET"
|
||||
echo "Run: qemu-$TARGET -L $SYSROOT <binary>"
|
||||
echo " (after setup the /lib64 symlink lets you run $TARGET binaries directly)"
|
||||
@@ -289,10 +289,10 @@ static GraphicsContext *CreateGraphicsContext(GPUCore gpuCore, std::string **dev
|
||||
default:
|
||||
return nullptr;
|
||||
}
|
||||
#elif PPSSPP_ARCH(LOONGARCH64)
|
||||
// The loongarch64 cross-compilation toolchain has no SDL3 packages available (see the
|
||||
// LOONGARCH64_DEVICE branch in CMakeLists.txt), so this build is compile-tested only and
|
||||
// never actually needs to create a graphics context at runtime.
|
||||
#elif PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64)
|
||||
// The loongarch64 and riscv64 cross-compilation sysroots have no SDL3 (see the HEADLESS_CROSS
|
||||
// branch in CMakeLists.txt). These builds still run fine under qemu with --graphics=software,
|
||||
// which needs no graphics context.
|
||||
*deviceSetting = nullptr;
|
||||
return nullptr;
|
||||
#elif PPSSPP_PLATFORM(ANDROID)
|
||||
@@ -904,7 +904,7 @@ int main(int argc, const char* argv[]) {
|
||||
// We don't bother with a window.
|
||||
graphicsContext = new NullGraphicsContext();
|
||||
} else {
|
||||
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64)
|
||||
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64)
|
||||
fprintf(stderr, "Headless graphics context creation is not supported on this platform.\n");
|
||||
return 1;
|
||||
#else
|
||||
@@ -1134,7 +1134,7 @@ int main(int argc, const char* argv[]) {
|
||||
ShutdownWebServer();
|
||||
}
|
||||
|
||||
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64)
|
||||
#if PPSSPP_PLATFORM(ANDROID) || PPSSPP_ARCH(LOONGARCH64) || PPSSPP_ARCH(RISCV64)
|
||||
// ... see above
|
||||
#else
|
||||
if (window) {
|
||||
|
||||
@@ -0,0 +1,551 @@
|
||||
// Copyright (c) 2012- PPSSPP Project.
|
||||
|
||||
// This program is free software: you can redistribute it and/or modify
|
||||
// it under the terms of the GNU General Public License as published by
|
||||
// the Free Software Foundation, version 2.0 or later versions.
|
||||
|
||||
// This program is distributed in the hope that it will be useful,
|
||||
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
// GNU General Public License 2.0 for more details.
|
||||
|
||||
// A copy of the GPL 2.0 should have been included with the program.
|
||||
// If not, see http://www.gnu.org/licenses/
|
||||
|
||||
// Official git repository and contact information can be found at
|
||||
// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/.
|
||||
|
||||
// Tests for Common/Math/CrossSIMD.h.
|
||||
//
|
||||
// CrossSIMD has four independent implementations - SSE2, NEON, LoongArch LSX and a plain scalar
|
||||
// fallback - and only one of them is compiled on any given machine. So the point of these tests is
|
||||
// to check whatever got compiled against results worked out by hand, rather than against each other;
|
||||
// that way every architecture we build for verifies its own copy. Running the unit tests under
|
||||
// qemu covers the ones we have no hardware for.
|
||||
//
|
||||
// Only functions that all four implementations provide are tested here, otherwise this wouldn't
|
||||
// build everywhere.
|
||||
|
||||
#include "ppsspp_config.h"
|
||||
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
|
||||
#include "Common/Common.h"
|
||||
#include "Common/CommonTypes.h"
|
||||
#include "Common/Math/CrossSIMD.h"
|
||||
#include "unittest/UnitTest.h"
|
||||
|
||||
static bool CompareFloats(const float *values, const float *known_good, int count, int line) {
|
||||
int wrongCount = 0;
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (values[i] != known_good[i]) {
|
||||
wrongCount++;
|
||||
}
|
||||
}
|
||||
if (wrongCount > 0) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
bool wrong = values[i] != known_good[i];
|
||||
printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : "");
|
||||
}
|
||||
printf("At TestCrossSIMD.cpp:%d: %d / %d were wrong\n", line, wrongCount, count);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// For the approximate operations (reciprocals), which are allowed to differ between architectures.
|
||||
static bool CompareFloatsApprox(const float *values, const float *known_good, int count, float tolerance, int line) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
const float diff = fabsf(values[i] - known_good[i]);
|
||||
const float scale = fabsf(known_good[i]) > 1.0f ? fabsf(known_good[i]) : 1.0f;
|
||||
if (diff / scale > tolerance) {
|
||||
printf("At TestCrossSIMD.cpp:%d: lane %d: %0.6f vs %0.6f\n", line, i, values[i], known_good[i]);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4S32() {
|
||||
const int a_values[4] = { 3, -7, 0x4000, -1 };
|
||||
const int b_values[4] = { 5, 11, -2, 0x7FFF };
|
||||
|
||||
Vec4S32 a = Vec4S32::Load(a_values);
|
||||
Vec4S32 b = Vec4S32::Load(b_values);
|
||||
|
||||
int result[4];
|
||||
|
||||
(a + b).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], a_values[i] + b_values[i]);
|
||||
}
|
||||
|
||||
(a - b).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], a_values[i] - b_values[i]);
|
||||
}
|
||||
|
||||
a.Mul(b).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], a_values[i] * b_values[i]);
|
||||
}
|
||||
|
||||
// Mul16 only promises correct results when both sides fit in 16 bits.
|
||||
const int small_a[4] = { 3, -7, 300, -1 };
|
||||
const int small_b[4] = { 5, 11, -2, 1000 };
|
||||
Vec4S32 sa = Vec4S32::Load(small_a);
|
||||
Vec4S32 sb = Vec4S32::Load(small_b);
|
||||
sa.Mul16(sb).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], small_a[i] * small_b[i]);
|
||||
}
|
||||
|
||||
Vec4S32::Splat(-5).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], -5);
|
||||
}
|
||||
|
||||
Vec4S32::Zero().Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], 0);
|
||||
}
|
||||
|
||||
a.Shl<2>().Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], a_values[i] << 2);
|
||||
}
|
||||
|
||||
// operator[] should agree with Store.
|
||||
a.Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(a[i], result[i]);
|
||||
}
|
||||
|
||||
// Min16/Max16 likewise operate on 16-bit lanes.
|
||||
sa.Min16(sb).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], small_a[i] < small_b[i] ? small_a[i] : small_b[i]);
|
||||
}
|
||||
sa.Max16(sb).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_INT(result[i], small_a[i] > small_b[i] ? small_a[i] : small_b[i]);
|
||||
}
|
||||
|
||||
const int sext_values[4] = { 0x0000FFFF, 0x00007FFF, 0x00008000, 0x00000001 };
|
||||
Vec4S32::Load(sext_values).SignExtend16().Store(result);
|
||||
EXPECT_EQ_INT(result[0], -1);
|
||||
EXPECT_EQ_INT(result[1], 32767);
|
||||
EXPECT_EQ_INT(result[2], -32768);
|
||||
EXPECT_EQ_INT(result[3], 1);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4S32Compares() {
|
||||
const int a_values[4] = { 1, 2, 3, 4 };
|
||||
const int b_values[4] = { 1, 5, 0, 4 };
|
||||
Vec4S32 a = Vec4S32::Load(a_values);
|
||||
Vec4S32 b = Vec4S32::Load(b_values);
|
||||
|
||||
int result[4];
|
||||
|
||||
// Compares produce all-ones or all-zeros per lane.
|
||||
a.CompareEq(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0u);
|
||||
EXPECT_EQ_HEX((u32)result[2], 0u);
|
||||
EXPECT_EQ_HEX((u32)result[3], 0xFFFFFFFFu);
|
||||
|
||||
a.CompareLt(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[0], 0u);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[2], 0u);
|
||||
EXPECT_EQ_HEX((u32)result[3], 0u);
|
||||
|
||||
a.CompareGt(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[0], 0u);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0u);
|
||||
EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[3], 0u);
|
||||
|
||||
// AllCompareBitsSet / AnyCompareBitsSet over those masks.
|
||||
EXPECT_FALSE(AllCompareBitsSet(a.CompareEq(b)));
|
||||
EXPECT_TRUE(AnyCompareBitsSet(a.CompareEq(b)));
|
||||
EXPECT_TRUE(AllCompareBitsSet(a.CompareEq(a)));
|
||||
EXPECT_FALSE(AnyCompareBitsSet(a.CompareLt(a)));
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4F32Arith() {
|
||||
const float a_values[4] = { 1.0f, -2.0f, 3.5f, 0.25f };
|
||||
const float b_values[4] = { 4.0f, 8.0f, -0.5f, 2.0f };
|
||||
|
||||
Vec4F32 a = Vec4F32::Load(a_values);
|
||||
Vec4F32 b = Vec4F32::Load(b_values);
|
||||
|
||||
float result[4];
|
||||
float expected[4];
|
||||
|
||||
(a + b).Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = a_values[i] + b_values[i];
|
||||
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
|
||||
|
||||
(a - b).Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = a_values[i] - b_values[i];
|
||||
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
|
||||
|
||||
(a * b).Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = a_values[i] * b_values[i];
|
||||
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
|
||||
|
||||
a.Mul(2.0f).Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = a_values[i] * 2.0f;
|
||||
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
|
||||
|
||||
a.Min(b).Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = a_values[i] < b_values[i] ? a_values[i] : b_values[i];
|
||||
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
|
||||
|
||||
a.Max(b).Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = a_values[i] > b_values[i] ? a_values[i] : b_values[i];
|
||||
if (!CompareFloats(result, expected, 4, __LINE__)) return false;
|
||||
|
||||
a.Clamp(0.0f, 3.0f).Store(result);
|
||||
static const float known_clamp[4] = { 1.0f, 0.0f, 3.0f, 0.25f };
|
||||
if (!CompareFloats(result, known_clamp, 4, __LINE__)) return false;
|
||||
|
||||
// Recip is allowed to be approximate on some architectures, RecipApprox definitely is.
|
||||
b.Recip().Store(result);
|
||||
for (int i = 0; i < 4; i++) expected[i] = 1.0f / b_values[i];
|
||||
if (!CompareFloatsApprox(result, expected, 4, 0.001f, __LINE__)) return false;
|
||||
|
||||
b.RecipApprox().Store(result);
|
||||
if (!CompareFloatsApprox(result, expected, 4, 0.01f, __LINE__)) return false;
|
||||
|
||||
Vec4F32::Splat(1.5f).Store(result);
|
||||
static const float known_splat[4] = { 1.5f, 1.5f, 1.5f, 1.5f };
|
||||
if (!CompareFloats(result, known_splat, 4, __LINE__)) return false;
|
||||
|
||||
Vec4F32::Zero().Store(result);
|
||||
static const float known_zero[4] = { 0.0f, 0.0f, 0.0f, 0.0f };
|
||||
if (!CompareFloats(result, known_zero, 4, __LINE__)) return false;
|
||||
|
||||
// Dot products. Dot3 ignores lane 3.
|
||||
EXPECT_EQ_FLOAT(a.Dot3(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2]);
|
||||
EXPECT_EQ_FLOAT(a.Dot4(b), a_values[0] * b_values[0] + a_values[1] * b_values[1] + a_values[2] * b_values[2] + a_values[3] * b_values[3]);
|
||||
|
||||
// operator[] and GetLane<> should agree with Store.
|
||||
a.Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(a[i], result[i]);
|
||||
}
|
||||
EXPECT_EQ_FLOAT(a.GetLane<0>(), a_values[0]);
|
||||
EXPECT_EQ_FLOAT(a.GetLane<3>(), a_values[3]);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4F32Compares() {
|
||||
const float a_values[4] = { 1.0f, 2.0f, 3.0f, 4.0f };
|
||||
const float b_values[4] = { 1.0f, 5.0f, 0.0f, 4.0f };
|
||||
Vec4F32 a = Vec4F32::Load(a_values);
|
||||
Vec4F32 b = Vec4F32::Load(b_values);
|
||||
|
||||
int result[4];
|
||||
|
||||
a.CompareEq(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0u);
|
||||
|
||||
a.CompareLt(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[2], 0u);
|
||||
|
||||
a.CompareGt(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[2], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0u);
|
||||
|
||||
a.CompareLe(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[2], 0u);
|
||||
|
||||
a.CompareGe(b).Store(result);
|
||||
EXPECT_EQ_HEX((u32)result[0], 0xFFFFFFFFu);
|
||||
EXPECT_EQ_HEX((u32)result[1], 0u);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4F32Lanes() {
|
||||
const float values[4] = { 1.0f, 2.0f, 3.0f, 4.0f };
|
||||
const float other_values[4] = { 9.0f, 9.0f, 9.0f, 42.0f };
|
||||
Vec4F32 v = Vec4F32::Load(values);
|
||||
Vec4F32 other = Vec4F32::Load(other_values);
|
||||
|
||||
float result[4];
|
||||
|
||||
v.WithLane3Zero().Store(result);
|
||||
static const float known_zero3[4] = { 1.0f, 2.0f, 3.0f, 0.0f };
|
||||
if (!CompareFloats(result, known_zero3, 4, __LINE__)) return false;
|
||||
|
||||
v.WithLane3One().Store(result);
|
||||
static const float known_one3[4] = { 1.0f, 2.0f, 3.0f, 1.0f };
|
||||
if (!CompareFloats(result, known_one3, 4, __LINE__)) return false;
|
||||
|
||||
v.WithLane3From(other).Store(result);
|
||||
static const float known_from3[4] = { 1.0f, 2.0f, 3.0f, 42.0f };
|
||||
if (!CompareFloats(result, known_from3, 4, __LINE__)) return false;
|
||||
|
||||
v.ShuffleXXXX().Store(result);
|
||||
static const float known_xxxx[4] = { 1.0f, 1.0f, 1.0f, 1.0f };
|
||||
if (!CompareFloats(result, known_xxxx, 4, __LINE__)) return false;
|
||||
|
||||
v.ShuffleYYYY().Store(result);
|
||||
static const float known_yyyy[4] = { 2.0f, 2.0f, 2.0f, 2.0f };
|
||||
if (!CompareFloats(result, known_yyyy, 4, __LINE__)) return false;
|
||||
|
||||
v.ShuffleZZZZ().Store(result);
|
||||
static const float known_zzzz[4] = { 3.0f, 3.0f, 3.0f, 3.0f };
|
||||
if (!CompareFloats(result, known_zzzz, 4, __LINE__)) return false;
|
||||
|
||||
v.ShuffleWWWW().Store(result);
|
||||
static const float known_wwww[4] = { 4.0f, 4.0f, 4.0f, 4.0f };
|
||||
if (!CompareFloats(result, known_wwww, 4, __LINE__)) return false;
|
||||
|
||||
v.ShuffleXXYY().Store(result);
|
||||
static const float known_xxyy[4] = { 1.0f, 1.0f, 2.0f, 2.0f };
|
||||
if (!CompareFloats(result, known_xxyy, 4, __LINE__)) return false;
|
||||
|
||||
v.ShuffleZZWW().Store(result);
|
||||
static const float known_zzww[4] = { 3.0f, 3.0f, 4.0f, 4.0f };
|
||||
if (!CompareFloats(result, known_zzww, 4, __LINE__)) return false;
|
||||
|
||||
// Store2/Store3 must leave the rest of the destination alone.
|
||||
float partial[4] = { -1.0f, -1.0f, -1.0f, -1.0f };
|
||||
v.Store2(partial);
|
||||
static const float known_store2[4] = { 1.0f, 2.0f, -1.0f, -1.0f };
|
||||
if (!CompareFloats(partial, known_store2, 4, __LINE__)) return false;
|
||||
|
||||
float partial3[4] = { -1.0f, -1.0f, -1.0f, -1.0f };
|
||||
v.Store3(partial3);
|
||||
static const float known_store3[4] = { 1.0f, 2.0f, 3.0f, -1.0f };
|
||||
if (!CompareFloats(partial3, known_store3, 4, __LINE__)) return false;
|
||||
|
||||
// Transpose of four columns.
|
||||
static const float col_values[4][4] = {
|
||||
{ 0.0f, 1.0f, 2.0f, 3.0f },
|
||||
{ 4.0f, 5.0f, 6.0f, 7.0f },
|
||||
{ 8.0f, 9.0f, 10.0f, 11.0f },
|
||||
{ 12.0f, 13.0f, 14.0f, 15.0f },
|
||||
};
|
||||
Vec4F32 c0 = Vec4F32::Load(col_values[0]);
|
||||
Vec4F32 c1 = Vec4F32::Load(col_values[1]);
|
||||
Vec4F32 c2 = Vec4F32::Load(col_values[2]);
|
||||
Vec4F32 c3 = Vec4F32::Load(col_values[3]);
|
||||
Vec4F32::Transpose(c0, c1, c2, c3);
|
||||
float transposed[16];
|
||||
c0.Store(transposed);
|
||||
c1.Store(transposed + 4);
|
||||
c2.Store(transposed + 8);
|
||||
c3.Store(transposed + 12);
|
||||
for (int row = 0; row < 4; row++) {
|
||||
for (int col = 0; col < 4; col++) {
|
||||
EXPECT_EQ_FLOAT(transposed[row * 4 + col], col_values[col][row]);
|
||||
}
|
||||
}
|
||||
|
||||
// LoadTranspose should match a plain Load of each row followed by Transpose.
|
||||
static const float flat[16] = {
|
||||
0.0f, 1.0f, 2.0f, 3.0f,
|
||||
4.0f, 5.0f, 6.0f, 7.0f,
|
||||
8.0f, 9.0f, 10.0f, 11.0f,
|
||||
12.0f, 13.0f, 14.0f, 15.0f,
|
||||
};
|
||||
Vec4F32 t0, t1, t2, t3;
|
||||
Vec4F32::LoadTranspose(flat, t0, t1, t2, t3);
|
||||
float loadTransposed[16];
|
||||
t0.Store(loadTransposed);
|
||||
t1.Store(loadTransposed + 4);
|
||||
t2.Store(loadTransposed + 8);
|
||||
t3.Store(loadTransposed + 12);
|
||||
if (!CompareFloats(loadTransposed, transposed, 16, __LINE__)) return false;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4F32NaNInf() {
|
||||
const float inf = INFINITY;
|
||||
const float nan = NAN;
|
||||
const float values[4] = { 1.0f, nan, inf, -inf };
|
||||
Vec4F32 v = Vec4F32::Load(values);
|
||||
|
||||
float result[4];
|
||||
|
||||
v.ZeroNaNs().Store(result);
|
||||
EXPECT_EQ_FLOAT(result[0], 1.0f);
|
||||
EXPECT_EQ_FLOAT(result[1], 0.0f);
|
||||
// ZeroNaNs leaves infinities alone.
|
||||
EXPECT_TRUE(std::isinf(result[2]));
|
||||
EXPECT_TRUE(std::isinf(result[3]));
|
||||
|
||||
// The contract is just "a value that yields zero when multiplied by zero" - whatever is
|
||||
// cheapest to get there. So the implementations legitimately differ on what a bad lane
|
||||
// becomes: SSE2 clamps to +-FLT_MAX, NEON, LSX and the scalar fallback zero it. Check the
|
||||
// property rather than the value, and that good lanes are left alone.
|
||||
v.CleanNaNInfs().Store(result);
|
||||
EXPECT_EQ_FLOAT(result[0], 1.0f);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i] * 0.0f, 0.0f);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestVec4F32Loads() {
|
||||
float result[4];
|
||||
|
||||
// LoadF24x3_One: three 24-bit values shifted up into floats, lane 3 forced to 1.0f.
|
||||
const uint32_t f24_values[4] = { 0x3F8000 >> 0, 0x400000, 0x3F0000, 0x123456 };
|
||||
Vec4F32::LoadF24x3_One(f24_values).Store(result);
|
||||
for (int i = 0; i < 3; i++) {
|
||||
float expected;
|
||||
uint32_t bits = f24_values[i] << 8;
|
||||
memcpy(&expected, &bits, 4);
|
||||
EXPECT_EQ_FLOAT(result[i], expected);
|
||||
}
|
||||
EXPECT_EQ_FLOAT(result[3], 1.0f);
|
||||
|
||||
// LoadF24x4: same, but all four lanes come from the source.
|
||||
Vec4F32::LoadF24x4(f24_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
float expected;
|
||||
uint32_t bits = f24_values[i] << 8;
|
||||
memcpy(&expected, &bits, 4);
|
||||
EXPECT_EQ_FLOAT(result[i], expected);
|
||||
}
|
||||
|
||||
// The normalizing loads. Some of these read 8 bytes, so give them room.
|
||||
const int8_t s8_values[8] = { -1, -128, 127, 45, 0, 0, 0, 0 };
|
||||
Vec4F32::LoadS8Norm(s8_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i], (float)s8_values[i] / 128.0f);
|
||||
}
|
||||
|
||||
const int16_t s16_values[8] = { -1, -32768, 32767, 1234, 0, 0, 0, 0 };
|
||||
Vec4F32::LoadS16Norm(s16_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i], (float)s16_values[i] / 32768.0f);
|
||||
}
|
||||
|
||||
const uint8_t u8_values[8] = { 0, 255, 128, 7, 0, 0, 0, 0 };
|
||||
Vec4F32::LoadU8Norm(u8_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_APPROX_EQ_FLOAT(result[i], (float)u8_values[i] / 255.0f);
|
||||
}
|
||||
|
||||
// The converting (non-normalizing) loads.
|
||||
Vec4F32::LoadConvertS16(s16_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i], (float)s16_values[i]);
|
||||
}
|
||||
|
||||
Vec4F32::LoadConvertS8(s8_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i], (float)s8_values[i]);
|
||||
}
|
||||
|
||||
Vec4F32::LoadConvertU8(u8_values).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i], (float)u8_values[i]);
|
||||
}
|
||||
|
||||
// StoreConvertToU8 truncates towards zero and saturates to 0-255.
|
||||
const float to_u8_values[4] = { 0.0f, 255.0f, 12.75f, 200.0f };
|
||||
uint8_t u8_out[4];
|
||||
Vec4F32::Load(to_u8_values).StoreConvertToU8(u8_out);
|
||||
EXPECT_EQ_INT(u8_out[0], 0);
|
||||
EXPECT_EQ_INT(u8_out[1], 255);
|
||||
EXPECT_EQ_INT(u8_out[2], 12);
|
||||
EXPECT_EQ_INT(u8_out[3], 200);
|
||||
|
||||
const float saturate_values[4] = { -1.0f, 300.0f, -0.5f, 1000.0f };
|
||||
Vec4F32::Load(saturate_values).StoreConvertToU8(u8_out);
|
||||
EXPECT_EQ_INT(u8_out[0], 0);
|
||||
EXPECT_EQ_INT(u8_out[1], 255);
|
||||
EXPECT_EQ_INT(u8_out[2], 0);
|
||||
EXPECT_EQ_INT(u8_out[3], 255);
|
||||
|
||||
// Int to float.
|
||||
const int int_values[4] = { 0, -7, 1000, -32768 };
|
||||
Vec4F32::FromVec4S32(Vec4S32::Load(int_values)).Store(result);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
EXPECT_EQ_FLOAT(result[i], (float)int_values[i]);
|
||||
}
|
||||
|
||||
// Load2 only fills the first two lanes.
|
||||
const float two_values[2] = { 3.0f, 4.0f };
|
||||
Vec4F32::Load2(two_values).Store(result);
|
||||
EXPECT_EQ_FLOAT(result[0], 3.0f);
|
||||
EXPECT_EQ_FLOAT(result[1], 4.0f);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool TestMatrices() {
|
||||
static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f };
|
||||
static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f };
|
||||
static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, };
|
||||
float result[16];
|
||||
Mat4F32 a(a_values);
|
||||
Mat4F32 b(b_values);
|
||||
|
||||
Mul4x4By4x4(a, b).Store(result);
|
||||
if (!CompareFloats(result, known_result, 16, __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Mat4x3F32 d = Mat4x3F32(b_values + 2);
|
||||
Mul4x3By4x4(d, a).Store(result);
|
||||
|
||||
static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, };
|
||||
if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f };
|
||||
Vec4F32 v = Vec4F32::Load(vec_values);
|
||||
|
||||
v.AsVec3ByMatrix44(b).Store3(result);
|
||||
|
||||
static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, };
|
||||
if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
Vec4F32 scale = Vec4F32::Load(a_values);
|
||||
Vec4F32 translate = Vec4F32::Load(b_values);
|
||||
|
||||
TranslateAndScaleInplace(a, scale, translate);
|
||||
a.Store(result);
|
||||
|
||||
static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,};
|
||||
if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool TestCrossSIMD() {
|
||||
if (!TestVec4S32()) return false;
|
||||
if (!TestVec4S32Compares()) return false;
|
||||
if (!TestVec4F32Arith()) return false;
|
||||
if (!TestVec4F32Compares()) return false;
|
||||
if (!TestVec4F32Lanes()) return false;
|
||||
if (!TestVec4F32NaNInf()) return false;
|
||||
if (!TestVec4F32Loads()) return false;
|
||||
if (!TestMatrices()) return false;
|
||||
return true;
|
||||
}
|
||||
+1
-82
@@ -2764,88 +2764,6 @@ bool TestSIMD() {
|
||||
return true;
|
||||
}
|
||||
|
||||
static void PrintFloats(const float *f, int count) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
printf("%.1ff, ", f[i]);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
static bool CompareFloats(const float *values, const float *known_good, int count, int line) {
|
||||
int wrongCount = 0;
|
||||
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (values[i] != known_good[i]) {
|
||||
wrongCount++;
|
||||
}
|
||||
}
|
||||
|
||||
if (wrongCount > 0) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
bool wrong = values[i] != known_good[i];
|
||||
printf("%d: %0.3f vs %0.3f %s\n", i + 1, values[i], known_good[i], wrong ? "!! MISMATCH" : "");
|
||||
}
|
||||
printf("At UnitTest.cpp:%d: %d / %d were wrong\n", line, wrongCount, count);
|
||||
return false;
|
||||
} else {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
bool TestCrossSIMD() {
|
||||
static const float a_values[16] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f };
|
||||
static const float b_values[16] = { -12.0f, 3.0f, -2.5f, 5.0f, 31.0f, 0.5f, 4.0f, 6.0f, 7.0f, 13.0f, 12.0f, 51.0f, 81.0f, 32.0f };
|
||||
static const float known_result[16] = { 395.0f, 171.0f, 41.5f, 170.0f, 942.0f, 410.5f, 111.5f, 475.0f, 1358.0f, 607.5f, 163.0f, 728.0f, 297.0f, 49.5f, 25.0f, 160.0f, };
|
||||
float result[16];
|
||||
Mat4F32 a(a_values);
|
||||
Mat4F32 b(b_values);
|
||||
|
||||
Mul4x4By4x4(a, b).Store(result);
|
||||
if (!CompareFloats(result, known_result, 16, __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Mat4x3F32 d = Mat4x3F32(b_values + 2);
|
||||
Mul4x3By4x4(d, a).Store(result);
|
||||
|
||||
static const float known_4x3_result[16] = { 332.5f, 371.0f, 404.5f, 438.0f, 80.5f, 95.0f, 105.5f, 116.0f, 192.0f, 237.0f, 269.0f, 301.0f, 790.0f, 1036.0f, 1185.0f, 1349.0f, };
|
||||
if (!CompareFloats(result, known_4x3_result, 16, __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
static const float vec_values[4] = { 3.0f, 5.0f, 7.0f, 10000000.0f };
|
||||
Vec4F32 v = Vec4F32::Load(vec_values);
|
||||
|
||||
v.AsVec3ByMatrix44(b).Store3(result);
|
||||
|
||||
static const float known_vec_result[3] = { 249.0f, 134.5f, 96.5f, };
|
||||
if (!CompareFloats(result, known_vec_result, ARRAY_SIZE(known_vec_result), __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
Vec4F32 scale = Vec4F32::Load(a_values);
|
||||
Vec4F32 translate = Vec4F32::Load(b_values);
|
||||
|
||||
TranslateAndScaleInplace(a, scale, translate);
|
||||
a.Store(result);
|
||||
|
||||
static const float known_scale_result[16] = { -47.0f, 16.0f, -1.0f, 36.0f, -103.0f, 41.0f, 1.5f, 81.0f, -146.0f, 61.0f, 3.5f, 117.0f, 14.0f, 30.0f, 0.0f, 0.0f,};
|
||||
if (!CompareFloats(result, known_scale_result, ARRAY_SIZE(known_scale_result), __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
s8 values[4] = {-1, -128, 127, 45};
|
||||
float fvalues[4];
|
||||
Vec4F32::LoadS8Norm(values).Store(fvalues);
|
||||
static const float known_s8norm_result[4] = {(float)values[0]/128.0f, (float)values[1]/128.0f, (float)values[2]/128.0f, (float)values[3]/128.0f,};
|
||||
if (!CompareFloats(fvalues, known_s8norm_result, ARRAY_SIZE(known_s8norm_result), __LINE__)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// PrintFloats(result, 16);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool TestVolumeFunc() {
|
||||
for (int i = 0; i <= 20; i++) {
|
||||
float mul = Volume10ToMultiplier(i);
|
||||
@@ -3044,6 +2962,7 @@ struct TestItem {
|
||||
|
||||
bool TestArmEmitter();
|
||||
bool TestArm64Emitter();
|
||||
bool TestCrossSIMD();
|
||||
bool TestX64Emitter();
|
||||
bool TestRiscVEmitter();
|
||||
bool TestLoongArch64Emitter();
|
||||
|
||||
@@ -287,6 +287,7 @@
|
||||
<ClCompile Include="..\Windows\CaptureDevice.cpp" />
|
||||
<ClCompile Include="JitHarness.cpp" />
|
||||
<ClCompile Include="TestArm64Emitter.cpp" />
|
||||
<ClCompile Include="TestCrossSIMD.cpp" />
|
||||
<ClCompile Include="TestIRPassSimplify.cpp" />
|
||||
<ClCompile Include="TestLoongArch64Emitter.cpp" />
|
||||
<ClCompile Include="TestRiscVEmitter.cpp" />
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
<ClCompile Include="TestShaderGenerators.cpp" />
|
||||
<ClCompile Include="TestThreadManager.cpp" />
|
||||
<ClCompile Include="TestSoftwareGPUJit.cpp" />
|
||||
<ClCompile Include="TestCrossSIMD.cpp" />
|
||||
<ClCompile Include="TestIRPassSimplify.cpp" />
|
||||
<ClCompile Include="TestDemangle.cpp" />
|
||||
<ClCompile Include="TestLzrc.cpp" />
|
||||
|
||||
Reference in new issue
Block a user