diff --git a/Common/Math/CrossSIMD.h b/Common/Math/CrossSIMD.h index 3f35343fd4..91f8bd4d2b 100644 --- a/Common/Math/CrossSIMD.h +++ b/Common/Math/CrossSIMD.h @@ -250,6 +250,25 @@ struct Vec4F32 { return Vec4F32{_mm_castsi128_ps(_mm_slli_epi32(_mm_loadu_si128((const __m128i *)src), 8))}; } + float Dot3(Vec4F32 b) { + // Zero out the W component before multiplying to ensure only X, Y, Z are summed + alignas(16) static const uint32_t mask[4] = { 0xFFFFFFFF, 0xFFFFFFFF, 0xFFFFFFFF, 0x0 }; + __m128 masked = _mm_and_ps(v, _mm_load_ps((const float *)mask)); + __m128 mul = _mm_mul_ps(masked, b.v); + __m128 shuf1 = _mm_shuffle_ps(mul, mul, _MM_SHUFFLE(2, 1, 0, 3)); + __m128 sum1 = _mm_add_ps(mul, shuf1); + __m128 shuf2 = _mm_shuffle_ps(sum1, sum1, _MM_SHUFFLE(1, 0, 3, 2)); + return _mm_cvtss_f32(_mm_add_ps(sum1, shuf2)); + } + + float Dot4(Vec4F32 b) { + __m128 mul = _mm_mul_ps(v, b.v); + __m128 shuf1 = _mm_shuffle_ps(mul, mul, _MM_SHUFFLE(2, 1, 0, 3)); + __m128 sum1 = _mm_add_ps(mul, shuf1); + __m128 shuf2 = _mm_shuffle_ps(sum1, sum1, _MM_SHUFFLE(1, 0, 3, 2)); + return _mm_cvtss_f32(_mm_add_ps(sum1, shuf2)); + } + void Store(float *dst) { _mm_storeu_ps(dst, v); } void Store2(float *dst) { _mm_storel_epi64((__m128i *)dst, _mm_castps_si128(v)); } void StoreAligned(float *dst) { _mm_store_ps(dst, v); } @@ -791,6 +810,34 @@ struct Vec4F32 { return Vec4F32{ sum }; } + float Dot3(Vec4F32 b) { + // Zero out the W component before multiplying to ensure only X, Y, Z are summed + float32x4_t masked = vsetq_lane_f32(0.0f, v, 3); + float32x4_t mul = vmulq_f32(masked, b.v); +#if PPSSPP_ARCH(ARM64_NEON) + return vaddvq_f32(mul); +#else + float32x2_t sum_low = vget_low_f32(mul); + float32x2_t sum_high = vget_high_f32(mul); + float32x2_t sum = vadd_f32(sum_low, sum_high); + sum = vpadd_f32(sum, sum); + return vget_lane_f32(sum, 0); +#endif + } + + float Dot4(Vec4F32 b) { + float32x4_t mul = vmulq_f32(v, b.v); +#if PPSSPP_ARCH(ARM64_NEON) + return vaddvq_f32(mul); +#else + float32x2_t sum_low = vget_low_f32(mul); + float32x2_t sum_high = vget_high_f32(mul); + float32x2_t sum = vadd_f32(sum_low, sum_high); + sum = vpadd_f32(sum, sum); + return vget_lane_f32(sum, 0); +#endif + } + template float GetLane() const { return vgetq_lane_f32(v, i); } @@ -1122,7 +1169,19 @@ struct Vec4F32 { } // NOTE: May be slow. - float operator[](size_t index) const { return ((float *)&v)[index]; } + float operator[](size_t index) const { + // index is a compile-time constant parameter to the intrinsic. + int ival; + switch (index) { + case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0); + case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1); + case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2); + default: ival = __lsx_vpickve2gr_w((__m128i)v, 3); + } + float fval; + memcpy(&fval, &ival, sizeof(float)); + return fval; + } Vec4F32 operator +(Vec4F32 other) const { return Vec4F32{ (__m128)__lsx_vfadd_s(v, other.v) }; } Vec4F32 operator -(Vec4F32 other) const { return Vec4F32{ (__m128)__lsx_vfsub_s(v, other.v) }; } @@ -1255,8 +1314,39 @@ struct Vec4F32 { return Vec4F32{ sum }; } + float Dot3(Vec4F32 b) { + // Zero out the W component before multiplying to ensure only X, Y, Z are summed + __m128 masked = (__m128)__lsx_vinsgr2vr_w((__m128i)v, 0, 3); + __m128 mul = (__m128)__lsx_vfmul_s(masked, b.v); + // Sum all elements: horizontal add + __m128 shuf1 = (__m128)__lsx_vshuf4i_w((__m128i)mul, 0b10110001); // Rotate elements + __m128 sum1 = (__m128)__lsx_vfadd_s(mul, shuf1); + __m128 shuf2 = (__m128)__lsx_vshuf4i_w((__m128i)sum1, 0b01001110); // Swap pairs + __m128 sum2 = (__m128)__lsx_vfadd_s(sum1, shuf2); + float fval; + int ival = __lsx_vpickve2gr_w((__m128i)sum2, 0); + memcpy(&fval, &ival, sizeof(float)); + return fval; + } + + float Dot4(Vec4F32 b) { + __m128 mul = (__m128)__lsx_vfmul_s(v, b.v); + // Sum all elements: horizontal add + __m128 shuf1 = (__m128)__lsx_vshuf4i_w((__m128i)mul, 0b10110001); // Rotate elements + __m128 sum1 = (__m128)__lsx_vfadd_s(mul, shuf1); + __m128 shuf2 = (__m128)__lsx_vshuf4i_w((__m128i)sum1, 0b01001110); // Swap pairs + __m128 sum2 = (__m128)__lsx_vfadd_s(sum1, shuf2); + float fval; + int ival = __lsx_vpickve2gr_w((__m128i)sum2, 0); + memcpy(&fval, &ival, sizeof(float)); + return fval; + } + template float GetLane() const { - return __lsx_vpickve2gr_w((__m128i)v, i); + int ival = __lsx_vpickve2gr_w((__m128i)v, i); + float fval; + memcpy(&fval, &ival, sizeof(float)); + return fval; } }; @@ -1758,6 +1848,15 @@ struct Vec4F32 { return Vec4F32{ { x, y, z, 1.0f } }; } + float Dot3(Vec4F32 b) { + // Only sum the first three elements (X, Y, Z), ignore W + return v[0] * b.v[0] + v[1] * b.v[1] + v[2] * b.v[2]; + } + + float Dot4(Vec4F32 b) { + return v[0] * b.v[0] + v[1] * b.v[1] + v[2] * b.v[2] + v[3] * b.v[3]; + } + template float GetLane() const { return v[i]; } diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 86f7e346ac..0148d95125 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -349,8 +349,8 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int // off to hardware, with whatever capabilities are available. // // NOTE: This doesn't handle through-mode or indexing (morph or skinning can be handled if they're implemented in software during decode). -template -static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, ClipInfoFlags *clipInfoFlags) { +template +static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, const void *idata, int vertexCount, const VertexDecoder *dec, ClipInfoFlags *clipInfoFlags) { Mat4F32 cullMat(cullMatrix); alignas(16) static const float planesXYData[4] = { 1, -1, 1, -1 }; Vec4F32 planesXY = Vec4F32::LoadAligned(planesXYData); @@ -362,14 +362,36 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int // In reality we should probably affect X and Y too, but meh. alignas(16) static const u32 vertexMaskData[4] = {0xFFFFFFFF, 0xFFFFFFFF, 0xFFFFFF00, 0xFFFFFFFF}; const int stride = dec->VertexSize(); - const s8 *data = (const s8 *)vdata + dec->posoff; + const s8 *srcdata = (const s8 *)vdata + dec->posoff; + const s8 *data = srcdata; const float vpZScale = gstate.getViewportZScale(); float minProjZ = FLT_MAX; float maxProjZ = -FLT_MAX; - for (int i = 0; i < vertexCount; i++, data += stride) { + for (int i = 0; i < vertexCount; i++) { + switch (idxFmt) { + case GE_VTYPE_IDX_8BIT: + { + u8 idx = ((u8 *)idata)[i]; + data = (const s8 *)srcdata + idx * stride; + break; + } + case GE_VTYPE_IDX_16BIT: + { + u16 idx = ((u16 *)idata)[i]; + data = (const s8 *)srcdata + idx * stride; + break; + } + case GE_VTYPE_IDX_32BIT: + { + u32 idx = ((u32 *)idata)[i]; + data = (const s8 *)srcdata + idx * stride; + break; + } + } + Vec4F32 objPos; switch (posFmt) { case GE_VTYPE_POS_8BIT: @@ -398,6 +420,10 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int if (projZ > maxProjZ) { // else ruins the minss/maxss optimization. maxProjZ = projZ; } + + if (idxFmt == GE_VTYPE_IDX_NONE) { + data += stride; + } } if (!AllCompareBitsSet(insideMaskXY) || !AllCompareBitsSet(insideMaskZ)) { @@ -447,7 +473,7 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int return true; } -bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *flags) { +bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void *vdata, const void *idata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *flags) { // Although this may lead to drawing that shouldn't happen, the viewport is more complex on VR. // Let's always say objects are within bounds. if (gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) { @@ -456,18 +482,50 @@ bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void * return false; } - switch (vertType & GE_VTYPE_POS_MASK) { - case GE_VTYPE_POS_8BIT: - return ::TestBoundingBoxFast(cullMatrix, vdata, vertexCount, dec, flags); - case GE_VTYPE_POS_16BIT: - return ::TestBoundingBoxFast(cullMatrix, vdata, vertexCount, dec, flags); - case GE_VTYPE_POS_FLOAT: - return ::TestBoundingBoxFast(cullMatrix, vdata, vertexCount, dec, flags); + // Dispatching like this is a bit ugly, but we want to avoid every possible overhead *inside* TestBoundingBoxFast. + // That said, I'm not 100% sure it's worth it.. + switch (vertType & GE_VTYPE_IDX_MASK) { + case GE_VTYPE_IDX_NONE: + switch (vertType & GE_VTYPE_POS_MASK) { + case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, nullptr, vertexCount, dec, flags); + case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, nullptr, vertexCount, dec, flags); + case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast(cullMatrix, vdata, nullptr, vertexCount, dec, flags); + default: + break; + } + break; + case GE_VTYPE_IDX_8BIT: + switch (vertType & GE_VTYPE_POS_MASK) { + case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + default: + break; + } + break; + case GE_VTYPE_IDX_16BIT: + switch (vertType & GE_VTYPE_POS_MASK) { + case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + default: + break; + } + break; + case GE_VTYPE_IDX_32BIT: + switch (vertType & GE_VTYPE_POS_MASK) { + case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast(cullMatrix, vdata, idata, vertexCount, dec, flags); + default: + break; + } + break; default: - // Shouldn't end up here with the checks outside this function. - _dbg_assert_(false); - return true; + break; } + _dbg_assert_(false); + return true; } // 2D bounding box test against scissor. No indexing yet. diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index cd00baffc7..a11e8c389f 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -116,7 +116,7 @@ public: // This is a less accurate version of TestBoundingBox, but faster. Can have more false positives. // Doesn't support indexing. - bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *clipInfoFlags); + bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, const void *idata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *clipInfoFlags); bool TestBoundingBoxThrough(const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, int *bytesRead); bool EstimateThroughPrimSafeSize(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, const VertexDecoder *dec, u32 vertType, int *safeWidth, int *safeHeight); diff --git a/GPU/Common/SoftwareTransformCommon.cpp b/GPU/Common/SoftwareTransformCommon.cpp index ae9e9d5d37..7c4de459ee 100644 --- a/GPU/Common/SoftwareTransformCommon.cpp +++ b/GPU/Common/SoftwareTransformCommon.cpp @@ -212,7 +212,7 @@ SoftwareTransformAction RunSoftwareTransform(SoftwareTransformParams ¶ms, in if (matchingComponents && stencilNotMasked) { DepthScaleFactors depthScale = GetDepthScaleFactors(gstate_c.UseFlags()); // Need to rescale from a [0, 1] float. This is the final transformed value. - float depth = depthScale.EncodeFromU16((float)(int)(transformed[1].z * 65535.0f)); + float depth = depthScale.EncodeFromU16(transformed[1].z); // Non-zero depth clears are unusual, but some drivers don't match drawn depth values to cleared values. // Games sometimes expect exact matches (see #12626, for example) for equal comparisons. if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && result->depth > 0.0f && result->depth < 1.0f)) { diff --git a/GPU/GPUCommonHW.cpp b/GPU/GPUCommonHW.cpp index 6164babf5e..b4c79808a9 100644 --- a/GPU/GPUCommonHW.cpp +++ b/GPU/GPUCommonHW.cpp @@ -1016,7 +1016,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) { // For now, turn off culling on platforms where we don't have SIMD bounding box tests, like RISC-V. #if PPSSPP_ARCH(ARM_NEON) || PPSSPP_ARCH(SSE2) -#define PASSES_CULLING ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_MORPHCOUNT_MASK | GE_VTYPE_WEIGHT_MASK | GE_VTYPE_IDX_MASK)) || count > MAX_CULL_CHECK_COUNT) +#define PASSES_CULLING ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_MORPHCOUNT_MASK | GE_VTYPE_WEIGHT_MASK)) || count > MAX_CULL_CHECK_COUNT) #else @@ -1030,7 +1030,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) { bool passCulling = PASSES_CULLING; if (!passCulling) { // Do software culling. - if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, count, decoder, vertexType, &flags)) { + if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, inds, count, decoder, vertexType, &flags)) { passCulling = true; } else { gpuStats.perFrame.numCulledDraws++; @@ -1120,8 +1120,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) { ClipInfoFlags flags{}; if (!passCulling) { // Do software culling. - _dbg_assert_((vertexType & GE_VTYPE_IDX_MASK) == GE_VTYPE_IDX_NONE); - if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, count, decoder, vertexType, &flags)) { + if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, inds, count, decoder, vertexType, &flags)) { passCulling = true; } else { gpuStats.perFrame.numCulledDraws++;