From 2a5c2fa4778a7c8e6ede883f39ca528b6007b3d3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 30 May 2026 10:15:26 +0200 Subject: [PATCH] Check if the viewport transform matches the clip space. If so we can skip the near clip plane. --- Common/Math/CrossSIMD.h | 2 +- GPU/Common/DrawEngineCommon.cpp | 31 ++++++++++++++++++++----------- GPU/GPUCommon.cpp | 10 ++++++++++ GPU/GPUState.h | 1 + 4 files changed, 32 insertions(+), 12 deletions(-) diff --git a/Common/Math/CrossSIMD.h b/Common/Math/CrossSIMD.h index 9f883354de..00e9c1646b 100644 --- a/Common/Math/CrossSIMD.h +++ b/Common/Math/CrossSIMD.h @@ -184,7 +184,7 @@ inline bool AnyZeroSignBit(Vec4S32 value) { return _mm_movemask_ps(_mm_castsi128_ps(value.v)) != 0xF; } -// These are for evaluating compare masks. +// These are for evaluating compare masks. On some archs it might just check the upper bit. inline bool AllCompareBitsSet(Vec4S32 value) { return _mm_movemask_ps(_mm_castsi128_ps(value.v)) == 0xF; } diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 54ff6d30f6..2bfce7f68b 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -30,6 +30,7 @@ #include "Core/Config.h" #include "GPU/GPUCommon.h" #include "GPU/Common/DrawEngineCommon.h" +#include "GPU/GPUStateSIMDUtil.h" #include "GPU/Common/SplineCommon.h" #include "GPU/Common/DepthRaster.h" #include "GPU/Common/ShaderId.h" @@ -368,12 +369,12 @@ static bool TestBoundingBoxFast(const float *worldViewProj, const void *vdata, i objPos = Vec4F32::Load((const float *)data); break; } - Vec4F32 clipPos = objPos.AsVec3ByMatrix44(worldViewProjMat) & vertexMask; + Vec4F32 clipPos = objPos.AsVec3ByMatrix44(worldViewProjMat) & vertexMask; // Not sure we should do the vertex mask thing. Vec4F32 posW = clipPos.ShuffleWWWW(); Vec4F32 posXY = clipPos.ShuffleXXYY(); Vec4F32 planeDistXY = posXY * planesXY + posW; insideMaskXY |= planeDistXY.CompareGe(Vec4F32::Zero()); - Vec4F32 posZ = clipPos.ShuffleZZZZ(); + Vec4F32 posZ = clipPos.ShuffleZZZZ(); // This means that we compute the Z sides twice. Oh well. Vec4F32 planeDistZ = posZ * planesXY + posW; anyOutsideMaskZ |= planeDistZ.CompareLt(Vec4F32::Zero()); insideMaskZ |= planeDistZ.CompareGe(Vec4F32::Zero()); @@ -386,17 +387,25 @@ static bool TestBoundingBoxFast(const float *worldViewProj, const void *vdata, i } } - if (AllCompareBitsSet(insideMaskXY) && AllCompareBitsSet(insideMaskZ)) { - depths->valid = true; - depths->hitClipSpaceZW = AnyCompareBitsSet(anyOutsideMaskZ); - // Before checking later, we apply a viewport to the projZ, so we can later compare against minZ/maxZ or the outer bounds. - // But to save operations we don't do it here. - depths->minProjZ = minProjZ + vpZOffset; - depths->maxProjZ = maxProjZ + vpZOffset; - return true; - } else { + if (!AllCompareBitsSet(insideMaskXY) || !AllCompareBitsSet(insideMaskZ)) { + // All vertices were outside one side of the clipping cube. We can skip the draw entirely. return false; } + + depths->valid = true; + depths->hitClipSpaceZW = AnyCompareBitsSet(anyOutsideMaskZ); + + // If the W=-Z plane was intersected, here we can go through the vertices again, and check for X/Y bounds for range culling. + // However! We need to find a valid way to do so by "backprojecting" the range culling into clip space, which may be a little tricky. + // + // If nothing is outside the box, the "inversion" cases (vertices hit the boundary after clipping like Flatout, Sengoku Cannon) + // cannot happen, and soft clipping is only needed if the viewport is smaller than the valid Z range. + + // Before checking later, we apply a viewport to the projZ, so we can later compare against minZ/maxZ or the outer bounds. + // But to save operations we don't do it here. + depths->minProjZ = minProjZ + vpZOffset; + depths->maxProjZ = maxProjZ + vpZOffset; + return true; } bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, BoundingDepths *depths) { diff --git a/GPU/GPUCommon.cpp b/GPU/GPUCommon.cpp index 46af1d1189..535b2b823e 100644 --- a/GPU/GPUCommon.cpp +++ b/GPU/GPUCommon.cpp @@ -2252,6 +2252,16 @@ void GPUCommon::UpdateMatrixProducts() { // No funny business, just use worldviewproj for culling. memcpy(gstate_c.cullMatrix, gstate_c.worldviewproj, sizeof(float) * 16); } + + // Now, check the Z range. If the viewport matches the limits of Z, we can avoid the need to do near clipping in many cases + // since the host hardware will take care of it automatically. + const float absZScale = fabsf(gstate.getViewportZScale()); + const float zCenter = gstate.getViewportZCenter(); + const float zMin = gstate.getDepthRangeMin(); + const float frontPlane = zCenter - absZScale; + if (frontPlane == zMin || frontPlane == zMin + 1.0f) { + gstate_c.viewportNearPlaneMatchesOutput = true; + } } // It's just a bit operation, cheaper to clean all three together. gstate_c.Clean(DIRTY_WORLD_VIEW_PROJ_MATRIX | DIRTY_VIEW_PROJ_MATRIX | DIRTY_CULL_MATRIX); diff --git a/GPU/GPUState.h b/GPU/GPUState.h index cc5812e797..707b5123be 100644 --- a/GPU/GPUState.h +++ b/GPU/GPUState.h @@ -668,6 +668,7 @@ public: float viewproj[16]; float worldviewproj[16]; float cullMatrix[16]; + bool viewportNearPlaneMatchesOutput; KnownVertexBounds vertBounds;