diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 47eb2dbe8a..c61c9aa003 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -32,6 +32,7 @@ #include "GPU/Common/DrawEngineCommon.h" #include "GPU/Common/SplineCommon.h" #include "GPU/Common/DepthRaster.h" +#include "GPU/Common/ShaderId.h" #include "GPU/Common/VertexDecoderCommon.h" #include "GPU/Common/SoftwareTransformCommon.h" #include "GPU/ge_constants.h" @@ -464,6 +465,35 @@ bool DrawEngineCommon::TestBoundingBoxFast(const float *worldViewProj, const voi } } +bool DrawEngineCommon::CheckBoundingDepths(bool useHWTransform) const { + if (useHWTransform && boundingDepths_.valid) { + if (boundingDepths_.hitClipSpaceZW) { + // Revert to software transform so we can clip more accurately. + // + // This is only really needed for two known games: Flatout (water) and Sengoku Cannon (pink geometry). But there may + // be some more. + return false; + } + + if (needFragmentMinMaxClipping()) { + if ((boundingDepths_.minProjZ < gstate.getDepthRangeMin() || boundingDepths_.maxProjZ > gstate.getDepthRangeMax())) { + // Revert to software transform so we can clamp more accurately. + return false; + } + } + + if (needFragmentDepthClamp()) { + if ((boundingDepths_.minProjZ < 0 || boundingDepths_.maxProjZ > 65535)) { + // Revert to software transform so we can clamp more accurately. + return false; + } + } + // Also handle clamping in software if it's not supported in hardware (or always?) + return true; + } + return useHWTransform; +} + // 2D bounding box test against scissor. No indexing yet. // Only supports non-indexed draws with float positions. TODO: Add more float formats. bool DrawEngineCommon::TestBoundingBoxThrough(const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, int *bytesRead) { diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index 78b8fd0326..97951e4720 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -186,6 +186,8 @@ public: protected: virtual bool UpdateUseHWTessellation(bool enabled) const { return enabled; } + bool CheckBoundingDepths(bool useHwTransform) const; + void DecodeVerts(const VertexDecoder *dec, u8 *dest); int DecodeInds(); diff --git a/GPU/Common/ShaderId.cpp b/GPU/Common/ShaderId.cpp index f8b6591381..1a47979f78 100644 --- a/GPU/Common/ShaderId.cpp +++ b/GPU/Common/ShaderId.cpp @@ -13,18 +13,6 @@ #include "GPU/Common/ShaderId.h" #include "GPU/Common/VertexDecoderCommon.h" - -// Shared ID checks for when the vertex and fragment shaders need to coordinate. - -// NOTE: Both of these assume non - through - mode.Don't check these if in through mode. -static bool needFragmentMinMaxClipping() { - return gstate.getDepthRangeMin() != 0 && gstate.getDepthRangeMax() != 0xFFFF && !gstate_c.Use(GPU_USE_CLIP_DISTANCE); -} -static bool needFragmentDepthClamp() { - // If gstate.isDepthClipEnabled is false, clamping does not happen, instead fragments are culled as normal. - return (gstate.getDepthRangeMin() == 0 || gstate.getDepthRangeMax() == 0xFFFF) && gstate.isDepthClipEnabled() && !gstate_c.Use(GPU_USE_DEPTH_CLAMP); -} - std::string VertexShaderDesc(const VShaderID &id) { std::stringstream desc; desc << StringFromFormat("%08x:%08x ", id.d[1], id.d[0]); diff --git a/GPU/Common/ShaderId.h b/GPU/Common/ShaderId.h index 04e3957750..e2e302019b 100644 --- a/GPU/Common/ShaderId.h +++ b/GPU/Common/ShaderId.h @@ -5,6 +5,20 @@ #include #include "Common/CommonFuncs.h" +#include "GPU/GPUState.h" + + +// Shared ID checks for when the vertex and fragment shaders (and host code) need to coordinate. + +// NOTE: Both of these assume non - through - mode.Don't check these if in through mode. +inline bool needFragmentMinMaxClipping() { + return gstate.getDepthRangeMin() != 0 && gstate.getDepthRangeMax() != 0xFFFF && !gstate_c.Use(GPU_USE_CLIP_DISTANCE); +} + +inline bool needFragmentDepthClamp() { + // If gstate.isDepthClipEnabled is false, clamping does not happen, instead fragments are culled as normal. + return (gstate.getDepthRangeMin() == 0 || gstate.getDepthRangeMax() == 0xFFFF) && gstate.isDepthClipEnabled() && !gstate_c.Use(GPU_USE_DEPTH_CLAMP); +} // VS_BIT_LIGHT_UBERSHADER indicates that some groups of these will be // sent to the shader and processed there. This cuts down the number of shaders ("ubershader approach"). diff --git a/GPU/Common/SoftwareTransformCommon.cpp b/GPU/Common/SoftwareTransformCommon.cpp index 28d8b30e69..fe540f8c42 100644 --- a/GPU/Common/SoftwareTransformCommon.cpp +++ b/GPU/Common/SoftwareTransformCommon.cpp @@ -409,7 +409,7 @@ void SoftwareTransform::ProjectVertices(TransformedVertex *transformed, int vert // TODO: Move this to ProjectClipAndExpand. const float w = transformed[i].pos_w; const float recip = 1.0f / w; - Lin::Vec3 xyz = vpOffset + vpScale.scaledBy(Lin::Vec3(transformed[i].x * recip, transformed[i].y * recip, transformed[i].z * recip)); + Lin::Vec3 xyz = vpOffset + vpScale.scaledBy(Lin::Vec3(transformed[i].x, transformed[i].y, transformed[i].z)) * recip; transformed[i].x = xyz.x; transformed[i].y = xyz.y; transformed[i].z = xyz.z; @@ -421,7 +421,7 @@ void SoftwareTransform::ProjectVertices(TransformedVertex *transformed, int vert for (int i = 0; i < vertexCount; i++) { Vec4F32 xyzw = Vec4F32::Load(&transformed[i].x); Vec4F32 wRecip = Vec4F32::Splat(1.0f / transformed[i].pos_w); - Vec4F32 projected = xyzw * wRecip * vpScale + vpOffset; + Vec4F32 projected = xyzw * vpScale * wRecip + vpOffset; // Now, we need to restore the W value as we'll still need it later. projected.WithLane3From(xyzw).Store(&transformed[i].x); } diff --git a/GPU/Common/VertexShaderGenerator.cpp b/GPU/Common/VertexShaderGenerator.cpp index cb165e6509..a37767e700 100644 --- a/GPU/Common/VertexShaderGenerator.cpp +++ b/GPU/Common/VertexShaderGenerator.cpp @@ -876,7 +876,8 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag // Perform the perspective projection and viewport transform. (We'll have to undo the division before passing the coordinate along). // In software transform mode, this is performed in on the CPU. - WRITE(p, " outPos.xyz = (outPos.xyz / outPos.w) * u_vpScale.xyz + u_vpOffset.xyz;\n"); + WRITE(p, " float recip = 1.0 / outPos.w;\n"); + WRITE(p, " outPos.xyz = outPos.xyz * u_vpScale.xyz * recip + u_vpOffset.xyz;\n"); if (fsMinmaxDiscard || fsDepthClamp) { WRITE(p, " %sv_zw = vec2(outPos.z * outPos.w, outPos.w);\n", compat.vsOutPrefix); diff --git a/GPU/D3D11/DrawEngineD3D11.cpp b/GPU/D3D11/DrawEngineD3D11.cpp index 6e5aa8f9d4..2e4c97759f 100644 --- a/GPU/D3D11/DrawEngineD3D11.cpp +++ b/GPU/D3D11/DrawEngineD3D11.cpp @@ -294,18 +294,7 @@ void DrawEngineD3D11::Flush() { // Always use software for flat shading to fix the provoking index. bool tess = gstate_c.submitType == SubmitType::HW_BEZIER || gstate_c.submitType == SubmitType::HW_SPLINE; bool useHWTransform = CanUseHardwareTransform(prim) && (tess || gstate.getShadeMode() != GE_SHADE_FLAT); - - if (useHWTransform && boundingDepths_.valid) { - if (boundingDepths_.hitClipSpaceZW) { - // Revert to software transform so we can clip more accurately. - useHWTransform = false; - } - if ((boundingDepths_.minProjZ < 0.0 || boundingDepths_.maxProjZ > 65535.0) && !gstate_c.Use(GPU_USE_DEPTH_CLAMP)) { - // Revert to software transform so we can clamp more accurately. - useHWTransform = false; - } - // Also handle clamping in software if it's not supported in hardware (or always?) - } + useHWTransform = CheckBoundingDepths(useHWTransform); if (useHWTransform != lastUseHwTransform_) { gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_RASTER_STATE); diff --git a/GPU/GLES/DrawEngineGLES.cpp b/GPU/GLES/DrawEngineGLES.cpp index cf582575d6..9f2001de2a 100644 --- a/GPU/GLES/DrawEngineGLES.cpp +++ b/GPU/GLES/DrawEngineGLES.cpp @@ -254,18 +254,7 @@ void DrawEngineGLES::Flush() { GEPrimitiveType prim = prevPrim_; bool useHWTransform = CanUseHardwareTransform(prim); - - if (useHWTransform && boundingDepths_.valid) { - if (boundingDepths_.hitClipSpaceZW) { - // Revert to software transform so we can clip more accurately. - useHWTransform = false; - } - if ((boundingDepths_.minProjZ < 0.0 || boundingDepths_.maxProjZ > 65535.0) && !gstate_c.Use(GPU_USE_DEPTH_CLAMP)) { - // Revert to software transform so we can clamp more accurately. - useHWTransform = false; - } - // Also handle clamping in software if it's not supported in hardware (or always?) - } + useHWTransform = CheckBoundingDepths(useHWTransform); if (useHWTransform != lastUseHwTransform_) { gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_RASTER_STATE); diff --git a/GPU/Vulkan/DrawEngineVulkan.cpp b/GPU/Vulkan/DrawEngineVulkan.cpp index 2f737c7c58..051fbe6454 100644 --- a/GPU/Vulkan/DrawEngineVulkan.cpp +++ b/GPU/Vulkan/DrawEngineVulkan.cpp @@ -241,17 +241,7 @@ void DrawEngineVulkan::Flush() { provokingVertexOk = true; } bool useHWTransform = CanUseHardwareTransform(prim) && provokingVertexOk; - if (useHWTransform && boundingDepths_.valid) { - if (boundingDepths_.hitClipSpaceZW) { - // Revert to software transform so we can clip more accurately. - useHWTransform = false; - } - if ((boundingDepths_.minProjZ < gstate.getDepthRangeMin() || boundingDepths_.maxProjZ > gstate.getDepthRangeMax()) && !gstate_c.Use(GPU_USE_DEPTH_CLAMP)) { - // Revert to software transform so we can clamp more accurately. - useHWTransform = false; - } - // Also handle clamping in software if it's not supported in hardware (or always?) - } + useHWTransform = CheckBoundingDepths(useHWTransform); if (useHWTransform != lastUseHwTransform_) { // Need to re-evaluate software transform fallbacks.