diff --git a/GPU/Common/FragmentShaderGenerator.cpp b/GPU/Common/FragmentShaderGenerator.cpp index c5658d9f73..c8d40987fc 100644 --- a/GPU/Common/FragmentShaderGenerator.cpp +++ b/GPU/Common/FragmentShaderGenerator.cpp @@ -399,6 +399,13 @@ bool GenerateFragmentShader(const FShaderID &id, char *buffer, const ShaderLangu WRITE(p, "uniform float u_mipBias;\n"); } + if (fsMinmaxDiscard) { + *uniformMask |= DIRTY_RASTER_OFFSET; // they're updated together + WRITE(p, "uniform vec2 u_minZmaxZ;\n"); + } + + // Varyings + WRITE(p, "%s %s lowp vec4 v_color0;\n", shading, compat.varying_fs); if (lmode) { WRITE(p, "%s %s lowp vec3 v_color1;\n", shading, compat.varying_fs); diff --git a/GPU/Common/ShaderUniforms.cpp b/GPU/Common/ShaderUniforms.cpp index db937e2024..85647a2ec9 100644 --- a/GPU/Common/ShaderUniforms.cpp +++ b/GPU/Common/ShaderUniforms.cpp @@ -6,6 +6,7 @@ #include "Common/Data/Convert/SmallDataConvert.h" #include "Common/Math/lin/matrix4x4.h" #include "Common/Math/math_util.h" +#include "Common/Math/CrossSIMD.h" #include "Common/Math/lin/vec3.h" #include "Common/TimeUtil.h" #include "GPU/GPUState.h" @@ -87,8 +88,8 @@ void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBuffe if (dirtyUniforms & DIRTY_RASTER_OFFSET) { ub->rasterOffset[0] = gstate.getOffsetX(); ub->rasterOffset[1] = gstate.getOffsetY(); - ub->minZmaxZ[0] = gstate.getDepthRangeMin(); - ub->minZmaxZ[1] = gstate.getDepthRangeMax(); + ub->minZmaxZ[0] = (float)gstate.getDepthRangeMin(); + ub->minZmaxZ[1] = (float)gstate.getDepthRangeMax(); // test sine wave // ub->minZmaxZ[0] = (sin(time_now_d()) * 0.5f + 0.5f) * 65536.0; @@ -96,17 +97,17 @@ void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBuffe if (dirtyUniforms & DIRTY_VIEWPORT_UNIFORMS) { // TODO: This should be a couple of SIMD instructions. - ub->vpScale[0] = gstate.getViewportXScale(); - ub->vpScale[1] = gstate.getViewportYScale(); - ub->vpScale[2] = gstate.getViewportZScale(); + Vec4F32 vpScale = Vec4F32::LoadF24x3_DontCare(&gstate.viewportxscale); + Vec4F32 vpOffset = Vec4F32::LoadF24x3_DontCare(&gstate.viewportxcenter); + vpScale.Store(ub->vpScale); + vpOffset.Store(ub->vpOffset); ub->NaN = std::numeric_limits::quiet_NaN(); // Used in the shader for range culling. - ub->vpOffset[0] = gstate.getViewportXCenter(); - ub->vpOffset[1] = gstate.getViewportYCenter(); - ub->vpOffset[2] = gstate.getViewportZCenter(); } // Transform if (dirtyUniforms & DIRTY_WORLDMATRIX) { + // TODO: We could change the shader to directly read these "malformed" matrices, but we'd + // be doing the matrix multiplication manually. ConvertMatrix4x3To3x4Transposed(ub->world, gstate.worldMatrix); } if (dirtyUniforms & DIRTY_VIEWMATRIX) { diff --git a/GPU/D3D11/DrawEngineD3D11.cpp b/GPU/D3D11/DrawEngineD3D11.cpp index 2e4c97759f..cbe11cdaa1 100644 --- a/GPU/D3D11/DrawEngineD3D11.cpp +++ b/GPU/D3D11/DrawEngineD3D11.cpp @@ -297,7 +297,7 @@ void DrawEngineD3D11::Flush() { useHWTransform = CheckBoundingDepths(useHWTransform); if (useHWTransform != lastUseHwTransform_) { - gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_RASTER_STATE); + gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_RASTER_STATE); lastUseHwTransform_ = useHWTransform; } diff --git a/GPU/GLES/DrawEngineGLES.cpp b/GPU/GLES/DrawEngineGLES.cpp index 9f2001de2a..db62285c62 100644 --- a/GPU/GLES/DrawEngineGLES.cpp +++ b/GPU/GLES/DrawEngineGLES.cpp @@ -257,7 +257,7 @@ void DrawEngineGLES::Flush() { useHWTransform = CheckBoundingDepths(useHWTransform); if (useHWTransform != lastUseHwTransform_) { - gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_RASTER_STATE); + gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_RASTER_STATE); lastUseHwTransform_ = useHWTransform; } diff --git a/GPU/GPUCommonHW.cpp b/GPU/GPUCommonHW.cpp index cb89adf832..ae040ab162 100644 --- a/GPU/GPUCommonHW.cpp +++ b/GPU/GPUCommonHW.cpp @@ -585,6 +585,8 @@ u32 GPUCommonHW::CheckGPUFeatures() const { features |= GPU_USE_BLEND_MINMAX; } + // The next three (clipDistance, cullDistance, depthClamp) are the critical features that let us correctly emulate PSP behavior without fallbacks. + // This is except for the case of clipped polygons still reaching outside the guardband, where we still need to software-clip if detected. if (draw_->GetDeviceCaps().maxClipDistances >= 3) { features |= GPU_USE_CLIP_DISTANCE; } @@ -593,15 +595,15 @@ u32 GPUCommonHW::CheckGPUFeatures() const { features |= GPU_USE_CULL_DISTANCE; } - if (draw_->GetDeviceCaps().textureDepthSupported) { - features |= GPU_USE_DEPTH_TEXTURE; - } - if (draw_->GetDeviceCaps().depthClampSupported) { // Some backends always do GPU_USE_ACCURATE_DEPTH, but it's required for depth clamp. features |= GPU_USE_DEPTH_CLAMP; } + if (draw_->GetDeviceCaps().textureDepthSupported) { + features |= GPU_USE_DEPTH_TEXTURE; + } + if (draw_->GetDeviceCaps().framebufferFetchSupported) { features |= GPU_USE_FRAMEBUFFER_FETCH; features |= GPU_USE_SHADER_BLENDING; // doesn't matter if we are buffered or not here. diff --git a/GPU/Math3D.h b/GPU/Math3D.h index eb85efa4d0..7cbbe796cb 100644 --- a/GPU/Math3D.h +++ b/GPU/Math3D.h @@ -1176,26 +1176,26 @@ inline void ConvertMatrix4x3To4x4Transposed(float *m4x4, const float *m4x3) { // 4567 // 89AB // Don't see a way to SIMD that. Should be pretty fast anyway. -inline void ConvertMatrix4x3To3x4Transposed(float *m4x4, const float *m4x3) { +inline void ConvertMatrix4x3To3x4Transposed(float *m3x4, const float *m4x3) { #if PPSSPP_ARCH(ARM_NEON) // vld3q is a perfect match here! float32x4x3_t packed = vld3q_f32(m4x3); - vst1q_f32(m4x4, packed.val[0]); - vst1q_f32(m4x4 + 4, packed.val[1]); - vst1q_f32(m4x4 + 8, packed.val[2]); + vst1q_f32(m3x4, packed.val[0]); + vst1q_f32(m3x4 + 4, packed.val[1]); + vst1q_f32(m3x4 + 8, packed.val[2]); #else - m4x4[0] = m4x3[0]; - m4x4[1] = m4x3[3]; - m4x4[2] = m4x3[6]; - m4x4[3] = m4x3[9]; - m4x4[4] = m4x3[1]; - m4x4[5] = m4x3[4]; - m4x4[6] = m4x3[7]; - m4x4[7] = m4x3[10]; - m4x4[8] = m4x3[2]; - m4x4[9] = m4x3[5]; - m4x4[10] = m4x3[8]; - m4x4[11] = m4x3[11]; + m3x4[0] = m4x3[0]; + m3x4[1] = m4x3[3]; + m3x4[2] = m4x3[6]; + m3x4[3] = m4x3[9]; + m3x4[4] = m4x3[1]; + m3x4[5] = m4x3[4]; + m3x4[6] = m4x3[7]; + m3x4[7] = m4x3[10]; + m3x4[8] = m4x3[2]; + m3x4[9] = m4x3[5]; + m3x4[10] = m4x3[8]; + m3x4[11] = m4x3[11]; #endif }