Bugfixes, cleanup

This commit is contained in:
Henrik Rydgård committed 2026-05-30 19:07:59 +02:00
1 parent e75d9f8a5f
commit 0e8ba0d6fe
6 files changed
+40 -30

No files matched your search

+7
View File
@@ -399,6 +399,13 @@ bool GenerateFragmentShader(const FShaderID &id, char *buffer, const ShaderLangu
WRITE(p, "uniform float u_mipBias;\n");
}
if (fsMinmaxDiscard) {
*uniformMask |= DIRTY_RASTER_OFFSET; // they're updated together
WRITE(p, "uniform vec2 u_minZmaxZ;\n");
}
// Varyings
WRITE(p, "%s %s lowp vec4 v_color0;\n", shading, compat.varying_fs);
if (lmode) {
WRITE(p, "%s %s lowp vec3 v_color1;\n", shading, compat.varying_fs);
+9 -8
View File
@@ -6,6 +6,7 @@
#include "Common/Data/Convert/SmallDataConvert.h"
#include "Common/Math/lin/matrix4x4.h"
#include "Common/Math/math_util.h"
#include "Common/Math/CrossSIMD.h"
#include "Common/Math/lin/vec3.h"
#include "Common/TimeUtil.h"
#include "GPU/GPUState.h"
@@ -87,8 +88,8 @@ void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBuffe
if (dirtyUniforms & DIRTY_RASTER_OFFSET) {
ub->rasterOffset[0] = gstate.getOffsetX();
ub->rasterOffset[1] = gstate.getOffsetY();
ub->minZmaxZ[0] = gstate.getDepthRangeMin();
ub->minZmaxZ[1] = gstate.getDepthRangeMax();
ub->minZmaxZ[0] = (float)gstate.getDepthRangeMin();
ub->minZmaxZ[1] = (float)gstate.getDepthRangeMax();
// test sine wave
// ub->minZmaxZ[0] = (sin(time_now_d()) * 0.5f + 0.5f) * 65536.0;
@@ -96,17 +97,17 @@ void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBuffe
if (dirtyUniforms & DIRTY_VIEWPORT_UNIFORMS) {
// TODO: This should be a couple of SIMD instructions.
ub->vpScale[0] = gstate.getViewportXScale();
ub->vpScale[1] = gstate.getViewportYScale();
ub->vpScale[2] = gstate.getViewportZScale();
Vec4F32 vpScale = Vec4F32::LoadF24x3_DontCare(&gstate.viewportxscale);
Vec4F32 vpOffset = Vec4F32::LoadF24x3_DontCare(&gstate.viewportxcenter);
vpScale.Store(ub->vpScale);
vpOffset.Store(ub->vpOffset);
ub->NaN = std::numeric_limits<float>::quiet_NaN(); // Used in the shader for range culling.
ub->vpOffset[0] = gstate.getViewportXCenter();
ub->vpOffset[1] = gstate.getViewportYCenter();
ub->vpOffset[2] = gstate.getViewportZCenter();
}
// Transform
if (dirtyUniforms & DIRTY_WORLDMATRIX) {
// TODO: We could change the shader to directly read these "malformed" matrices, but we'd
// be doing the matrix multiplication manually.
ConvertMatrix4x3To3x4Transposed(ub->world, gstate.worldMatrix);
}
if (dirtyUniforms & DIRTY_VIEWMATRIX) {
+1 -1
View File
@@ -297,7 +297,7 @@ void DrawEngineD3D11::Flush() {
useHWTransform = CheckBoundingDepths(useHWTransform);
if (useHWTransform != lastUseHwTransform_) {
gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_RASTER_STATE);
gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_RASTER_STATE);
lastUseHwTransform_ = useHWTransform;
}
+1 -1
View File
@@ -257,7 +257,7 @@ void DrawEngineGLES::Flush() {
useHWTransform = CheckBoundingDepths(useHWTransform);
if (useHWTransform != lastUseHwTransform_) {
gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_RASTER_STATE);
gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_RASTER_STATE);
lastUseHwTransform_ = useHWTransform;
}
+6 -4
View File
@@ -585,6 +585,8 @@ u32 GPUCommonHW::CheckGPUFeatures() const {
features |= GPU_USE_BLEND_MINMAX;
}
// The next three (clipDistance, cullDistance, depthClamp) are the critical features that let us correctly emulate PSP behavior without fallbacks.
// This is except for the case of clipped polygons still reaching outside the guardband, where we still need to software-clip if detected.
if (draw_->GetDeviceCaps().maxClipDistances >= 3) {
features |= GPU_USE_CLIP_DISTANCE;
}
@@ -593,15 +595,15 @@ u32 GPUCommonHW::CheckGPUFeatures() const {
features |= GPU_USE_CULL_DISTANCE;
}
if (draw_->GetDeviceCaps().textureDepthSupported) {
features |= GPU_USE_DEPTH_TEXTURE;
}
if (draw_->GetDeviceCaps().depthClampSupported) {
// Some backends always do GPU_USE_ACCURATE_DEPTH, but it's required for depth clamp.
features |= GPU_USE_DEPTH_CLAMP;
}
if (draw_->GetDeviceCaps().textureDepthSupported) {
features |= GPU_USE_DEPTH_TEXTURE;
}
if (draw_->GetDeviceCaps().framebufferFetchSupported) {
features |= GPU_USE_FRAMEBUFFER_FETCH;
features |= GPU_USE_SHADER_BLENDING; // doesn't matter if we are buffered or not here.
+16 -16
View File
@@ -1176,26 +1176,26 @@ inline void ConvertMatrix4x3To4x4Transposed(float *m4x4, const float *m4x3) {
// 4567
// 89AB
// Don't see a way to SIMD that. Should be pretty fast anyway.
inline void ConvertMatrix4x3To3x4Transposed(float *m4x4, const float *m4x3) {
inline void ConvertMatrix4x3To3x4Transposed(float *m3x4, const float *m4x3) {
#if PPSSPP_ARCH(ARM_NEON)
// vld3q is a perfect match here!
float32x4x3_t packed = vld3q_f32(m4x3);
vst1q_f32(m4x4, packed.val[0]);
vst1q_f32(m4x4 + 4, packed.val[1]);
vst1q_f32(m4x4 + 8, packed.val[2]);
vst1q_f32(m3x4, packed.val[0]);
vst1q_f32(m3x4 + 4, packed.val[1]);
vst1q_f32(m3x4 + 8, packed.val[2]);
#else
m4x4[0] = m4x3[0];
m4x4[1] = m4x3[3];
m4x4[2] = m4x3[6];
m4x4[3] = m4x3[9];
m4x4[4] = m4x3[1];
m4x4[5] = m4x3[4];
m4x4[6] = m4x3[7];
m4x4[7] = m4x3[10];
m4x4[8] = m4x3[2];
m4x4[9] = m4x3[5];
m4x4[10] = m4x3[8];
m4x4[11] = m4x3[11];
m3x4[0] = m4x3[0];
m3x4[1] = m4x3[3];
m3x4[2] = m4x3[6];
m3x4[3] = m4x3[9];
m3x4[4] = m4x3[1];
m3x4[5] = m4x3[4];
m3x4[6] = m4x3[7];
m3x4[7] = m4x3[10];
m3x4[8] = m4x3[2];
m3x4[9] = m4x3[5];
m3x4[10] = m4x3[8];
m3x4[11] = m4x3[11];
#endif
}