From 3f44da709b0a9aa66c6f266d52d3a9e6a8b1a357 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 11:22:01 -0600 Subject: [PATCH] Clamp sampling of video textures and direct-displayed video to the 480x272 frame Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Common/FramebufferManagerCommon.cpp | 8 +++--- GPU/Common/ShaderId.cpp | 5 ++-- GPU/Common/ShaderUniforms.cpp | 35 +++++++++++++++---------- GPU/Common/ShaderUniforms.h | 3 +++ GPU/Common/TextureCacheCommon.cpp | 10 ++++++- GPU/GLES/ShaderManagerGLES.cpp | 21 +++------------ 6 files changed, 44 insertions(+), 38 deletions(-) diff --git a/GPU/Common/FramebufferManagerCommon.cpp b/GPU/Common/FramebufferManagerCommon.cpp index 06ff45d7e9..3e7f55df37 100644 --- a/GPU/Common/FramebufferManagerCommon.cpp +++ b/GPU/Common/FramebufferManagerCommon.cpp @@ -1546,7 +1546,9 @@ Draw::Texture *FramebufferManagerCommon::MakePixelTexture(const u8 *srcPixels, G bool FramebufferManagerCommon::DrawFramebufferToOutput(const DisplayLayoutConfig &config, const u8 *srcPixels, int srcStride, GEBufferFormat srcPixelFormat) { textureCache_->ForgetLastTexture(); - Draw::Texture *pixelsTex = MakePixelTexture(srcPixels, srcPixelFormat, srcStride, 512, 272); + // Upload exactly the displayed 480x272, so that filtering (and post shaders) clamp at the edge of the + // image instead of pulling in what's past it in memory - often garbage, like when this is video. + Draw::Texture *pixelsTex = MakePixelTexture(srcPixels, srcPixelFormat, srcStride, 480, 272); if (!pixelsTex) { return false; } @@ -1557,12 +1559,12 @@ bool FramebufferManagerCommon::DrawFramebufferToOutput(const DisplayLayoutConfig flags |= OutputFlags::BACKBUFFER_FLIPPED; } - constexpr float u0 = 0.0f, u1 = 480.0f / 512.0f; + constexpr float u0 = 0.0f, u1 = 1.0f; constexpr float v0 = 0.0f, v1 = 1.0f; if (useBufferedRendering_) { presentation_->UpdateUniforms(gpu->VideoIsPlaying()); - presentation_->SourceTexture(pixelsTex, 512, 272); + presentation_->SourceTexture(pixelsTex, 480, 272); presentation_->RunPostshaderPasses(config, flags, uvRotation, u0, v0, u1, v1); } diff --git a/GPU/Common/ShaderId.cpp b/GPU/Common/ShaderId.cpp index 72d9932131..b3013e9619 100644 --- a/GPU/Common/ShaderId.cpp +++ b/GPU/Common/ShaderId.cpp @@ -337,8 +337,9 @@ void ComputeFragmentShaderID(FShaderID *id_out, const ComputedPipelineState &pip if (gstate_c.needShaderTexClamp) { // 4 bits total. id.SetBit(FS_BIT_SHADER_TEX_CLAMP); - id.SetBit(FS_BIT_CLAMP_S, gstate.isTexCoordClampedS()); - id.SetBit(FS_BIT_CLAMP_T, gstate.isTexCoordClampedT()); + // Video is always clamped, so it can't wrap around into the garbage outside the frame. + id.SetBit(FS_BIT_CLAMP_S, gstate.isTexCoordClampedS() || gstate_c.textureIsVideo); + id.SetBit(FS_BIT_CLAMP_T, gstate.isTexCoordClampedT() || gstate_c.textureIsVideo); } id.SetBits(FS_BIT_SHADER_DEPAL_MODE, 2, (int)shaderDepalMode); id.SetBits(FS_BIT_SHADER_DEPAL_FORMAT, 3, (int)shaderDepalFormat); diff --git a/GPU/Common/ShaderUniforms.cpp b/GPU/Common/ShaderUniforms.cpp index 4b55d21ba3..748aef9802 100644 --- a/GPU/Common/ShaderUniforms.cpp +++ b/GPU/Common/ShaderUniforms.cpp @@ -32,6 +32,26 @@ void UpdateRotation(float rotMatrix[4], bool useBufferedRendering) { } } +void CalcTexClamp(float texClamp[4], float texClampOffset[2]) { + const float invW = 1.0f / (float)gstate_c.curTextureWidth; + const float invH = 1.0f / (float)gstate_c.curTextureHeight; + int w = gstate.getTextureWidth(0); + int h = gstate.getTextureHeight(0); + if (gstate_c.textureIsVideo) { + // Only the video frame itself, whatever is in memory around it may be garbage. + w = std::min(w, 480); + h = std::min(h, 272); + } + + // First wrap xy, then half texel xy (for clamp.) + texClamp[0] = (float)w * invW; + texClamp[1] = (float)h * invH; + texClamp[2] = invW * 0.5f; + texClamp[3] = invH * 0.5f; + texClampOffset[0] = gstate_c.curTextureXOffset * invW; + texClampOffset[1] = gstate_c.curTextureYOffset * invH; +} + void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBufferedRendering, bool pixelMapped) { if (dirtyUniforms & DIRTY_TEXENV) { Uint8x3ToFloat3(ub->texEnvColor, gstate.texenvcolor); @@ -50,20 +70,7 @@ void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBuffe Uint8x3ToFloat3(ub->blendFixB, gstate.getFixB()); } if (dirtyUniforms & DIRTY_TEXCLAMP) { - const float invW = 1.0f / (float)gstate_c.curTextureWidth; - const float invH = 1.0f / (float)gstate_c.curTextureHeight; - const int w = gstate.getTextureWidth(0); - const int h = gstate.getTextureHeight(0); - const float widthFactor = (float)w * invW; - const float heightFactor = (float)h * invH; - - // First wrap xy, then half texel xy (for clamp.) - ub->texClamp[0] = widthFactor; - ub->texClamp[1] = heightFactor; - ub->texClamp[2] = invW * 0.5f; - ub->texClamp[3] = invH * 0.5f; - ub->texClampOffset[0] = gstate_c.curTextureXOffset * invW; - ub->texClampOffset[1] = gstate_c.curTextureYOffset * invH; + CalcTexClamp(ub->texClamp, ub->texClampOffset); } if (dirtyUniforms & DIRTY_MIPBIAS) { diff --git a/GPU/Common/ShaderUniforms.h b/GPU/Common/ShaderUniforms.h index cc8b9f793f..ebecb5258c 100644 --- a/GPU/Common/ShaderUniforms.h +++ b/GPU/Common/ShaderUniforms.h @@ -113,6 +113,9 @@ uint32_t PackDepalBits(bool pixelMapped); void UpdateFogCoef(const GEState &state, float fogCoef[2]); +// Computes u_texclamp and u_texclampoff. Only meaningful when gstate_c.needShaderTexClamp is set. +void CalcTexClamp(float texClamp[4], float texClampOffset[2]); + // This happens so much that I want it inline. inline void UpdateUVScaleOff(const GEState &state, float uvScaleOff[4]) { if (gstate_c.textureIsFramebuffer) { diff --git a/GPU/Common/TextureCacheCommon.cpp b/GPU/Common/TextureCacheCommon.cpp index 19bacfbb8f..912f50aece 100644 --- a/GPU/Common/TextureCacheCommon.cpp +++ b/GPU/Common/TextureCacheCommon.cpp @@ -937,7 +937,15 @@ TextureApplyResult TextureCacheCommon::ApplyTexture(bool doBind) { TextureApplyResult TextureCacheCommon::ApplyTextureFinish(TexCacheEntry *entry, bool doBind) { _dbg_assert_(entry); - gstate_c.SetTextureIsVideo((entry->status & TexStatus::VIDEO) != 0); + const bool isVideo = (entry->status & TexStatus::VIDEO) != 0; + gstate_c.SetTextureIsVideo(isVideo); + if (isVideo) { + // Restrict sampling to the 480x272 frame (see CalcTexClamp), since with the forced linear + // filtering, whatever is next to it in memory bleeds in at the edges. + gstate_c.curTextureXOffset = 0; + gstate_c.curTextureYOffset = 0; + gstate_c.SetNeedShaderTexclamp(true); + } gstate_c.SetTextureIsArray(false); // Ordinary 2D textures still aren't used by array view in VK. We probably might as well, though, at this point.. gstate_c.SetTextureIsFramebuffer(false); entry->lastFrame = gpuStats.totals.numFlips; diff --git a/GPU/GLES/ShaderManagerGLES.cpp b/GPU/GLES/ShaderManagerGLES.cpp index d1a41862d4..5b027cb6eb 100644 --- a/GPU/GLES/ShaderManagerGLES.cpp +++ b/GPU/GLES/ShaderManagerGLES.cpp @@ -453,24 +453,9 @@ void LinkedShader::UpdateUniforms(const ShaderID &vsid, const ShaderLanguageDesc } if ((dirty & DIRTY_TEXCLAMP) && u_texclamp != -1) { - const float invW = 1.0f / (float)gstate_c.curTextureWidth; - const float invH = 1.0f / (float)gstate_c.curTextureHeight; - const int w = gstate.getTextureWidth(0); - const int h = gstate.getTextureHeight(0); - const float widthFactor = (float)w * invW; - const float heightFactor = (float)h * invH; - - // First wrap xy, then half texel xy (for clamp.) - const float texclamp[4] = { - widthFactor, - heightFactor, - invW * 0.5f, - invH * 0.5f, - }; - const float texclampoff[2] = { - gstate_c.curTextureXOffset * invW, - gstate_c.curTextureYOffset * invH, - }; + float texclamp[4]; + float texclampoff[2]; + CalcTexClamp(texclamp, texclampoff); render_->SetUniformF(&u_texclamp, 4, texclamp); if (u_texclampoff != -1) { render_->SetUniformF(&u_texclampoff, 2, texclampoff);