From 16ea2cbf88e48646c66a0621aa8cc0e89c6c2322 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 14:52:49 -0600 Subject: [PATCH 1/9] GPU: Fix draw engine buffer overruns and stale vertex data - Flush before the queued draws would decode more than VERTEX_BUFFER_MAX vertices. The batch was limited by index count, which doesn't bound a sparse index range, and DecodeVerts silently stopped while DecodeInds still emitted indices for the undecoded draws. - Give TestBoundingBox its own scratch buffer. It used offsets in decoded_, which can hold decoded vertices that aren't flushed yet. - Read 32-bit indices the way the PSP does, ignoring the upper 16 bits. IndexConverter and the fast bounding box test used all 32, so a game setting them indexed far past the decoded vertices. - D3D11: Flush in FinishDeferred like the other backends, since indices are still read from PSP memory at flush time (#10095). - Don't JIT new vertex decoders once the code space is full. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Common/DrawEngineCommon.cpp | 62 ++++++++++++++++++++---------- GPU/Common/DrawEngineCommon.h | 15 ++++++++ GPU/Common/IndexGenerator.cpp | 1 + GPU/Common/VertexDecoderCommon.cpp | 9 ++++- GPU/Common/VertexDecoderCommon.h | 3 +- GPU/D3D11/DrawEngineD3D11.h | 4 +- GPU/GLES/DrawEngineGLES.cpp | 1 + GPU/Vulkan/DrawEngineVulkan.cpp | 1 + GPU/ge_constants.h | 2 +- 9 files changed, 73 insertions(+), 25 deletions(-) diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 09bf16b891..527f21b256 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -51,11 +51,13 @@ DrawEngineCommon::DrawEngineCommon() : decoderMap_(32) { transformedExpanded_ = (TransformedVertex *)AllocateMemoryPages(3 * TRANSFORMED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); decoded_ = (u8 *)AllocateMemoryPages(DECODED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); decIndex_ = (u16 *)AllocateMemoryPages(DECODED_INDEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); + bboxScratch_ = (u8 *)AllocateMemoryPages(BBOX_SCRATCH_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); _dbg_assert_(transformed_); _dbg_assert_(transformedExpanded_); _dbg_assert_(decoded_); _dbg_assert_(decIndex_); + _dbg_assert_(bboxScratch_); indexGen.Setup(decIndex_); @@ -65,6 +67,7 @@ DrawEngineCommon::DrawEngineCommon() : decoderMap_(32) { DrawEngineCommon::~DrawEngineCommon() { FreeMemoryPages(decoded_, DECODED_VERTEX_BUFFER_SIZE); FreeMemoryPages(decIndex_, DECODED_INDEX_BUFFER_SIZE); + FreeMemoryPages(bboxScratch_, BBOX_SCRATCH_SIZE); FreeMemoryPages(transformed_, TRANSFORMED_VERTEX_BUFFER_SIZE); FreeMemoryPages(transformedExpanded_, 3 * TRANSFORMED_VERTEX_BUFFER_SIZE); ShutdownDepthRaster(); @@ -183,15 +186,14 @@ void DrawEngineCommon::DispatchSubmitImm(GEPrimitiveType prim, TransformedVertex // - Less accurate, but.. // - Only requires six plane evaluations then. bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, const VertexDecoder *dec, u32 vertType) { - // Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws. - // Although this may lead to drawing that shouldn't happen, the viewport is more complex on VR. - // Let's always say objects are within bounds. + // The scratch buffer is sized for 1024 vertices. Although this may lead to drawing that shouldn't happen, + // the viewport is more complex on VR. Let's always say objects are within bounds. if (vertexCount > 1024 || gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) { return true; } - SimpleVertex *corners = (SimpleVertex *)(decoded_ + 65536 * 12); - float *verts = (float *)(decoded_ + 65536 * 18); + SimpleVertex *corners = (SimpleVertex *)(bboxScratch_ + BBOX_SCRATCH_CORNERS_OFFSET); + float *verts = (float *)(bboxScratch_ + BBOX_SCRATCH_VERTS_OFFSET); // Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder // and a large vertex format. @@ -218,7 +220,7 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int } } else { // Simplify away indices, bones, and morph before proceeding. - u8 *temp_buffer = decoded_ + 65536 * 24; + u8 *temp_buffer = bboxScratch_ + BBOX_SCRATCH_TEMP_OFFSET; if ((inds || (vertType & (GE_VTYPE_WEIGHT_MASK | GE_VTYPE_MORPHCOUNT_MASK)))) { // Need for Speed Carbon ends up on this path! With a single bone weight. @@ -392,7 +394,8 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, cons } case GE_VTYPE_IDX_32BIT: { - u32 idx = ((u32 *)idata)[i]; + // The PSP ignores the upper 16 bits. + u16 idx = (u16)((u32 *)idata)[i]; data = (const s8 *)srcdata + idx * stride; break; } @@ -759,7 +762,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t * if (IsTrianglePrim(newPrim) != isTriangle) break; int vertexCount = data & 0xFFFF; - if (numDrawInds >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + offset + vertexCount > VERTEX_BUFFER_MAX) { + if (numDrawInds >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + offset + vertexCount > VERTEX_BUFFER_MAX || numVertsToDecode_ + (offset - dv.vertexCount) + vertexCount > VERTEX_BUFFER_MAX) { break; } DeferredInds &di = drawInds_[numDrawInds++]; @@ -780,6 +783,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t * dv.vertexCount = offset; dv.indexUpperBound = dv.vertexCount - 1; vertexCountInDrawCalls_ += totalCount; + numVertsToDecode_ += totalCount; *bytesRead = totalCount * dec->VertexSize(); return cmd - start; } @@ -805,7 +809,21 @@ void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, const Ver // vertTypeID is the vertex type but with the UVGen mode smashed into the top bits. bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, const VertexDecoder *dec, u32 vertTypeID, bool clockwise, int *bytesRead, ClipInfoFlags clipInfoFlags) { - if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawVerts_ >= MAX_DEFERRED_DRAW_VERTS || numDrawInds_ >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + vertexCount > VERTEX_BUFFER_MAX) { + // The index count doesn't bound how many vertices DecodeVerts will produce (the index range can be + // sparse), so track that separately, as the growth of the range to decode. + u16 lowerBound = 0; + u16 upperBound = 0; + int decodeGrowth = 0; + if (vertexCount > 0) { + GetIndexBounds(inds, vertexCount, vertTypeID, &lowerBound, &upperBound); + decodeGrowth = upperBound - lowerBound + 1; + if (CanExtendDecode(verts, inds, dec)) { + const DeferredVerts &last = drawVerts_[numDrawVerts_ - 1]; + decodeGrowth = std::max(upperBound, last.indexUpperBound) - std::min(lowerBound, last.indexLowerBound) - (last.indexUpperBound - last.indexLowerBound); + } + } + + if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawVerts_ >= MAX_DEFERRED_DRAW_VERTS || numDrawInds_ >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + vertexCount > VERTEX_BUFFER_MAX || numVertsToDecode_ + decodeGrowth > VERTEX_BUFFER_MAX) { Flush(); } @@ -855,6 +873,7 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti const int rem = vertexCount % 3; if (rem != 0) { vertexCount -= rem; + GetIndexBounds(inds, vertexCount, vertTypeID, &lowerBound, &upperBound); } } @@ -881,17 +900,16 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti _dbg_assert_(numDrawVerts <= MAX_DEFERRED_DRAW_VERTS); - if (inds && numDrawVerts > decodeVertsCounter_ && drawVerts_[numDrawVerts - 1].verts == verts && !applySkin) { + if (CanExtendDecode(verts, inds, dec_)) { // Same vertex pointer as a previous un-decoded draw call - let's just extend the decode! di.vertDecodeIndex = numDrawVerts - 1; - u16 lb; - u16 ub; - GetIndexBounds(inds, vertexCount, vertTypeID, &lb, &ub); DeferredVerts &dv = drawVerts_[numDrawVerts - 1]; - if (lb < dv.indexLowerBound) - dv.indexLowerBound = lb; - if (ub > dv.indexUpperBound) - dv.indexUpperBound = ub; + const int oldCount = dv.indexUpperBound - dv.indexLowerBound + 1; + if (lowerBound < dv.indexLowerBound) + dv.indexLowerBound = lowerBound; + if (upperBound > dv.indexUpperBound) + dv.indexUpperBound = upperBound; + numVertsToDecode_ += dv.indexUpperBound - dv.indexLowerBound + 1 - oldCount; } else { // Record a new draw, and a new index gen. DeferredVerts &dv = drawVerts_[numDrawVerts]; @@ -899,8 +917,9 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti dv.verts = verts; dv.vertexCount = vertexCount; dv.uvScale = LoadUVScaleOffset(gstate); - // Does handle the unindexed case. - GetIndexBounds(inds, vertexCount, vertTypeID, &dv.indexLowerBound, &dv.indexUpperBound); + dv.indexLowerBound = lowerBound; + dv.indexUpperBound = upperBound; + numVertsToDecode_ += upperBound - lowerBound + 1; } vertexCountInDrawCalls_ += vertexCount; @@ -930,8 +949,9 @@ void DrawEngineCommon::DecodeVerts(const VertexDecoder *dec, u8 *dest) { drawVertexOffsets_[i] = numDecodedVerts - indexLowerBound; const int indexUpperBound = dv.indexUpperBound; const int count = indexUpperBound - indexLowerBound + 1; - if (count + numDecodedVerts >= VERTEX_BUFFER_MAX) { - // Hit our limit! Stop decoding in this draw. + if (count + numDecodedVerts > VERTEX_BUFFER_MAX) { + // SubmitPrim flushes before this can happen. + _dbg_assert_(false); break; } diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index f361e55ff3..0333809963 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -38,6 +38,11 @@ enum { VERTEX_BUFFER_MAX = 65536, DECODED_VERTEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 2 * 36, // 36 == sizeof(SimpleVertex) DECODED_INDEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 6 * 6 * 2, // * 6 for spline tessellation, then * 6 again for converting into points/lines, and * 2 for 2 bytes per index + // TestBoundingBox handles up to 1025 vertices: corners (SimpleVertex), then positions, then decoded vertices. + BBOX_SCRATCH_CORNERS_OFFSET = 0, + BBOX_SCRATCH_VERTS_OFFSET = 64 * 1024, + BBOX_SCRATCH_TEMP_OFFSET = 128 * 1024, + BBOX_SCRATCH_SIZE = 256 * 1024, }; enum { @@ -161,6 +166,11 @@ protected: void DecodeVerts(const VertexDecoder *dec, u8 *dest); int DecodeInds(); + // Whether an indexed draw can share the previous draw's vertex decode, by widening its index range. + bool CanExtendDecode(const void *verts, const void *inds, const VertexDecoder *dec) const { + return inds && numDrawVerts_ > decodeVertsCounter_ && drawVerts_[numDrawVerts_ - 1].verts == verts && !dec->skinInDecode; + } + int ComputeNumVertsToDecode() const; void ApplyFramebufferRead(FBOTexState *fboTexState); @@ -211,6 +221,7 @@ protected: numDrawVerts_ = 0; numDrawInds_ = 0; vertexCountInDrawCalls_ = 0; + numVertsToDecode_ = 0; decodeIndsCounter_ = 0; decodeVertsCounter_ = 0; seenPrims_ = 0; @@ -273,6 +284,8 @@ protected: // Vertex collector buffers u8 *decoded_ = nullptr; u16 *decIndex_ = nullptr; + // Separate from decoded_, which can hold decoded vertices that haven't been flushed yet. + u8 *bboxScratch_ = nullptr; // Cached vertex decoders DenseHashMap decoderMap_; @@ -312,6 +325,8 @@ protected: int numDrawVerts_ = 0; int numDrawInds_ = 0; int vertexCountInDrawCalls_ = 0; + // How many vertices DecodeVerts will produce for the queued draws. Must stay <= VERTEX_BUFFER_MAX. + int numVertsToDecode_ = 0; int decodeVertsCounter_ = 0; int decodeIndsCounter_ = 0; diff --git a/GPU/Common/IndexGenerator.cpp b/GPU/Common/IndexGenerator.cpp index ee84be35e3..e83eeb597c 100644 --- a/GPU/Common/IndexGenerator.cpp +++ b/GPU/Common/IndexGenerator.cpp @@ -342,6 +342,7 @@ void IndexGenerator::TranslatePrim(int prim, int numInds, const u16_le *inds, in } } +// The PSP ignores the upper 16 bits of 32-bit indices. The u16 output drops them the same way. void IndexGenerator::TranslatePrim(int prim, int numInds, const u32_le *inds, int indexOffset, bool clockwise) { switch (prim) { case GE_PRIM_POINTS: TranslatePoints(numInds, inds, indexOffset); break; diff --git a/GPU/Common/VertexDecoderCommon.cpp b/GPU/Common/VertexDecoderCommon.cpp index cd452a04c2..e98f5a302b 100644 --- a/GPU/Common/VertexDecoderCommon.cpp +++ b/GPU/Common/VertexDecoderCommon.cpp @@ -170,8 +170,9 @@ void GetIndexBounds(const void *inds, int count, u32 vertType, u16 *indexLowerBo bool oob = false; const u32_le *ind32 = (const u32_le *)inds; for (int i = 0; i < count; i++) { + // The PSP ignores the upper 16 bits, so only the low ones count. const u16 value = (u16)ind32[i]; - // These aren't documented and should be rare. Let's bounds check each one. + // Games setting the upper bits should be rare, so report them. if (ind32[i] != value) { oob = true; } @@ -1373,6 +1374,12 @@ void VertexDecoder::SetVertexType(u32 fmt, const VertexDecoderOptions &options, // Attempt to JIT as well. But only do that if the main CPU JIT is enabled, in order to aid // debugging attempts - if the main JIT doesn't work, this one won't do any better, probably. if (jitCache) { + // Compile doesn't check for space. We can't clear the cache here since other decoders point into it, + // so when it's full (only seen with garbage display lists), new decoders use the interpreter. + if (jitCache->GetSpaceLeft() < 4096) { + WARN_LOG(Log::G3D, "Vertex decoder JIT cache full, using the interpreter for %08x", fmt_); + return; + } jitted_ = jitCache->Compile(*this, &jittedSize_); if (!jitted_) { WARN_LOG(Log::G3D, "Vertex decoder JIT failed! fmt = %08x (%s)", fmt_, GetString(SHADER_STRING_SHORT_DESC).c_str()); diff --git a/GPU/Common/VertexDecoderCommon.h b/GPU/Common/VertexDecoderCommon.h index 23a9babde4..5bb99a919e 100644 --- a/GPU/Common/VertexDecoderCommon.h +++ b/GPU/Common/VertexDecoderCommon.h @@ -115,7 +115,8 @@ public: case GE_VTYPE_IDX_16BIT: return indices16[index]; case GE_VTYPE_IDX_32BIT: - return indices32[index]; + // The PSP supports 32-bit indices in name only: it ignores the upper 16 bits. + return indices32[index] & 0xFFFF; default: return index; } diff --git a/GPU/D3D11/DrawEngineD3D11.h b/GPU/D3D11/DrawEngineD3D11.h index 3944f72a36..3be8d21983 100644 --- a/GPU/D3D11/DrawEngineD3D11.h +++ b/GPU/D3D11/DrawEngineD3D11.h @@ -63,7 +63,9 @@ public: void Flush() override; void FinishDeferred() { - DecodeVerts(dec_, decoded_); + // Decoding only the vertices isn't enough: the indices are still read from PSP memory at flush + // time, and the game may change them once it regains control (#10095). + Flush(); } void NotifyConfigChanged() override; diff --git a/GPU/GLES/DrawEngineGLES.cpp b/GPU/GLES/DrawEngineGLES.cpp index a86ff17852..4111aa2bd4 100644 --- a/GPU/GLES/DrawEngineGLES.cpp +++ b/GPU/GLES/DrawEngineGLES.cpp @@ -229,6 +229,7 @@ void DrawEngineGLES::Flush() { numDrawVerts_ = 0; numDrawInds_ = 0; vertexCountInDrawCalls_ = 0; + numVertsToDecode_ = 0; decodeVertsCounter_ = 0; decodeIndsCounter_ = 0; return; diff --git a/GPU/Vulkan/DrawEngineVulkan.cpp b/GPU/Vulkan/DrawEngineVulkan.cpp index 05b32d69b0..17dc21a94f 100644 --- a/GPU/Vulkan/DrawEngineVulkan.cpp +++ b/GPU/Vulkan/DrawEngineVulkan.cpp @@ -550,6 +550,7 @@ void DrawEngineVulkan::ResetAfterSkippedDraw() { numDrawVerts_ = 0; numDrawInds_ = 0; vertexCountInDrawCalls_ = 0; + numVertsToDecode_ = 0; decodeIndsCounter_ = 0; decodeVertsCounter_ = 0; gstate_c.vertexFullAlpha = true; diff --git a/GPU/ge_constants.h b/GPU/ge_constants.h index 779faa6e81..52a481c066 100644 --- a/GPU/ge_constants.h +++ b/GPU/ge_constants.h @@ -330,7 +330,7 @@ enum GEVertexType : uint32_t { GE_VTYPE_IDX_NONE = (0<<11), GE_VTYPE_IDX_8BIT = (1<<11), GE_VTYPE_IDX_16BIT = (2<<11), - GE_VTYPE_IDX_32BIT = (3<<11), + GE_VTYPE_IDX_32BIT = (3<<11), // In name only: the hardware ignores the upper 16 bits of each index. GE_VTYPE_IDX_MASK = (3<<11), #define GE_VTYPE_IDX_SHIFT 11 }; From 507bb0801b05fabeab9b6d9469a3b89f46b42287 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 14:52:50 -0600 Subject: [PATCH 2/9] GE: Complete lists dropped on error, flush before immediate draws - A list dropped for a bad pc or a GE error stayed RUNNING: its ID was never freed and sceGeListSync on it never returned. Complete it like a finished one. - FlushImm switches to through mode and another vertex decoder, so flush the queued draws first even when the immediate flags match. - Clear leftover temporary GE breakpoints when setting or clearing the next break, so a step that never got there doesn't trip later. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/GPUCommon.cpp | 26 ++++++++++++++++++++++++-- GPU/GPUCommon.h | 1 + 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/GPU/GPUCommon.cpp b/GPU/GPUCommon.cpp index 0bd1aadab7..67a1ea807f 100644 --- a/GPU/GPUCommon.cpp +++ b/GPU/GPUCommon.cpp @@ -957,6 +957,7 @@ DLResult GPUCommon::ProcessDLQueue() { // Nothing to execute here, and leaving it at the head of the queue would block everything // behind it for good. Treat it like a list that ran into an error. ERROR_LOG(Log::G3D, "Display list %d has a bad pc %08x (state %d), dropping it", listIndex, list.pc, (int)list.state); + CompleteFailedList(list); dlQueue.erase(std::remove(dlQueue.begin(), dlQueue.end(), listIndex), dlQueue.end()); continue; } @@ -1045,7 +1046,8 @@ DLResult GPUCommon::ProcessDLQueue() { } break; case GPUSTATE_ERROR: - // don't do anything - though dunno about error... + // The list can't continue. It's removed from the queue below. + CompleteFailedList(list); break; case GPUSTATE_STALL: // Resume work on this same display list later. The GE is still busy with what it has @@ -1390,6 +1392,20 @@ void GPUCommon::Execute_End(u32 op, u32 diff) { } } +// A list dropped on an error still has to complete like a finished one, or its ID is never freed and +// sceGeListSync on it never returns. +void GPUCommon::CompleteFailedList(DisplayList &list) { + if (list.started && list.context.IsValid()) { + gstate.Restore(list.context); + ReapplyGfxState(); + list.started = false; + } + list.state = PSP_GE_DL_STATE_COMPLETED; + list.waitUntilTicks = startingTicks + cyclesExecuted; + busyTicks = std::max(busyTicks, list.waitUntilTicks); + __GeTriggerSync(GPU_SYNC_LIST, list.id, list.waitUntilTicks); +} + void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) { // Just resetting, nothing to check bounds for. const u32 count = op & 0xFFFF; @@ -1552,8 +1568,10 @@ void GPUCommon::FlushImm() { bool changed = texturing != prevTexturing || cullEnable != prevCullEnable || dither != prevDither; changed = changed || prevShading != shading || prevFog != fog; + // Always flush, even if the flags match: DispatchSubmitImm switches to through mode and a different + // vertex decoder, which would otherwise apply to the draws already queued. + Flush(); if (changed) { - Flush(); gstate.antiAliasEnable = (GE_CMD_ANTIALIASENABLE << 24) | (int)antialias; gstate.shademodel = (GE_CMD_SHADEMODE << 24) | (int)shading; gstate.cullfaceEnable = (GE_CMD_CULLFACEENABLE << 24) | (int)cullEnable; @@ -2290,12 +2308,16 @@ bool GPUCommon::NeedsSlowInterpreter() const { void GPUCommon::ClearBreakNext() { breakNext_ = GPUDebug::BreakNext::NONE; breakAtCount_ = -1; + // A step that never reached its target leaves these behind, and they'd trip unexpectedly later. + breakpoints_.ClearTempBreakpoints(); GPUStepping::ResumeFromStepping(); } void GPUCommon::SetBreakNext(GPUDebug::BreakNext next) { breakNext_ = next; breakAtCount_ = -1; + // Drop the ones from a previous step that didn't get there, before adding this one's. + breakpoints_.ClearTempBreakpoints(); switch (next) { case GPUDebug::BreakNext::TEX: breakpoints_.AddTextureChangeTempBreakpoint(); diff --git a/GPU/GPUCommon.h b/GPU/GPUCommon.h index 384a18102f..a295a970c3 100644 --- a/GPU/GPUCommon.h +++ b/GPU/GPUCommon.h @@ -452,4 +452,5 @@ protected: private: void DoExecuteCall(u32 target); void PopDLQueue(); + void CompleteFailedList(DisplayList &list); }; From d6f8615d2a045daaf37e8657a93b8fc43871afdc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 14:52:50 -0600 Subject: [PATCH 3/9] SoftGPU: Fix block transfer and self-render overlap tracking - The block transfer overlap check passed the stride in pixels where bytes are expected, so it only covered part of the rectangle. - A selfrender/selfdepth flush in UpdateState dropped the current draw's pending writes and reads, so later transfers didn't wait for it. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Software/BinManager.cpp | 15 +++++++++++++-- GPU/Software/SoftGpu.cpp | 4 ++-- 2 files changed, 15 insertions(+), 4 deletions(-) diff --git a/GPU/Software/BinManager.cpp b/GPU/Software/BinManager.cpp index 6430fad5ed..cd2c553ab1 100644 --- a/GPU/Software/BinManager.cpp +++ b/GPU/Software/BinManager.cpp @@ -246,21 +246,32 @@ void BinManager::UpdateState() { if (newMaxTasks > MAX_POSSIBLE_TASKS) newMaxTasks = MAX_POSSIBLE_TASKS; // We don't want to overlap wrong, so flush any pending. + bool flushed = false; if (maxTasks_ != newMaxTasks) { maxTasks_ = newMaxTasks; Flush("selfrender"); + flushed = true; } - pendingOverlap_ = pendingOverlap_ || selfRender; // Lastly, we have to check if we're newly writing depth we were texturing before. // This happens in Call of Duty (depth clear after depth texture), for example. - if (!hadDepth && state.pixelID.depthWrite) { + if (!flushed && !hadDepth && state.pixelID.depthWrite) { for (size_t i = 0; i < states_.Size(); ++i) { if (HasTextureWrite(states_.Peek(i))) { Flush("selfdepth"); + flushed = true; + break; } } } + + if (flushed) { + // The flush forgot what this draw writes and reads, so record it again. + MarkPendingWrites(state); + MarkPendingReads(state); + ClearDirty(SoftDirty::BINNER_RANGE); + } + pendingOverlap_ = pendingOverlap_ || selfRender; ClearDirty(SoftDirty::BINNER_OVERLAP); } } diff --git a/GPU/Software/SoftGpu.cpp b/GPU/Software/SoftGpu.cpp index e84dc2a9d6..5cb910ae92 100644 --- a/GPU/Software/SoftGpu.cpp +++ b/GPU/Software/SoftGpu.cpp @@ -826,8 +826,8 @@ void SoftGPU::Execute_BlockTransferStart(u32 op, u32 diff) { // Need to flush both source and target, so we overwrite properly. if (Memory::IsValidRange(src, srcSize) && Memory::IsValidRange(dst, dstSize)) { - drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", false, src, srcStride, width * bpp, height); - drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", true, dst, dstStride, width * bpp, height); + drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", false, src, srcStride * bpp, width * bpp, height); + drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", true, dst, dstStride * bpp, width * bpp, height); } else { drawEngine_->transformUnit.Flush(this, "blockxfer_wrap"); } From 71210f3fa23ca5a761dcb5994fdeb0bb38d4ce5f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 14:52:51 -0600 Subject: [PATCH 4/9] GE recorder: Don't let a second request hijack a recording RecordNextFrame now refuses while a recording is active and during frame dump playback (which asserted on the next replay). The callback handoff to the CPU thread is locked, and a failed file open no longer crashes. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Debugger/Record.cpp | 44 +++++++++++++++++++++++++++++------------ GPU/Debugger/Record.h | 5 ++++- 2 files changed, 35 insertions(+), 14 deletions(-) diff --git a/GPU/Debugger/Record.cpp b/GPU/Debugger/Record.cpp index 38330805c3..5a542e22cf 100644 --- a/GPU/Debugger/Record.cpp +++ b/GPU/Debugger/Record.cpp @@ -126,13 +126,16 @@ void Recorder::DirtyDrawnVRAM() { } bool Recorder::BeginRecording() { + std::unique_lock guard(callbackLock_); + nextFrame = false; if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) { - // Can't record a GE dump. + // Can't record a GE dump. RecordNextFrame refuses this too. + writeCallback = nullptr; return false; } active = true; - nextFrame = false; + guard.unlock(); lastTextures.clear(); lastRenderTargets.clear(); flipLastAction = gpuStats.totals.numFlips; @@ -180,6 +183,10 @@ Path Recorder::WriteRecording() { NOTICE_LOG(Log::G3D, "Recording filename: %s", filename.c_str()); FILE *fp = File::OpenCFile(filename, "wb"); + if (!fp) { + ERROR_LOG(Log::G3D, "Failed to open '%s' for writing the recording", filename.c_str()); + return Path(); + } Header header{}; memcpy(header.magic, HEADER_MAGIC, sizeof(header.magic)); header.version = VERSION; @@ -574,14 +581,19 @@ void Recorder::EmitBezierSpline(u32 op) { } bool Recorder::RecordNextFrame(const std::function callback) { - if (!nextFrame) { - flipLastAction = gpuStats.totals.numFlips; - flipFinishAt = -1; - writeCallback = callback; - nextFrame = true; - return true; + if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) { + return false; } - return false; + std::lock_guard guard(callbackLock_); + // Don't take over a recording in progress, it would get the wrong callback and end point. + if (nextFrame || active) { + return false; + } + flipLastAction = gpuStats.totals.numFlips; + flipFinishAt = -1; + writeCallback = callback; + nextFrame = true; + return true; } void Recorder::FinishRecording() { @@ -596,15 +608,21 @@ void Recorder::FinishRecording() { lastVRAM.clear(); NOTICE_LOG(Log::System, "Recording finished"); - active = false; flipLastAction = gpuStats.totals.numFlips; flipFinishAt = -1; lastEdramTrans = 0x400; - if (writeCallback) { - writeCallback(filename); + std::function callback; + { + std::lock_guard guard(callbackLock_); + callback = std::move(writeCallback); + writeCallback = nullptr; + active = false; + } + + if (callback && !filename.empty()) { + callback(filename); } - writeCallback = nullptr; } void Recorder::CheckEdramTrans() { diff --git a/GPU/Debugger/Record.h b/GPU/Debugger/Record.h index 514ca1c04c..3c291057af 100644 --- a/GPU/Debugger/Record.h +++ b/GPU/Debugger/Record.h @@ -19,6 +19,7 @@ #include #include +#include #include #include @@ -50,7 +51,7 @@ public: } bool RecordNextFrame(const std::function callback); void ClearCallback() { - // Not super thread safe.. + std::lock_guard guard(callbackLock_); writeCallback = nullptr; } @@ -95,6 +96,8 @@ private: int flipFinishAt = -1; uint32_t lastEdramTrans = 0x400; std::function writeCallback; + // RecordNextFrame is called from other threads. Guards writeCallback, nextFrame and the writes to active. + std::mutex callbackLock_; std::vector pushbuf; std::vector commands; From 71bc3187dbed272a1b6e1033aafd0c64f6372619 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 14:52:51 -0600 Subject: [PATCH 5/9] Vulkan: Reload the shader cache after a device restore DeviceLost saves the cache and clears everything, and the next save wrote back only what was drawn since, so each Android background/resume cycle shrank the cache. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Vulkan/GPU_Vulkan.cpp | 26 +++++++++++++++++--------- GPU/Vulkan/GPU_Vulkan.h | 2 +- 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/GPU/Vulkan/GPU_Vulkan.cpp b/GPU/Vulkan/GPU_Vulkan.cpp index dfda1df455..5d69cc61fc 100644 --- a/GPU/Vulkan/GPU_Vulkan.cpp +++ b/GPU/Vulkan/GPU_Vulkan.cpp @@ -90,7 +90,7 @@ GPU_Vulkan::GPU_Vulkan(GraphicsContext *gfxCtx, Draw::DrawContext *draw) if (discID.size()) { File::CreateFullPath(GetSysDirectory(DIRECTORY_APP_CACHE)); shaderCachePath_ = GetSysDirectory(DIRECTORY_APP_CACHE) / (discID + ".vkshadercache"); - LoadCache(shaderCachePath_); + LoadCache(shaderCachePath_, true); } InitDeviceObjects(); @@ -101,7 +101,7 @@ void GPU_Vulkan::FinishInitOnMainThread() { framebufferManagerVulkan_->Init(msaaLevel_); } -void GPU_Vulkan::LoadCache(const Path &filename) { +void GPU_Vulkan::LoadCache(const Path &filename, bool waitForPipelines) { _dbg_assert_(draw_); if (!g_Config.bShaderCache) { WARN_LOG(Log::G3D, "Shader cache disabled. Not loading."); @@ -134,13 +134,15 @@ void GPU_Vulkan::LoadCache(const Path &filename) { } fclose(f); - // Now, since we're on the loader thread, we can just block here until all pipelines are actually created. - // This makes it so that the on-screen spinner keeps spinning until we are done. - double start = time_now_d(); - VulkanRenderManager *rm = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER); - int maxTasksSeen = rm->WaitForPipelines(); - double seconds = time_now_d() - start; - INFO_LOG(Log::G3D, "Waited %0.1fms for at least %d pipeline tasks to finish compiling.", seconds * 1000.0, maxTasksSeen); + if (waitForPipelines) { + // Now, since we're on the loader thread, we can just block here until all pipelines are actually created. + // This makes it so that the on-screen spinner keeps spinning until we are done. + double start = time_now_d(); + VulkanRenderManager *rm = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER); + int maxTasksSeen = rm->WaitForPipelines(); + double seconds = time_now_d() - start; + INFO_LOG(Log::G3D, "Waited %0.1fms for at least %d pipeline tasks to finish compiling.", seconds * 1000.0, maxTasksSeen); + } if (!result) { WARN_LOG(Log::G3D, "Incompatible Vulkan pipeline cache - rebuilding."); @@ -402,6 +404,12 @@ void GPU_Vulkan::DeviceRestore(Draw::DrawContext *draw) { pipelineManager_->DeviceRestore(vulkan); InitDeviceObjects(); + + // DeviceLost saved the cache and then threw everything away. Load it again, or the next save would + // overwrite the file with only what gets drawn from now on. + if (shaderCachePath_.Valid()) { + LoadCache(shaderCachePath_, false); + } } void GPU_Vulkan::GetStats(StringWriter &w) { diff --git a/GPU/Vulkan/GPU_Vulkan.h b/GPU/Vulkan/GPU_Vulkan.h index 12871e4fdf..264a0b51db 100644 --- a/GPU/Vulkan/GPU_Vulkan.h +++ b/GPU/Vulkan/GPU_Vulkan.h @@ -69,7 +69,7 @@ private: void InitDeviceObjects(); void DestroyDeviceObjects(); - void LoadCache(const Path &filename); + void LoadCache(const Path &filename, bool waitForPipelines); void SaveCache(const Path &filename); FramebufferManagerVulkan *framebufferManagerVulkan_; From e108e4167545ade56e33c2b0dd9de01cc1344982 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 14:52:52 -0600 Subject: [PATCH 6/9] GPU: Fix assorted leaks, null derefs and small rendering bugs - Put the anisotropy level in the sampler key, so changing it applies on Vulkan and D3D11. - Release CLUT textures at shutdown on GLES and D3D11. - Fix the depth readback viewport, which squeezed the image whenever the read rectangle was smaller than the fbo. - Test the computed depth, not the unset result, in the equal-depth clear check. - Don't read back a CLUT from a framebuffer without an fbo. - Tolerate null entries when releasing post-shader objects and CLUT textures after a failed creation. - ImGe: Don't crash on a framebuffer without an fbo, or on GetVFB under the software renderer. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Common/DepthBufferCommon.cpp | 4 +++- GPU/Common/FramebufferManagerCommon.cpp | 3 ++- GPU/Common/PresentationCommon.cpp | 8 ++++++-- GPU/Common/SoftwareTransformCommon.cpp | 2 +- GPU/Common/TextureCacheCommon.cpp | 5 +++++ GPU/Common/TextureCacheCommon.h | 2 ++ GPU/Common/TextureShaderCommon.cpp | 8 ++++++-- GPU/D3D11/TextureCacheD3D11.cpp | 2 +- GPU/Vulkan/TextureCacheVulkan.cpp | 2 +- UI/ImDebugger/ImGe.cpp | 12 +++++++++--- 10 files changed, 36 insertions(+), 12 deletions(-) diff --git a/GPU/Common/DepthBufferCommon.cpp b/GPU/Common/DepthBufferCommon.cpp index a480994e3c..2c2af33184 100644 --- a/GPU/Common/DepthBufferCommon.cpp +++ b/GPU/Common/DepthBufferCommon.cpp @@ -199,7 +199,9 @@ bool FramebufferManagerCommon::ReadbackDepthbuffer(Draw::Framebuffer *fbo, int x auto *blitFBO = GetTempFBO(TempFBO::Z_COPY, fbo->Width() * scaleX, fbo->Height() * scaleY); draw_->BindFramebufferAsRenderTarget(blitFBO, { RPAction::DONT_CARE, RPAction::DONT_CARE, RPAction::DONT_CARE }, "ReadbackDepthbufferSync"); - Draw::Viewport viewport = { 0.0f, 0.0f, (float)destW, (float)destH, 0.0f, 1.0f }; + // The whole fbo is drawn, so the viewport has to cover all of it at the destination scale. Not just + // destW x destH, which would squeeze it whenever the read rectangle is smaller than the fbo. + Draw::Viewport viewport = { 0.0f, 0.0f, fbo->Width() * scaleX, fbo->Height() * scaleY, 0.0f, 1.0f }; draw_->SetViewport(viewport); draw_->SetScissorRect(0, 0, fbo->Width() * scaleX, fbo->Height() * scaleY); diff --git a/GPU/Common/FramebufferManagerCommon.cpp b/GPU/Common/FramebufferManagerCommon.cpp index b6c0d2cbe4..9250151122 100644 --- a/GPU/Common/FramebufferManagerCommon.cpp +++ b/GPU/Common/FramebufferManagerCommon.cpp @@ -3354,7 +3354,8 @@ void FramebufferManagerCommon::FlushBeforeCopy() { // TODO: Replace with with depal, reading the palette from the texture on the GPU directly. void FramebufferManagerCommon::DownloadFramebufferForClut(u32 fb_address, u32 loadBytes) { VirtualFramebuffer *vfb = GetVFBAt(fb_address); - if (vfb && vfb->fb_stride != 0) { + // Without an fbo there's nothing to read back (ReadbackFramebuffer would read the backbuffer instead). + if (vfb && vfb->fb_stride != 0 && vfb->fbo) { const u32 bpp = BufferFormatBytesPerPixel(vfb->fb_format); int x = 0; int y = 0; diff --git a/GPU/Common/PresentationCommon.cpp b/GPU/Common/PresentationCommon.cpp index 2ac066f917..e96a91819b 100644 --- a/GPU/Common/PresentationCommon.cpp +++ b/GPU/Common/PresentationCommon.cpp @@ -555,8 +555,12 @@ static void DoRelease(T *&obj) { template static void DoReleaseVector(std::vector &list) { - for (auto &obj : list) - obj->Release(); + for (auto &obj : list) { + // Can be null when a creation failed partway. + if (obj) { + obj->Release(); + } + } list.clear(); } diff --git a/GPU/Common/SoftwareTransformCommon.cpp b/GPU/Common/SoftwareTransformCommon.cpp index b8f5ff2c2a..319f6bb07a 100644 --- a/GPU/Common/SoftwareTransformCommon.cpp +++ b/GPU/Common/SoftwareTransformCommon.cpp @@ -198,7 +198,7 @@ SoftwareTransformAction RunSoftwareTransform(SoftwareTransformParams ¶ms, in float depth = std::clamp(transformed[1].z, 0.0f, 65535.0f) / 65535.0f; // Non-zero depth clears are unusual, but some drivers don't match drawn depth values to cleared values. // Games sometimes expect exact matches (see #12626, for example) for equal comparisons. - if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && result->depth > 0.0f && result->depth < 1.0f)) { + if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && depth > 0.0f && depth < 1.0f)) { result->color = transformed[1].color0_32; result->depth = depth; gpuStats.perFrame.numClears++; diff --git a/GPU/Common/TextureCacheCommon.cpp b/GPU/Common/TextureCacheCommon.cpp index 30c8d8c247..ded34b7a93 100644 --- a/GPU/Common/TextureCacheCommon.cpp +++ b/GPU/Common/TextureCacheCommon.cpp @@ -114,6 +114,8 @@ TextureCacheCommon::TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D) } TextureCacheCommon::~TextureCacheCommon() { + // Only DeviceLost cleared these, which the GLES and D3D11 backends don't go through at shutdown. + clutTextureCache_.Clear(); FreeAlignedMemory(clutBufConverted_); FreeAlignedMemory(clutBufRaw_); FreeAlignedMemory(expandClut_); @@ -342,6 +344,9 @@ SamplerCacheKey TextureCacheCommon::GetSamplingParams(int maxLevel, const TexCac break; } + if (key.aniso) { + key.anisoLevel = (uint8_t)g_Config.iAnisotropyLevel; + } return key; } diff --git a/GPU/Common/TextureCacheCommon.h b/GPU/Common/TextureCacheCommon.h index 96bf2c1795..22fa4e7517 100644 --- a/GPU/Common/TextureCacheCommon.h +++ b/GPU/Common/TextureCacheCommon.h @@ -86,6 +86,8 @@ struct SamplerCacheKey { bool tClamp : 1; bool aniso : 1; bool texture3d : 1; + // Baked into the sampler objects, so it has to be part of the key, or a change wouldn't apply. + uint8_t anisoLevel; }; }; bool operator < (const SamplerCacheKey &other) const { diff --git a/GPU/Common/TextureShaderCommon.cpp b/GPU/Common/TextureShaderCommon.cpp index dc34f0702c..39362ee65f 100644 --- a/GPU/Common/TextureShaderCommon.cpp +++ b/GPU/Common/TextureShaderCommon.cpp @@ -252,7 +252,9 @@ ClutTexture ClutTextureCache::GetClutTexture(GEPaletteFormat clutFormat, const u void ClutTextureCache::Clear() { for (auto tex = texCache_.begin(); tex != texCache_.end(); ++tex) { - tex->second->texture->Release(); + if (tex->second->texture) { + tex->second->texture->Release(); + } delete tex->second; } texCache_.clear(); @@ -261,7 +263,9 @@ void ClutTextureCache::Clear() { void ClutTextureCache::Decimate() { for (auto tex = texCache_.begin(); tex != texCache_.end(); ) { if (tex->second->lastFrame + DEPAL_TEXTURE_OLD_AGE < gpuStats.totals.numFlips) { - tex->second->texture->Release(); + if (tex->second->texture) { + tex->second->texture->Release(); + } delete tex->second; texCache_.erase(tex++); } else { diff --git a/GPU/D3D11/TextureCacheD3D11.cpp b/GPU/D3D11/TextureCacheD3D11.cpp index 9be6d61b0c..9b65c403e3 100644 --- a/GPU/D3D11/TextureCacheD3D11.cpp +++ b/GPU/D3D11/TextureCacheD3D11.cpp @@ -90,7 +90,7 @@ HRESULT SamplerCacheD3D11::GetOrCreateSampler(ID3D11Device *device, const Sample samp.AddressV = key.tClamp ? D3D11_TEXTURE_ADDRESS_CLAMP : D3D11_TEXTURE_ADDRESS_WRAP; samp.AddressW = samp.AddressU; // Mali benefits from all clamps being the same, and this one is irrelevant. if (key.aniso) { - samp.MaxAnisotropy = (float)(1 << g_Config.iAnisotropyLevel); + samp.MaxAnisotropy = (float)(1 << key.anisoLevel); } else { samp.MaxAnisotropy = 1.0f; } diff --git a/GPU/Vulkan/TextureCacheVulkan.cpp b/GPU/Vulkan/TextureCacheVulkan.cpp index d75e1e8b3d..63c06a35e0 100644 --- a/GPU/Vulkan/TextureCacheVulkan.cpp +++ b/GPU/Vulkan/TextureCacheVulkan.cpp @@ -143,7 +143,7 @@ VkSampler SamplerCache::GetOrCreateSampler(const SamplerCacheKey &key) { if (key.aniso) { // Docs say the min of this value and the supported max are used. - samp.maxAnisotropy = 1 << g_Config.iAnisotropyLevel; + samp.maxAnisotropy = 1 << key.anisoLevel; samp.anisotropyEnable = true; } else { samp.maxAnisotropy = 1.0f; diff --git a/UI/ImDebugger/ImGe.cpp b/UI/ImDebugger/ImGe.cpp index 09bafb1676..855f5ed03b 100644 --- a/UI/ImDebugger/ImGe.cpp +++ b/UI/ImDebugger/ImGe.cpp @@ -96,9 +96,14 @@ void DrawFramebuffersWindow(ImConfig &cfg, FramebufferManagerCommon *framebuffer ImGui::SliderFloat("Scale", &cfg.fbViewerZoom, 0.5f, 16.0f, "%.2f", ImGuiSliderFlags_Logarithmic); // Now, draw the image of the selected framebuffer. + // Null without buffered rendering, or if creating it failed. Draw::Framebuffer *fb = vfbs[cfg.selectedFramebuffer]->fbo; - ImTextureID texId = ImGui_ImplThin3d_AddFBAsTextureTemp(fb, Draw::Aspect::COLOR_BIT, ImGuiPipeline::TexturedOpaque); - ImGui::Image(texId, ImVec2(fb->Width() * cfg.fbViewerZoom, fb->Height() * cfg.fbViewerZoom)); + if (fb) { + ImTextureID texId = ImGui_ImplThin3d_AddFBAsTextureTemp(fb, Draw::Aspect::COLOR_BIT, ImGuiPipeline::TexturedOpaque); + ImGui::Image(texId, ImVec2(fb->Width() * cfg.fbViewerZoom, fb->Height() * cfg.fbViewerZoom)); + } else { + ImGui::TextUnformatted("(no framebuffer object)"); + } } ImGui::End(); @@ -602,7 +607,8 @@ ImGeReadbackViewer::~ImGeReadbackViewer() { } VirtualFramebuffer *ImGeReadbackViewer::GetVFB(FramebufferManagerCommon *fbMan) const { - return fbMan->GetExactVFB(gstate.getFrameBufAddress(), gstate.FrameBufStride(), gstate.FrameBufFormat()); + // fbMan is null with the software renderer. + return fbMan ? fbMan->GetExactVFB(gstate.getFrameBufAddress(), gstate.FrameBufStride(), gstate.FrameBufFormat()) : nullptr; } void ImGeReadbackViewer::DeviceLost() { From b72927bbeb7d8357094395c8bc405cb2fbaea4c3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 15:05:25 -0600 Subject: [PATCH 7/9] Texture replacement: Plug leaks on error paths - Release the file reference when loading a level fails or finds nothing, since only a loaded level takes ownership of it. - Free the PNG image when the size changed since the header was read. - Delete the VFS when a pack without an ini has no hash-named textures. Co-Authored-By: Claude Opus 5.5 (1M context) --- GPU/Common/ReplacedTexture.cpp | 9 +++++++++ GPU/Common/TextureReplacer.cpp | 1 + 2 files changed, 10 insertions(+) diff --git a/GPU/Common/ReplacedTexture.cpp b/GPU/Common/ReplacedTexture.cpp index 4c2cbb0dd4..2c823d86b6 100644 --- a/GPU/Common/ReplacedTexture.cpp +++ b/GPU/Common/ReplacedTexture.cpp @@ -278,6 +278,14 @@ void ReplacedTexture::Prepare(VFSBackend *vfs) { } result = LoadLevelData(fileRef, desc_.filenames[i], i, &pixelFormat); + // A level that got loaded owns the reference. Otherwise (error, or nothing to load) it's ours to free. + bool kept = false; + for (const ReplacedTextureLevel &level : levels_) { + kept = kept || level.fileRef == fileRef; + } + if (!kept) { + vfs_->ReleaseFile(fileRef); + } if (result == LoadLevelResult::DONE) { // Loaded all the levels we're gonna get. fmt = pixelFormat; @@ -730,6 +738,7 @@ ReplacedTexture::LoadLevelResult ReplacedTexture::LoadLevelData(VFSFileReference } if (png.width > (uint32_t)level.w || png.height > (uint32_t)level.h) { ERROR_LOG(Log::TexReplacement, "Texture replacement changed since header read: %s", filename.c_str()); + png_image_free(&png); return LoadLevelResult::LOAD_ERROR; } diff --git a/GPU/Common/TextureReplacer.cpp b/GPU/Common/TextureReplacer.cpp index d8de13bfba..44e3b7009f 100644 --- a/GPU/Common/TextureReplacer.cpp +++ b/GPU/Common/TextureReplacer.cpp @@ -230,6 +230,7 @@ bool TextureReplacer::LoadIni(std::string *error, bool notify) { if (filenameMap.empty()) { WARN_LOG(Log::TexReplacement, "No replacement textures found."); + delete dir; return false; } From 39049a67fd83863a86e362329d9f9b82a2e57c78 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 28 Sep 2026 15:05:26 -0600 Subject: [PATCH 8/9] GLES: Free textures on device lost, and unsubmitted step data at exit - The texture and fragment test caches dropped their GLRTexture objects on DeviceLost without queueing them for deletion, leaking them on every Android background/resume. The deleter already skips the GL calls when the context is gone. - GLRenderManager::ThreadEnd cleared unsubmitted init and render steps without freeing the data they own. Run them through the dry run instead, which now also frees stereo matrices and shader code. Co-Authored-By: Claude Opus 5.5 (1M context) --- Common/GPU/OpenGL/GLQueueRunner.cpp | 4 ++++ Common/GPU/OpenGL/GLRenderManager.cpp | 8 ++++---- GPU/GLES/FragmentTestCacheGLES.cpp | 2 +- GPU/GLES/FragmentTestCacheGLES.h | 3 ++- GPU/GLES/TextureCacheGLES.cpp | 8 ++++---- 5 files changed, 15 insertions(+), 10 deletions(-) diff --git a/Common/GPU/OpenGL/GLQueueRunner.cpp b/Common/GPU/OpenGL/GLQueueRunner.cpp index 4c27693f75..fdf71f6dd6 100644 --- a/Common/GPU/OpenGL/GLQueueRunner.cpp +++ b/Common/GPU/OpenGL/GLQueueRunner.cpp @@ -147,6 +147,7 @@ void GLQueueRunner::RunInitSteps(const FastVec &steps, bool skipGLC case GLRInitStepType::CREATE_SHADER: { WARN_LOG(Log::G3D, "CREATE_SHADER found with skipGLCalls, not good"); + delete[] step.create_shader.code; break; } default: @@ -678,6 +679,9 @@ void GLQueueRunner::RunSteps(const std::vector &steps, GLFrameData &f } } break; + case GLRRenderCommand::UNIFORMSTEREOMATRIX: + delete[] c.uniformStereoMatrix4.mData; + break; default: break; } diff --git a/Common/GPU/OpenGL/GLRenderManager.cpp b/Common/GPU/OpenGL/GLRenderManager.cpp index 50216756b5..4a62896e03 100644 --- a/Common/GPU/OpenGL/GLRenderManager.cpp +++ b/Common/GPU/OpenGL/GLRenderManager.cpp @@ -110,11 +110,11 @@ void GLRenderManager::ThreadEnd() { frameData_[i].deleter_prev.Perform(this, skipGLCalls_); } deleter_.Perform(this, skipGLCalls_); - for (int i = 0; i < (int)steps_.size(); i++) { - delete steps_[i]; - } - steps_.clear(); + // Steps that never got submitted. A dry run frees the data they own (texture uploads etc), and the steps. + queueRunner_.RunInitSteps(initSteps_, true); initSteps_.clear(); + queueRunner_.RunSteps(steps_, frameData_[0], true, false, false); + steps_.clear(); INFO_LOG(Log::G3D, "GLRenderManager::ThreadEnd end"); } diff --git a/GPU/GLES/FragmentTestCacheGLES.cpp b/GPU/GLES/FragmentTestCacheGLES.cpp index 973a3e58a2..066ca0dfae 100644 --- a/GPU/GLES/FragmentTestCacheGLES.cpp +++ b/GPU/GLES/FragmentTestCacheGLES.cpp @@ -144,7 +144,7 @@ GLRTexture *FragmentTestCacheGLES::CreateTestTexture(const GEComparison funcs[4] } void FragmentTestCacheGLES::Clear(bool deleteThem) { - if (deleteThem) { + if (deleteThem && render_) { for (const auto &[_, v] : cache_) { render_->DeleteTexture(v.texture); } diff --git a/GPU/GLES/FragmentTestCacheGLES.h b/GPU/GLES/FragmentTestCacheGLES.h index ddb9eca154..174c1bae6d 100644 --- a/GPU/GLES/FragmentTestCacheGLES.h +++ b/GPU/GLES/FragmentTestCacheGLES.h @@ -68,7 +68,8 @@ public: void BindTestTexture(int slot); void DeviceLost() { - Clear(false); + // Queue the deletes anyway, the deleter frees the GLRTexture objects and skips the GL calls if needed. + Clear(true); render_ = nullptr; } void DeviceRestore(Draw::DrawContext *draw); diff --git a/GPU/GLES/TextureCacheGLES.cpp b/GPU/GLES/TextureCacheGLES.cpp index 8cd71a4dec..e1a61021ca 100644 --- a/GPU/GLES/TextureCacheGLES.cpp +++ b/GPU/GLES/TextureCacheGLES.cpp @@ -51,10 +51,10 @@ void TextureCacheGLES::SetFramebufferManager(FramebufferManagerGLES *fbManager) } void TextureCacheGLES::ReleaseTexture(TexCacheEntry *entry, bool delete_them) { - if (delete_them) { - if (entry->textureName) { - render_->DeleteTexture(entry->textureName); - } + // Delete even when !delete_them (device lost): the GLRTexture is a heap object that only the deleter + // frees, and the deleter skips the GL calls itself once the context is gone. + if (entry->textureName && render_) { + render_->DeleteTexture(entry->textureName); } entry->textureName = nullptr; } From 93f57a4f3829bf1e1b2f394e22794cf06e73433d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Tue, 29 Sep 2026 10:24:26 -0600 Subject: [PATCH 9/9] GPU: Delete copy operations on classes that own resources These own GPU objects, memory or refcounts in their destructors (or assert there that they were torn down), so a copy would double-free. Nothing copies them today; this keeps it that way. The manager base classes cover every backend's subclass. Co-Authored-By: Claude Opus 5.5 (1M context) --- Common/GPU/OpenGL/GLMemory.h | 2 ++ Common/GPU/OpenGL/GLRenderManager.h | 9 +++++++++ Common/GPU/Vulkan/VulkanBarrier.h | 2 ++ Common/GPU/Vulkan/VulkanDescSet.h | 2 ++ Common/GPU/Vulkan/VulkanFramebuffer.h | 2 ++ Common/GPU/Vulkan/VulkanImage.h | 2 ++ Common/GPU/Vulkan/VulkanRenderManager.h | 5 +++++ Common/Render/DrawBuffer.h | 2 ++ Common/Render/TextureAtlas.h | 6 ++++++ GPU/Common/DrawEngineCommon.h | 2 ++ GPU/Common/FramebufferManagerCommon.h | 2 ++ GPU/Common/PresentationCommon.h | 2 ++ GPU/Common/ReplacedTexture.h | 2 ++ GPU/Common/ShaderCommon.h | 2 ++ GPU/Common/TextureCacheCommon.h | 5 +++++ GPU/Common/TextureReplacer.h | 2 ++ GPU/GLES/ShaderManagerGLES.h | 4 ++++ GPU/GPUCommon.h | 2 ++ GPU/Software/BinManager.h | 4 ++++ GPU/Software/TransformUnit.h | 2 ++ GPU/Vulkan/PipelineManagerVulkan.h | 3 +++ GPU/Vulkan/ShaderManagerVulkan.h | 4 ++++ GPU/Vulkan/TextureCacheVulkan.h | 2 ++ 23 files changed, 70 insertions(+) diff --git a/Common/GPU/OpenGL/GLMemory.h b/Common/GPU/OpenGL/GLMemory.h index 3d7780f621..e5ba129a29 100644 --- a/Common/GPU/OpenGL/GLMemory.h +++ b/Common/GPU/OpenGL/GLMemory.h @@ -30,6 +30,8 @@ static inline int operator &(const GLBufferStrategy &lhs, const GLBufferStrategy class GLRBuffer { public: + GLRBuffer(const GLRBuffer &) = delete; + GLRBuffer &operator=(const GLRBuffer &) = delete; GLRBuffer(GLuint target, size_t size) : target_(target), size_((int)size) {} ~GLRBuffer() { if (buffer_) { diff --git a/Common/GPU/OpenGL/GLRenderManager.h b/Common/GPU/OpenGL/GLRenderManager.h index 687307c67d..edfe69b373 100644 --- a/Common/GPU/OpenGL/GLRenderManager.h +++ b/Common/GPU/OpenGL/GLRenderManager.h @@ -30,6 +30,8 @@ constexpr int MAX_GL_TEXTURE_SLOTS = 8; class GLRTexture { public: + GLRTexture(const GLRTexture &) = delete; + GLRTexture &operator=(const GLRTexture &) = delete; GLRTexture(const Draw::DeviceCaps &caps, int width, int height, int depth, int numMips); ~GLRTexture(); @@ -53,6 +55,8 @@ public: class GLRFramebuffer { public: + GLRFramebuffer(const GLRFramebuffer &) = delete; + GLRFramebuffer &operator=(const GLRFramebuffer &) = delete; GLRFramebuffer(const Draw::DeviceCaps &caps, int _width, int _height, bool z_stencil, const char *tag) : color_texture(caps, _width, _height, 1, 1), z_stencil_texture(caps, _width, _height, 1, 1), width(_width), height(_height), z_stencil_(z_stencil) { @@ -83,6 +87,8 @@ private: class GLRShader { public: + GLRShader(const GLRShader &) = delete; + GLRShader &operator=(const GLRShader &) = delete; explicit GLRShader(std::string_view _desc) : desc(_desc) {} ~GLRShader() { if (shader) { @@ -116,6 +122,9 @@ public: class GLRProgram { public: + GLRProgram() = default; + GLRProgram(const GLRProgram &) = delete; + GLRProgram &operator=(const GLRProgram &) = delete; ~GLRProgram() { if (deleteCallback_) { deleteCallback_(deleteParam_); diff --git a/Common/GPU/Vulkan/VulkanBarrier.h b/Common/GPU/Vulkan/VulkanBarrier.h index e41df8f9b1..734fc77f81 100644 --- a/Common/GPU/Vulkan/VulkanBarrier.h +++ b/Common/GPU/Vulkan/VulkanBarrier.h @@ -12,6 +12,8 @@ struct VKRImage; class VulkanBarrierBatch { public: + VulkanBarrierBatch(const VulkanBarrierBatch &) = delete; + VulkanBarrierBatch &operator=(const VulkanBarrierBatch &) = delete; VulkanBarrierBatch() : imageBarriers_(4) {} ~VulkanBarrierBatch(); diff --git a/Common/GPU/Vulkan/VulkanDescSet.h b/Common/GPU/Vulkan/VulkanDescSet.h index 5123ba3460..17833db7c6 100644 --- a/Common/GPU/Vulkan/VulkanDescSet.h +++ b/Common/GPU/Vulkan/VulkanDescSet.h @@ -18,6 +18,8 @@ enum class BindingType { // Only appropriate for use in a per-frame pool. class VulkanDescSetPool { public: + VulkanDescSetPool(const VulkanDescSetPool &) = delete; + VulkanDescSetPool &operator=(const VulkanDescSetPool &) = delete; VulkanDescSetPool(const char *tag, bool grow = true) : tag_(tag), grow_(grow) {} ~VulkanDescSetPool(); diff --git a/Common/GPU/Vulkan/VulkanFramebuffer.h b/Common/GPU/Vulkan/VulkanFramebuffer.h index 0b686b3929..291609c06c 100644 --- a/Common/GPU/Vulkan/VulkanFramebuffer.h +++ b/Common/GPU/Vulkan/VulkanFramebuffer.h @@ -60,6 +60,8 @@ struct VKRImage { class VKRFramebuffer { public: + VKRFramebuffer(const VKRFramebuffer &) = delete; + VKRFramebuffer &operator=(const VKRFramebuffer &) = delete; VKRFramebuffer(VulkanContext *vk, VulkanBarrierBatch *barriers, int _width, int _height, int _numLayers, int _multiSampleLevel, bool createDepthStencilBuffer, const char *tag); ~VKRFramebuffer(); diff --git a/Common/GPU/Vulkan/VulkanImage.h b/Common/GPU/Vulkan/VulkanImage.h index e087a8a1c1..d6a53ec11a 100644 --- a/Common/GPU/Vulkan/VulkanImage.h +++ b/Common/GPU/Vulkan/VulkanImage.h @@ -22,6 +22,8 @@ struct TextureCopyBatch { // ALWAYS use an allocator when calling CreateDirect. class VulkanTexture { public: + VulkanTexture(const VulkanTexture &) = delete; + VulkanTexture &operator=(const VulkanTexture &) = delete; VulkanTexture(VulkanContext *vulkan, const char *tag); ~VulkanTexture() { Destroy(); diff --git a/Common/GPU/Vulkan/VulkanRenderManager.h b/Common/GPU/Vulkan/VulkanRenderManager.h index 8f4b7b0d0f..42bb5f2dd7 100644 --- a/Common/GPU/Vulkan/VulkanRenderManager.h +++ b/Common/GPU/Vulkan/VulkanRenderManager.h @@ -115,6 +115,8 @@ public: // Wrapped pipeline. Does own desc! struct VKRGraphicsPipeline { + VKRGraphicsPipeline(const VKRGraphicsPipeline &) = delete; + VKRGraphicsPipeline &operator=(const VKRGraphicsPipeline &) = delete; VKRGraphicsPipeline(PipelineFlags flags, const char *tag) : flags_(flags), tag_(tag) {} ~VKRGraphicsPipeline(); @@ -195,6 +197,9 @@ static_assert(sizeof(PackedDescriptor::buffer) == 16, "PackedDescriptor should b // Note that we only support a single descriptor set due to compatibility with some ancient devices. // We should probably eventually give that up eventually. struct VKRPipelineLayout { + VKRPipelineLayout() = default; + VKRPipelineLayout(const VKRPipelineLayout &) = delete; + VKRPipelineLayout &operator=(const VKRPipelineLayout &) = delete; ~VKRPipelineLayout(); enum { MAX_DESC_SET_BINDINGS = 5 }; diff --git a/Common/Render/DrawBuffer.h b/Common/Render/DrawBuffer.h index 94e5e56095..f5c9b90e60 100644 --- a/Common/Render/DrawBuffer.h +++ b/Common/Render/DrawBuffer.h @@ -51,6 +51,8 @@ class TextDrawer; class DrawBuffer { public: + DrawBuffer(const DrawBuffer &) = delete; + DrawBuffer &operator=(const DrawBuffer &) = delete; DrawBuffer(); ~DrawBuffer(); diff --git a/Common/Render/TextureAtlas.h b/Common/Render/TextureAtlas.h index 493a73c3ba..15ae9e0fb3 100644 --- a/Common/Render/TextureAtlas.h +++ b/Common/Render/TextureAtlas.h @@ -96,6 +96,9 @@ struct AtlasFontHeader { }; struct AtlasFont { + AtlasFont() = default; + AtlasFont(const AtlasFont &) = delete; + AtlasFont &operator=(const AtlasFont &) = delete; ~AtlasFont(); float padding; @@ -126,6 +129,9 @@ struct AtlasHeader { }; struct Atlas { + Atlas() = default; + Atlas(const Atlas &) = delete; + Atlas &operator=(const Atlas &) = delete; ~Atlas(); bool LoadMeta(const uint8_t *data, size_t data_size); bool IsMetadataLoaded() const { diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index 0333809963..fee97a90b6 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -73,6 +73,8 @@ struct alignas(16) Plane8 { class DrawEngineCommon { public: + DrawEngineCommon(const DrawEngineCommon &) = delete; + DrawEngineCommon &operator=(const DrawEngineCommon &) = delete; DrawEngineCommon(); virtual ~DrawEngineCommon(); diff --git a/GPU/Common/FramebufferManagerCommon.h b/GPU/Common/FramebufferManagerCommon.h index 365d4d050e..bfe5db0d47 100644 --- a/GPU/Common/FramebufferManagerCommon.h +++ b/GPU/Common/FramebufferManagerCommon.h @@ -291,6 +291,8 @@ struct DisplayLayoutConfig; class FramebufferManagerCommon { public: + FramebufferManagerCommon(const FramebufferManagerCommon &) = delete; + FramebufferManagerCommon &operator=(const FramebufferManagerCommon &) = delete; FramebufferManagerCommon(Draw::DrawContext *draw); virtual ~FramebufferManagerCommon(); diff --git a/GPU/Common/PresentationCommon.h b/GPU/Common/PresentationCommon.h index abdab8f032..eb6caf1969 100644 --- a/GPU/Common/PresentationCommon.h +++ b/GPU/Common/PresentationCommon.h @@ -80,6 +80,8 @@ ENUM_CLASS_BITOPS(OutputFlags); class PresentationCommon { public: + PresentationCommon(const PresentationCommon &) = delete; + PresentationCommon &operator=(const PresentationCommon &) = delete; PresentationCommon(Draw::DrawContext *draw); ~PresentationCommon(); diff --git a/GPU/Common/ReplacedTexture.h b/GPU/Common/ReplacedTexture.h index 6ff33d7367..17817f77f8 100644 --- a/GPU/Common/ReplacedTexture.h +++ b/GPU/Common/ReplacedTexture.h @@ -140,6 +140,8 @@ struct ReplacedTextureLevel { class ReplacedTexture { public: + ReplacedTexture(const ReplacedTexture &) = delete; + ReplacedTexture &operator=(const ReplacedTexture &) = delete; ReplacedTexture(VFSBackend *vfs, const ReplacementDesc &desc); ~ReplacedTexture(); diff --git a/GPU/Common/ShaderCommon.h b/GPU/Common/ShaderCommon.h index 6e31d9dfaa..56f449179a 100644 --- a/GPU/Common/ShaderCommon.h +++ b/GPU/Common/ShaderCommon.h @@ -123,6 +123,8 @@ enum : uint64_t { class ShaderManagerCommon { public: + ShaderManagerCommon(const ShaderManagerCommon &) = delete; + ShaderManagerCommon &operator=(const ShaderManagerCommon &) = delete; ShaderManagerCommon(Draw::DrawContext *draw) : draw_(draw) {} virtual ~ShaderManagerCommon() {} diff --git a/GPU/Common/TextureCacheCommon.h b/GPU/Common/TextureCacheCommon.h index 22fa4e7517..30549035c1 100644 --- a/GPU/Common/TextureCacheCommon.h +++ b/GPU/Common/TextureCacheCommon.h @@ -160,6 +160,9 @@ ENUM_CLASS_BITOPS(TexStatus); // TODO: Shrink this struct. There is some fluff. struct TexCacheEntry { + TexCacheEntry() = default; + TexCacheEntry(const TexCacheEntry &) = delete; + TexCacheEntry &operator=(const TexCacheEntry &) = delete; ~TexCacheEntry() { #ifdef _DEBUG if (texturePtr || textureName || vkTex) @@ -338,6 +341,8 @@ struct TextureApplyResult { class TextureCacheCommon { public: + TextureCacheCommon(const TextureCacheCommon &) = delete; + TextureCacheCommon &operator=(const TextureCacheCommon &) = delete; TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D); virtual ~TextureCacheCommon(); diff --git a/GPU/Common/TextureReplacer.h b/GPU/Common/TextureReplacer.h index 833d21cd96..210221562e 100644 --- a/GPU/Common/TextureReplacer.h +++ b/GPU/Common/TextureReplacer.h @@ -76,6 +76,8 @@ enum class ReplacerDecimateMode { class TextureReplacer { public: + TextureReplacer(const TextureReplacer &) = delete; + TextureReplacer &operator=(const TextureReplacer &) = delete; // The draw context is checked for supported texture formats. TextureReplacer(Draw::DrawContext *draw); ~TextureReplacer(); diff --git a/GPU/GLES/ShaderManagerGLES.h b/GPU/GLES/ShaderManagerGLES.h index a0f519de3e..2f79093711 100644 --- a/GPU/GLES/ShaderManagerGLES.h +++ b/GPU/GLES/ShaderManagerGLES.h @@ -37,6 +37,8 @@ class IOFile; class LinkedShader { public: + LinkedShader(const LinkedShader &) = delete; + LinkedShader &operator=(const LinkedShader &) = delete; LinkedShader(GLRenderManager *render, VShaderID VSID, Shader *vs, FShaderID FSID, Shader *fs, bool useHWTransform, bool preloading = false); ~LinkedShader(); @@ -139,6 +141,8 @@ struct ShaderDescGLES { class Shader { public: + Shader(const Shader &) = delete; + Shader &operator=(const Shader &) = delete; Shader(GLRenderManager *render, const char *code, const std::string &desc, const ShaderDescGLES ¶ms); ~Shader(); GLRShader *shader; diff --git a/GPU/GPUCommon.h b/GPU/GPUCommon.h index a295a970c3..5cca5d9433 100644 --- a/GPU/GPUCommon.h +++ b/GPU/GPUCommon.h @@ -56,6 +56,8 @@ inline bool IsTrianglePrim(GEPrimitiveType prim) { struct TransformStats; class GPUCommon { public: + GPUCommon(const GPUCommon &) = delete; + GPUCommon &operator=(const GPUCommon &) = delete; // The constructor might run on the loader thread. GPUCommon(GraphicsContext *gfxCtx, Draw::DrawContext *draw); virtual ~GPUCommon() = default; diff --git a/GPU/Software/BinManager.h b/GPU/Software/BinManager.h index c2b2a7bfdd..8f453251a4 100644 --- a/GPU/Software/BinManager.h +++ b/GPU/Software/BinManager.h @@ -57,6 +57,8 @@ struct BinItem { template struct BinQueue { + BinQueue(const BinQueue &) = delete; + BinQueue &operator=(const BinQueue &) = delete; BinQueue() { Reset(); } @@ -185,6 +187,8 @@ struct BinDirtyRange { class StringWriter; class BinManager { public: + BinManager(const BinManager &) = delete; + BinManager &operator=(const BinManager &) = delete; BinManager(); ~BinManager(); diff --git a/GPU/Software/TransformUnit.h b/GPU/Software/TransformUnit.h index 17fc2053f4..d0e2a5bb78 100644 --- a/GPU/Software/TransformUnit.h +++ b/GPU/Software/TransformUnit.h @@ -111,6 +111,8 @@ class StringWriter; class TransformUnit { public: + TransformUnit(const TransformUnit &) = delete; + TransformUnit &operator=(const TransformUnit &) = delete; TransformUnit(); ~TransformUnit(); diff --git a/GPU/Vulkan/PipelineManagerVulkan.h b/GPU/Vulkan/PipelineManagerVulkan.h index bd05b9b243..a722fd8b29 100644 --- a/GPU/Vulkan/PipelineManagerVulkan.h +++ b/GPU/Vulkan/PipelineManagerVulkan.h @@ -63,6 +63,9 @@ private: // Simply wraps a Vulkan pipeline, providing some metadata. struct VulkanPipeline { + VulkanPipeline() = default; + VulkanPipeline(const VulkanPipeline &) = delete; + VulkanPipeline &operator=(const VulkanPipeline &) = delete; ~VulkanPipeline() { desc->Release(); } diff --git a/GPU/Vulkan/ShaderManagerVulkan.h b/GPU/Vulkan/ShaderManagerVulkan.h index 3ba870b94d..86848b60ad 100644 --- a/GPU/Vulkan/ShaderManagerVulkan.h +++ b/GPU/Vulkan/ShaderManagerVulkan.h @@ -40,6 +40,8 @@ class VulkanPushPool; class VulkanFragmentShader { public: + VulkanFragmentShader(const VulkanFragmentShader &) = delete; + VulkanFragmentShader &operator=(const VulkanFragmentShader &) = delete; VulkanFragmentShader(VulkanContext *vulkan, FShaderID id, FragmentShaderFlags flags, const char *code, SPIRVCache *cache); ~VulkanFragmentShader(); @@ -63,6 +65,8 @@ protected: class VulkanVertexShader { public: + VulkanVertexShader(const VulkanVertexShader &) = delete; + VulkanVertexShader &operator=(const VulkanVertexShader &) = delete; VulkanVertexShader(VulkanContext *vulkan, VShaderID id, VertexShaderFlags flags, const char *code, bool useHWTransform, SPIRVCache *cache); ~VulkanVertexShader(); diff --git a/GPU/Vulkan/TextureCacheVulkan.h b/GPU/Vulkan/TextureCacheVulkan.h index 72b15b585f..0a95a74d53 100644 --- a/GPU/Vulkan/TextureCacheVulkan.h +++ b/GPU/Vulkan/TextureCacheVulkan.h @@ -37,6 +37,8 @@ class StringWriter; class SamplerCache { public: + SamplerCache(const SamplerCache &) = delete; + SamplerCache &operator=(const SamplerCache &) = delete; SamplerCache(VulkanContext *vulkan) : vulkan_(vulkan), cache_(16) {} ~SamplerCache(); VkSampler GetOrCreateSampler(const SamplerCacheKey &key);