Merge pull request #22382 from hrydgard/gpu-lifecycle-fixes-2

Claude code review: GPU fixes 2
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-09-29 11:04:38 -06:00
commit 806618f9a3
54 files changed
+297 -77

No files matched your search

+2
View File
@@ -30,6 +30,8 @@ static inline int operator &(const GLBufferStrategy &lhs, const GLBufferStrategy
class GLRBuffer { class GLRBuffer {
public: public:
GLRBuffer(const GLRBuffer &) = delete;
GLRBuffer &operator=(const GLRBuffer &) = delete;
GLRBuffer(GLuint target, size_t size) : target_(target), size_((int)size) {} GLRBuffer(GLuint target, size_t size) : target_(target), size_((int)size) {}
~GLRBuffer() { ~GLRBuffer() {
if (buffer_) { if (buffer_) {
+4
View File
@@ -147,6 +147,7 @@ void GLQueueRunner::RunInitSteps(const FastVec<GLRInitStep> &steps, bool skipGLC
case GLRInitStepType::CREATE_SHADER: case GLRInitStepType::CREATE_SHADER:
{ {
WARN_LOG(Log::G3D, "CREATE_SHADER found with skipGLCalls, not good"); WARN_LOG(Log::G3D, "CREATE_SHADER found with skipGLCalls, not good");
delete[] step.create_shader.code;
break; break;
} }
default: default:
@@ -678,6 +679,9 @@ void GLQueueRunner::RunSteps(const std::vector<GLRStep *> &steps, GLFrameData &f
} }
} }
break; break;
case GLRRenderCommand::UNIFORMSTEREOMATRIX:
delete[] c.uniformStereoMatrix4.mData;
break;
default: default:
break; break;
} }
+4 -4
View File
@@ -110,11 +110,11 @@ void GLRenderManager::ThreadEnd() {
frameData_[i].deleter_prev.Perform(this, skipGLCalls_); frameData_[i].deleter_prev.Perform(this, skipGLCalls_);
} }
deleter_.Perform(this, skipGLCalls_); deleter_.Perform(this, skipGLCalls_);
for (int i = 0; i < (int)steps_.size(); i++) { // Steps that never got submitted. A dry run frees the data they own (texture uploads etc), and the steps.
delete steps_[i]; queueRunner_.RunInitSteps(initSteps_, true);
}
steps_.clear();
initSteps_.clear(); initSteps_.clear();
queueRunner_.RunSteps(steps_, frameData_[0], true, false, false);
steps_.clear();
INFO_LOG(Log::G3D, "GLRenderManager::ThreadEnd end"); INFO_LOG(Log::G3D, "GLRenderManager::ThreadEnd end");
} }
+9
View File
@@ -30,6 +30,8 @@ constexpr int MAX_GL_TEXTURE_SLOTS = 8;
class GLRTexture { class GLRTexture {
public: public:
GLRTexture(const GLRTexture &) = delete;
GLRTexture &operator=(const GLRTexture &) = delete;
GLRTexture(const Draw::DeviceCaps &caps, int width, int height, int depth, int numMips); GLRTexture(const Draw::DeviceCaps &caps, int width, int height, int depth, int numMips);
~GLRTexture(); ~GLRTexture();
@@ -53,6 +55,8 @@ public:
class GLRFramebuffer { class GLRFramebuffer {
public: public:
GLRFramebuffer(const GLRFramebuffer &) = delete;
GLRFramebuffer &operator=(const GLRFramebuffer &) = delete;
GLRFramebuffer(const Draw::DeviceCaps &caps, int _width, int _height, bool z_stencil, const char *tag) GLRFramebuffer(const Draw::DeviceCaps &caps, int _width, int _height, bool z_stencil, const char *tag)
: color_texture(caps, _width, _height, 1, 1), z_stencil_texture(caps, _width, _height, 1, 1), : color_texture(caps, _width, _height, 1, 1), z_stencil_texture(caps, _width, _height, 1, 1),
width(_width), height(_height), z_stencil_(z_stencil) { width(_width), height(_height), z_stencil_(z_stencil) {
@@ -83,6 +87,8 @@ private:
class GLRShader { class GLRShader {
public: public:
GLRShader(const GLRShader &) = delete;
GLRShader &operator=(const GLRShader &) = delete;
explicit GLRShader(std::string_view _desc) : desc(_desc) {} explicit GLRShader(std::string_view _desc) : desc(_desc) {}
~GLRShader() { ~GLRShader() {
if (shader) { if (shader) {
@@ -116,6 +122,9 @@ public:
class GLRProgram { class GLRProgram {
public: public:
GLRProgram() = default;
GLRProgram(const GLRProgram &) = delete;
GLRProgram &operator=(const GLRProgram &) = delete;
~GLRProgram() { ~GLRProgram() {
if (deleteCallback_) { if (deleteCallback_) {
deleteCallback_(deleteParam_); deleteCallback_(deleteParam_);
+2
View File
@@ -12,6 +12,8 @@ struct VKRImage;
class VulkanBarrierBatch { class VulkanBarrierBatch {
public: public:
VulkanBarrierBatch(const VulkanBarrierBatch &) = delete;
VulkanBarrierBatch &operator=(const VulkanBarrierBatch &) = delete;
VulkanBarrierBatch() : imageBarriers_(4) {} VulkanBarrierBatch() : imageBarriers_(4) {}
~VulkanBarrierBatch(); ~VulkanBarrierBatch();
+2
View File
@@ -18,6 +18,8 @@ enum class BindingType {
// Only appropriate for use in a per-frame pool. // Only appropriate for use in a per-frame pool.
class VulkanDescSetPool { class VulkanDescSetPool {
public: public:
VulkanDescSetPool(const VulkanDescSetPool &) = delete;
VulkanDescSetPool &operator=(const VulkanDescSetPool &) = delete;
VulkanDescSetPool(const char *tag, bool grow = true) : tag_(tag), grow_(grow) {} VulkanDescSetPool(const char *tag, bool grow = true) : tag_(tag), grow_(grow) {}
~VulkanDescSetPool(); ~VulkanDescSetPool();
+2
View File
@@ -60,6 +60,8 @@ struct VKRImage {
class VKRFramebuffer { class VKRFramebuffer {
public: public:
VKRFramebuffer(const VKRFramebuffer &) = delete;
VKRFramebuffer &operator=(const VKRFramebuffer &) = delete;
VKRFramebuffer(VulkanContext *vk, VulkanBarrierBatch *barriers, int _width, int _height, int _numLayers, int _multiSampleLevel, bool createDepthStencilBuffer, const char *tag); VKRFramebuffer(VulkanContext *vk, VulkanBarrierBatch *barriers, int _width, int _height, int _numLayers, int _multiSampleLevel, bool createDepthStencilBuffer, const char *tag);
~VKRFramebuffer(); ~VKRFramebuffer();
+2
View File
@@ -22,6 +22,8 @@ struct TextureCopyBatch {
// ALWAYS use an allocator when calling CreateDirect. // ALWAYS use an allocator when calling CreateDirect.
class VulkanTexture { class VulkanTexture {
public: public:
VulkanTexture(const VulkanTexture &) = delete;
VulkanTexture &operator=(const VulkanTexture &) = delete;
VulkanTexture(VulkanContext *vulkan, const char *tag); VulkanTexture(VulkanContext *vulkan, const char *tag);
~VulkanTexture() { ~VulkanTexture() {
Destroy(); Destroy();
+5
View File
@@ -115,6 +115,8 @@ public:
// Wrapped pipeline. Does own desc! // Wrapped pipeline. Does own desc!
struct VKRGraphicsPipeline { struct VKRGraphicsPipeline {
VKRGraphicsPipeline(const VKRGraphicsPipeline &) = delete;
VKRGraphicsPipeline &operator=(const VKRGraphicsPipeline &) = delete;
VKRGraphicsPipeline(PipelineFlags flags, const char *tag) : flags_(flags), tag_(tag) {} VKRGraphicsPipeline(PipelineFlags flags, const char *tag) : flags_(flags), tag_(tag) {}
~VKRGraphicsPipeline(); ~VKRGraphicsPipeline();
@@ -195,6 +197,9 @@ static_assert(sizeof(PackedDescriptor::buffer) == 16, "PackedDescriptor should b
// Note that we only support a single descriptor set due to compatibility with some ancient devices. // Note that we only support a single descriptor set due to compatibility with some ancient devices.
// We should probably eventually give that up eventually. // We should probably eventually give that up eventually.
struct VKRPipelineLayout { struct VKRPipelineLayout {
VKRPipelineLayout() = default;
VKRPipelineLayout(const VKRPipelineLayout &) = delete;
VKRPipelineLayout &operator=(const VKRPipelineLayout &) = delete;
~VKRPipelineLayout(); ~VKRPipelineLayout();
enum { MAX_DESC_SET_BINDINGS = 5 }; enum { MAX_DESC_SET_BINDINGS = 5 };
+2
View File
@@ -51,6 +51,8 @@ class TextDrawer;
class DrawBuffer { class DrawBuffer {
public: public:
DrawBuffer(const DrawBuffer &) = delete;
DrawBuffer &operator=(const DrawBuffer &) = delete;
DrawBuffer(); DrawBuffer();
~DrawBuffer(); ~DrawBuffer();
+6
View File
@@ -96,6 +96,9 @@ struct AtlasFontHeader {
}; };
struct AtlasFont { struct AtlasFont {
AtlasFont() = default;
AtlasFont(const AtlasFont &) = delete;
AtlasFont &operator=(const AtlasFont &) = delete;
~AtlasFont(); ~AtlasFont();
float padding; float padding;
@@ -126,6 +129,9 @@ struct AtlasHeader {
}; };
struct Atlas { struct Atlas {
Atlas() = default;
Atlas(const Atlas &) = delete;
Atlas &operator=(const Atlas &) = delete;
~Atlas(); ~Atlas();
bool LoadMeta(const uint8_t *data, size_t data_size); bool LoadMeta(const uint8_t *data, size_t data_size);
bool IsMetadataLoaded() const { bool IsMetadataLoaded() const {
+3 -1
View File
@@ -199,7 +199,9 @@ bool FramebufferManagerCommon::ReadbackDepthbuffer(Draw::Framebuffer *fbo, int x
auto *blitFBO = GetTempFBO(TempFBO::Z_COPY, fbo->Width() * scaleX, fbo->Height() * scaleY); auto *blitFBO = GetTempFBO(TempFBO::Z_COPY, fbo->Width() * scaleX, fbo->Height() * scaleY);
draw_->BindFramebufferAsRenderTarget(blitFBO, { RPAction::DONT_CARE, RPAction::DONT_CARE, RPAction::DONT_CARE }, "ReadbackDepthbufferSync"); draw_->BindFramebufferAsRenderTarget(blitFBO, { RPAction::DONT_CARE, RPAction::DONT_CARE, RPAction::DONT_CARE }, "ReadbackDepthbufferSync");
Draw::Viewport viewport = { 0.0f, 0.0f, (float)destW, (float)destH, 0.0f, 1.0f }; // The whole fbo is drawn, so the viewport has to cover all of it at the destination scale. Not just
// destW x destH, which would squeeze it whenever the read rectangle is smaller than the fbo.
Draw::Viewport viewport = { 0.0f, 0.0f, fbo->Width() * scaleX, fbo->Height() * scaleY, 0.0f, 1.0f };
draw_->SetViewport(viewport); draw_->SetViewport(viewport);
draw_->SetScissorRect(0, 0, fbo->Width() * scaleX, fbo->Height() * scaleY); draw_->SetScissorRect(0, 0, fbo->Width() * scaleX, fbo->Height() * scaleY);
+41 -21
View File
@@ -51,11 +51,13 @@ DrawEngineCommon::DrawEngineCommon() : decoderMap_(32) {
transformedExpanded_ = (TransformedVertex *)AllocateMemoryPages(3 * TRANSFORMED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); transformedExpanded_ = (TransformedVertex *)AllocateMemoryPages(3 * TRANSFORMED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
decoded_ = (u8 *)AllocateMemoryPages(DECODED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); decoded_ = (u8 *)AllocateMemoryPages(DECODED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
decIndex_ = (u16 *)AllocateMemoryPages(DECODED_INDEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE); decIndex_ = (u16 *)AllocateMemoryPages(DECODED_INDEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
bboxScratch_ = (u8 *)AllocateMemoryPages(BBOX_SCRATCH_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
_dbg_assert_(transformed_); _dbg_assert_(transformed_);
_dbg_assert_(transformedExpanded_); _dbg_assert_(transformedExpanded_);
_dbg_assert_(decoded_); _dbg_assert_(decoded_);
_dbg_assert_(decIndex_); _dbg_assert_(decIndex_);
_dbg_assert_(bboxScratch_);
indexGen.Setup(decIndex_); indexGen.Setup(decIndex_);
@@ -65,6 +67,7 @@ DrawEngineCommon::DrawEngineCommon() : decoderMap_(32) {
DrawEngineCommon::~DrawEngineCommon() { DrawEngineCommon::~DrawEngineCommon() {
FreeMemoryPages(decoded_, DECODED_VERTEX_BUFFER_SIZE); FreeMemoryPages(decoded_, DECODED_VERTEX_BUFFER_SIZE);
FreeMemoryPages(decIndex_, DECODED_INDEX_BUFFER_SIZE); FreeMemoryPages(decIndex_, DECODED_INDEX_BUFFER_SIZE);
FreeMemoryPages(bboxScratch_, BBOX_SCRATCH_SIZE);
FreeMemoryPages(transformed_, TRANSFORMED_VERTEX_BUFFER_SIZE); FreeMemoryPages(transformed_, TRANSFORMED_VERTEX_BUFFER_SIZE);
FreeMemoryPages(transformedExpanded_, 3 * TRANSFORMED_VERTEX_BUFFER_SIZE); FreeMemoryPages(transformedExpanded_, 3 * TRANSFORMED_VERTEX_BUFFER_SIZE);
ShutdownDepthRaster(); ShutdownDepthRaster();
@@ -183,15 +186,14 @@ void DrawEngineCommon::DispatchSubmitImm(GEPrimitiveType prim, TransformedVertex
// - Less accurate, but.. // - Less accurate, but..
// - Only requires six plane evaluations then. // - Only requires six plane evaluations then.
bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, const VertexDecoder *dec, u32 vertType) { bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, const VertexDecoder *dec, u32 vertType) {
// Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws. // The scratch buffer is sized for 1024 vertices. Although this may lead to drawing that shouldn't happen,
// Although this may lead to drawing that shouldn't happen, the viewport is more complex on VR. // the viewport is more complex on VR. Let's always say objects are within bounds.
// Let's always say objects are within bounds.
if (vertexCount > 1024 || gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) { if (vertexCount > 1024 || gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) {
return true; return true;
} }
SimpleVertex *corners = (SimpleVertex *)(decoded_ + 65536 * 12); SimpleVertex *corners = (SimpleVertex *)(bboxScratch_ + BBOX_SCRATCH_CORNERS_OFFSET);
float *verts = (float *)(decoded_ + 65536 * 18); float *verts = (float *)(bboxScratch_ + BBOX_SCRATCH_VERTS_OFFSET);
// Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder // Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder
// and a large vertex format. // and a large vertex format.
@@ -218,7 +220,7 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
} }
} else { } else {
// Simplify away indices, bones, and morph before proceeding. // Simplify away indices, bones, and morph before proceeding.
u8 *temp_buffer = decoded_ + 65536 * 24; u8 *temp_buffer = bboxScratch_ + BBOX_SCRATCH_TEMP_OFFSET;
if ((inds || (vertType & (GE_VTYPE_WEIGHT_MASK | GE_VTYPE_MORPHCOUNT_MASK)))) { if ((inds || (vertType & (GE_VTYPE_WEIGHT_MASK | GE_VTYPE_MORPHCOUNT_MASK)))) {
// Need for Speed Carbon ends up on this path! With a single bone weight. // Need for Speed Carbon ends up on this path! With a single bone weight.
@@ -392,7 +394,8 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, cons
} }
case GE_VTYPE_IDX_32BIT: case GE_VTYPE_IDX_32BIT:
{ {
u32 idx = ((u32 *)idata)[i]; // The PSP ignores the upper 16 bits.
u16 idx = (u16)((u32 *)idata)[i];
data = (const s8 *)srcdata + idx * stride; data = (const s8 *)srcdata + idx * stride;
break; break;
} }
@@ -759,7 +762,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *
if (IsTrianglePrim(newPrim) != isTriangle) if (IsTrianglePrim(newPrim) != isTriangle)
break; break;
int vertexCount = data & 0xFFFF; int vertexCount = data & 0xFFFF;
if (numDrawInds >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + offset + vertexCount > VERTEX_BUFFER_MAX) { if (numDrawInds >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + offset + vertexCount > VERTEX_BUFFER_MAX || numVertsToDecode_ + (offset - dv.vertexCount) + vertexCount > VERTEX_BUFFER_MAX) {
break; break;
} }
DeferredInds &di = drawInds_[numDrawInds++]; DeferredInds &di = drawInds_[numDrawInds++];
@@ -780,6 +783,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *
dv.vertexCount = offset; dv.vertexCount = offset;
dv.indexUpperBound = dv.vertexCount - 1; dv.indexUpperBound = dv.vertexCount - 1;
vertexCountInDrawCalls_ += totalCount; vertexCountInDrawCalls_ += totalCount;
numVertsToDecode_ += totalCount;
*bytesRead = totalCount * dec->VertexSize(); *bytesRead = totalCount * dec->VertexSize();
return cmd - start; return cmd - start;
} }
@@ -805,7 +809,21 @@ void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, const Ver
// vertTypeID is the vertex type but with the UVGen mode smashed into the top bits. // vertTypeID is the vertex type but with the UVGen mode smashed into the top bits.
bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, const VertexDecoder *dec, u32 vertTypeID, bool clockwise, int *bytesRead, ClipInfoFlags clipInfoFlags) { bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, const VertexDecoder *dec, u32 vertTypeID, bool clockwise, int *bytesRead, ClipInfoFlags clipInfoFlags) {
if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawVerts_ >= MAX_DEFERRED_DRAW_VERTS || numDrawInds_ >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + vertexCount > VERTEX_BUFFER_MAX) { // The index count doesn't bound how many vertices DecodeVerts will produce (the index range can be
// sparse), so track that separately, as the growth of the range to decode.
u16 lowerBound = 0;
u16 upperBound = 0;
int decodeGrowth = 0;
if (vertexCount > 0) {
GetIndexBounds(inds, vertexCount, vertTypeID, &lowerBound, &upperBound);
decodeGrowth = upperBound - lowerBound + 1;
if (CanExtendDecode(verts, inds, dec)) {
const DeferredVerts &last = drawVerts_[numDrawVerts_ - 1];
decodeGrowth = std::max(upperBound, last.indexUpperBound) - std::min(lowerBound, last.indexLowerBound) - (last.indexUpperBound - last.indexLowerBound);
}
}
if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawVerts_ >= MAX_DEFERRED_DRAW_VERTS || numDrawInds_ >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + vertexCount > VERTEX_BUFFER_MAX || numVertsToDecode_ + decodeGrowth > VERTEX_BUFFER_MAX) {
Flush(); Flush();
} }
@@ -855,6 +873,7 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti
const int rem = vertexCount % 3; const int rem = vertexCount % 3;
if (rem != 0) { if (rem != 0) {
vertexCount -= rem; vertexCount -= rem;
GetIndexBounds(inds, vertexCount, vertTypeID, &lowerBound, &upperBound);
} }
} }
@@ -881,17 +900,16 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti
_dbg_assert_(numDrawVerts <= MAX_DEFERRED_DRAW_VERTS); _dbg_assert_(numDrawVerts <= MAX_DEFERRED_DRAW_VERTS);
if (inds && numDrawVerts > decodeVertsCounter_ && drawVerts_[numDrawVerts - 1].verts == verts && !applySkin) { if (CanExtendDecode(verts, inds, dec_)) {
// Same vertex pointer as a previous un-decoded draw call - let's just extend the decode! // Same vertex pointer as a previous un-decoded draw call - let's just extend the decode!
di.vertDecodeIndex = numDrawVerts - 1; di.vertDecodeIndex = numDrawVerts - 1;
u16 lb;
u16 ub;
GetIndexBounds(inds, vertexCount, vertTypeID, &lb, &ub);
DeferredVerts &dv = drawVerts_[numDrawVerts - 1]; DeferredVerts &dv = drawVerts_[numDrawVerts - 1];
if (lb < dv.indexLowerBound) const int oldCount = dv.indexUpperBound - dv.indexLowerBound + 1;
dv.indexLowerBound = lb; if (lowerBound < dv.indexLowerBound)
if (ub > dv.indexUpperBound) dv.indexLowerBound = lowerBound;
dv.indexUpperBound = ub; if (upperBound > dv.indexUpperBound)
dv.indexUpperBound = upperBound;
numVertsToDecode_ += dv.indexUpperBound - dv.indexLowerBound + 1 - oldCount;
} else { } else {
// Record a new draw, and a new index gen. // Record a new draw, and a new index gen.
DeferredVerts &dv = drawVerts_[numDrawVerts]; DeferredVerts &dv = drawVerts_[numDrawVerts];
@@ -899,8 +917,9 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti
dv.verts = verts; dv.verts = verts;
dv.vertexCount = vertexCount; dv.vertexCount = vertexCount;
dv.uvScale = LoadUVScaleOffset(gstate); dv.uvScale = LoadUVScaleOffset(gstate);
// Does handle the unindexed case. dv.indexLowerBound = lowerBound;
GetIndexBounds(inds, vertexCount, vertTypeID, &dv.indexLowerBound, &dv.indexUpperBound); dv.indexUpperBound = upperBound;
numVertsToDecode_ += upperBound - lowerBound + 1;
} }
vertexCountInDrawCalls_ += vertexCount; vertexCountInDrawCalls_ += vertexCount;
@@ -930,8 +949,9 @@ void DrawEngineCommon::DecodeVerts(const VertexDecoder *dec, u8 *dest) {
drawVertexOffsets_[i] = numDecodedVerts - indexLowerBound; drawVertexOffsets_[i] = numDecodedVerts - indexLowerBound;
const int indexUpperBound = dv.indexUpperBound; const int indexUpperBound = dv.indexUpperBound;
const int count = indexUpperBound - indexLowerBound + 1; const int count = indexUpperBound - indexLowerBound + 1;
if (count + numDecodedVerts >= VERTEX_BUFFER_MAX) { if (count + numDecodedVerts > VERTEX_BUFFER_MAX) {
// Hit our limit! Stop decoding in this draw. // SubmitPrim flushes before this can happen.
_dbg_assert_(false);
break; break;
} }
+17
View File
@@ -38,6 +38,11 @@ enum {
VERTEX_BUFFER_MAX = 65536, VERTEX_BUFFER_MAX = 65536,
DECODED_VERTEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 2 * 36, // 36 == sizeof(SimpleVertex) DECODED_VERTEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 2 * 36, // 36 == sizeof(SimpleVertex)
DECODED_INDEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 6 * 6 * 2, // * 6 for spline tessellation, then * 6 again for converting into points/lines, and * 2 for 2 bytes per index DECODED_INDEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 6 * 6 * 2, // * 6 for spline tessellation, then * 6 again for converting into points/lines, and * 2 for 2 bytes per index
// TestBoundingBox handles up to 1025 vertices: corners (SimpleVertex), then positions, then decoded vertices.
BBOX_SCRATCH_CORNERS_OFFSET = 0,
BBOX_SCRATCH_VERTS_OFFSET = 64 * 1024,
BBOX_SCRATCH_TEMP_OFFSET = 128 * 1024,
BBOX_SCRATCH_SIZE = 256 * 1024,
}; };
enum { enum {
@@ -68,6 +73,8 @@ struct alignas(16) Plane8 {
class DrawEngineCommon { class DrawEngineCommon {
public: public:
DrawEngineCommon(const DrawEngineCommon &) = delete;
DrawEngineCommon &operator=(const DrawEngineCommon &) = delete;
DrawEngineCommon(); DrawEngineCommon();
virtual ~DrawEngineCommon(); virtual ~DrawEngineCommon();
@@ -161,6 +168,11 @@ protected:
void DecodeVerts(const VertexDecoder *dec, u8 *dest); void DecodeVerts(const VertexDecoder *dec, u8 *dest);
int DecodeInds(); int DecodeInds();
// Whether an indexed draw can share the previous draw's vertex decode, by widening its index range.
bool CanExtendDecode(const void *verts, const void *inds, const VertexDecoder *dec) const {
return inds && numDrawVerts_ > decodeVertsCounter_ && drawVerts_[numDrawVerts_ - 1].verts == verts && !dec->skinInDecode;
}
int ComputeNumVertsToDecode() const; int ComputeNumVertsToDecode() const;
void ApplyFramebufferRead(FBOTexState *fboTexState); void ApplyFramebufferRead(FBOTexState *fboTexState);
@@ -211,6 +223,7 @@ protected:
numDrawVerts_ = 0; numDrawVerts_ = 0;
numDrawInds_ = 0; numDrawInds_ = 0;
vertexCountInDrawCalls_ = 0; vertexCountInDrawCalls_ = 0;
numVertsToDecode_ = 0;
decodeIndsCounter_ = 0; decodeIndsCounter_ = 0;
decodeVertsCounter_ = 0; decodeVertsCounter_ = 0;
seenPrims_ = 0; seenPrims_ = 0;
@@ -273,6 +286,8 @@ protected:
// Vertex collector buffers // Vertex collector buffers
u8 *decoded_ = nullptr; u8 *decoded_ = nullptr;
u16 *decIndex_ = nullptr; u16 *decIndex_ = nullptr;
// Separate from decoded_, which can hold decoded vertices that haven't been flushed yet.
u8 *bboxScratch_ = nullptr;
// Cached vertex decoders // Cached vertex decoders
DenseHashMap<u32, VertexDecoder *> decoderMap_; DenseHashMap<u32, VertexDecoder *> decoderMap_;
@@ -312,6 +327,8 @@ protected:
int numDrawVerts_ = 0; int numDrawVerts_ = 0;
int numDrawInds_ = 0; int numDrawInds_ = 0;
int vertexCountInDrawCalls_ = 0; int vertexCountInDrawCalls_ = 0;
// How many vertices DecodeVerts will produce for the queued draws. Must stay <= VERTEX_BUFFER_MAX.
int numVertsToDecode_ = 0;
int decodeVertsCounter_ = 0; int decodeVertsCounter_ = 0;
int decodeIndsCounter_ = 0; int decodeIndsCounter_ = 0;
+2 -1
View File
@@ -3354,7 +3354,8 @@ void FramebufferManagerCommon::FlushBeforeCopy() {
// TODO: Replace with with depal, reading the palette from the texture on the GPU directly. // TODO: Replace with with depal, reading the palette from the texture on the GPU directly.
void FramebufferManagerCommon::DownloadFramebufferForClut(u32 fb_address, u32 loadBytes) { void FramebufferManagerCommon::DownloadFramebufferForClut(u32 fb_address, u32 loadBytes) {
VirtualFramebuffer *vfb = GetVFBAt(fb_address); VirtualFramebuffer *vfb = GetVFBAt(fb_address);
if (vfb && vfb->fb_stride != 0) { // Without an fbo there's nothing to read back (ReadbackFramebuffer would read the backbuffer instead).
if (vfb && vfb->fb_stride != 0 && vfb->fbo) {
const u32 bpp = BufferFormatBytesPerPixel(vfb->fb_format); const u32 bpp = BufferFormatBytesPerPixel(vfb->fb_format);
int x = 0; int x = 0;
int y = 0; int y = 0;
+2
View File
@@ -291,6 +291,8 @@ struct DisplayLayoutConfig;
class FramebufferManagerCommon { class FramebufferManagerCommon {
public: public:
FramebufferManagerCommon(const FramebufferManagerCommon &) = delete;
FramebufferManagerCommon &operator=(const FramebufferManagerCommon &) = delete;
FramebufferManagerCommon(Draw::DrawContext *draw); FramebufferManagerCommon(Draw::DrawContext *draw);
virtual ~FramebufferManagerCommon(); virtual ~FramebufferManagerCommon();
+1
View File
@@ -342,6 +342,7 @@ void IndexGenerator::TranslatePrim(int prim, int numInds, const u16_le *inds, in
} }
} }
// The PSP ignores the upper 16 bits of 32-bit indices. The u16 output drops them the same way.
void IndexGenerator::TranslatePrim(int prim, int numInds, const u32_le *inds, int indexOffset, bool clockwise) { void IndexGenerator::TranslatePrim(int prim, int numInds, const u32_le *inds, int indexOffset, bool clockwise) {
switch (prim) { switch (prim) {
case GE_PRIM_POINTS: TranslatePoints<u32_le>(numInds, inds, indexOffset); break; case GE_PRIM_POINTS: TranslatePoints<u32_le>(numInds, inds, indexOffset); break;
+6 -2
View File
@@ -555,8 +555,12 @@ static void DoRelease(T *&obj) {
template <typename T> template <typename T>
static void DoReleaseVector(std::vector<T *> &list) { static void DoReleaseVector(std::vector<T *> &list) {
for (auto &obj : list) for (auto &obj : list) {
obj->Release(); // Can be null when a creation failed partway.
if (obj) {
obj->Release();
}
}
list.clear(); list.clear();
} }
+2
View File
@@ -80,6 +80,8 @@ ENUM_CLASS_BITOPS(OutputFlags);
class PresentationCommon { class PresentationCommon {
public: public:
PresentationCommon(const PresentationCommon &) = delete;
PresentationCommon &operator=(const PresentationCommon &) = delete;
PresentationCommon(Draw::DrawContext *draw); PresentationCommon(Draw::DrawContext *draw);
~PresentationCommon(); ~PresentationCommon();
+9
View File
@@ -278,6 +278,14 @@ void ReplacedTexture::Prepare(VFSBackend *vfs) {
} }
result = LoadLevelData(fileRef, desc_.filenames[i], i, &pixelFormat); result = LoadLevelData(fileRef, desc_.filenames[i], i, &pixelFormat);
// A level that got loaded owns the reference. Otherwise (error, or nothing to load) it's ours to free.
bool kept = false;
for (const ReplacedTextureLevel &level : levels_) {
kept = kept || level.fileRef == fileRef;
}
if (!kept) {
vfs_->ReleaseFile(fileRef);
}
if (result == LoadLevelResult::DONE) { if (result == LoadLevelResult::DONE) {
// Loaded all the levels we're gonna get. // Loaded all the levels we're gonna get.
fmt = pixelFormat; fmt = pixelFormat;
@@ -730,6 +738,7 @@ ReplacedTexture::LoadLevelResult ReplacedTexture::LoadLevelData(VFSFileReference
} }
if (png.width > (uint32_t)level.w || png.height > (uint32_t)level.h) { if (png.width > (uint32_t)level.w || png.height > (uint32_t)level.h) {
ERROR_LOG(Log::TexReplacement, "Texture replacement changed since header read: %s", filename.c_str()); ERROR_LOG(Log::TexReplacement, "Texture replacement changed since header read: %s", filename.c_str());
png_image_free(&png);
return LoadLevelResult::LOAD_ERROR; return LoadLevelResult::LOAD_ERROR;
} }
+2
View File
@@ -140,6 +140,8 @@ struct ReplacedTextureLevel {
class ReplacedTexture { class ReplacedTexture {
public: public:
ReplacedTexture(const ReplacedTexture &) = delete;
ReplacedTexture &operator=(const ReplacedTexture &) = delete;
ReplacedTexture(VFSBackend *vfs, const ReplacementDesc &desc); ReplacedTexture(VFSBackend *vfs, const ReplacementDesc &desc);
~ReplacedTexture(); ~ReplacedTexture();
+2
View File
@@ -123,6 +123,8 @@ enum : uint64_t {
class ShaderManagerCommon { class ShaderManagerCommon {
public: public:
ShaderManagerCommon(const ShaderManagerCommon &) = delete;
ShaderManagerCommon &operator=(const ShaderManagerCommon &) = delete;
ShaderManagerCommon(Draw::DrawContext *draw) : draw_(draw) {} ShaderManagerCommon(Draw::DrawContext *draw) : draw_(draw) {}
virtual ~ShaderManagerCommon() {} virtual ~ShaderManagerCommon() {}
+1 -1
View File
@@ -198,7 +198,7 @@ SoftwareTransformAction RunSoftwareTransform(SoftwareTransformParams &params, in
float depth = std::clamp(transformed[1].z, 0.0f, 65535.0f) / 65535.0f; float depth = std::clamp(transformed[1].z, 0.0f, 65535.0f) / 65535.0f;
// Non-zero depth clears are unusual, but some drivers don't match drawn depth values to cleared values. // Non-zero depth clears are unusual, but some drivers don't match drawn depth values to cleared values.
// Games sometimes expect exact matches (see #12626, for example) for equal comparisons. // Games sometimes expect exact matches (see #12626, for example) for equal comparisons.
if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && result->depth > 0.0f && result->depth < 1.0f)) { if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && depth > 0.0f && depth < 1.0f)) {
result->color = transformed[1].color0_32; result->color = transformed[1].color0_32;
result->depth = depth; result->depth = depth;
gpuStats.perFrame.numClears++; gpuStats.perFrame.numClears++;
+5
View File
@@ -114,6 +114,8 @@ TextureCacheCommon::TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D)
} }
TextureCacheCommon::~TextureCacheCommon() { TextureCacheCommon::~TextureCacheCommon() {
// Only DeviceLost cleared these, which the GLES and D3D11 backends don't go through at shutdown.
clutTextureCache_.Clear();
FreeAlignedMemory(clutBufConverted_); FreeAlignedMemory(clutBufConverted_);
FreeAlignedMemory(clutBufRaw_); FreeAlignedMemory(clutBufRaw_);
FreeAlignedMemory(expandClut_); FreeAlignedMemory(expandClut_);
@@ -342,6 +344,9 @@ SamplerCacheKey TextureCacheCommon::GetSamplingParams(int maxLevel, const TexCac
break; break;
} }
if (key.aniso) {
key.anisoLevel = (uint8_t)g_Config.iAnisotropyLevel;
}
return key; return key;
} }
+7
View File
@@ -86,6 +86,8 @@ struct SamplerCacheKey {
bool tClamp : 1; bool tClamp : 1;
bool aniso : 1; bool aniso : 1;
bool texture3d : 1; bool texture3d : 1;
// Baked into the sampler objects, so it has to be part of the key, or a change wouldn't apply.
uint8_t anisoLevel;
}; };
}; };
bool operator < (const SamplerCacheKey &other) const { bool operator < (const SamplerCacheKey &other) const {
@@ -158,6 +160,9 @@ ENUM_CLASS_BITOPS(TexStatus);
// TODO: Shrink this struct. There is some fluff. // TODO: Shrink this struct. There is some fluff.
struct TexCacheEntry { struct TexCacheEntry {
TexCacheEntry() = default;
TexCacheEntry(const TexCacheEntry &) = delete;
TexCacheEntry &operator=(const TexCacheEntry &) = delete;
~TexCacheEntry() { ~TexCacheEntry() {
#ifdef _DEBUG #ifdef _DEBUG
if (texturePtr || textureName || vkTex) if (texturePtr || textureName || vkTex)
@@ -336,6 +341,8 @@ struct TextureApplyResult {
class TextureCacheCommon { class TextureCacheCommon {
public: public:
TextureCacheCommon(const TextureCacheCommon &) = delete;
TextureCacheCommon &operator=(const TextureCacheCommon &) = delete;
TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D); TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D);
virtual ~TextureCacheCommon(); virtual ~TextureCacheCommon();
+1
View File
@@ -230,6 +230,7 @@ bool TextureReplacer::LoadIni(std::string *error, bool notify) {
if (filenameMap.empty()) { if (filenameMap.empty()) {
WARN_LOG(Log::TexReplacement, "No replacement textures found."); WARN_LOG(Log::TexReplacement, "No replacement textures found.");
delete dir;
return false; return false;
} }
+2
View File
@@ -76,6 +76,8 @@ enum class ReplacerDecimateMode {
class TextureReplacer { class TextureReplacer {
public: public:
TextureReplacer(const TextureReplacer &) = delete;
TextureReplacer &operator=(const TextureReplacer &) = delete;
// The draw context is checked for supported texture formats. // The draw context is checked for supported texture formats.
TextureReplacer(Draw::DrawContext *draw); TextureReplacer(Draw::DrawContext *draw);
~TextureReplacer(); ~TextureReplacer();
+6 -2
View File
@@ -252,7 +252,9 @@ ClutTexture ClutTextureCache::GetClutTexture(GEPaletteFormat clutFormat, const u
void ClutTextureCache::Clear() { void ClutTextureCache::Clear() {
for (auto tex = texCache_.begin(); tex != texCache_.end(); ++tex) { for (auto tex = texCache_.begin(); tex != texCache_.end(); ++tex) {
tex->second->texture->Release(); if (tex->second->texture) {
tex->second->texture->Release();
}
delete tex->second; delete tex->second;
} }
texCache_.clear(); texCache_.clear();
@@ -261,7 +263,9 @@ void ClutTextureCache::Clear() {
void ClutTextureCache::Decimate() { void ClutTextureCache::Decimate() {
for (auto tex = texCache_.begin(); tex != texCache_.end(); ) { for (auto tex = texCache_.begin(); tex != texCache_.end(); ) {
if (tex->second->lastFrame + DEPAL_TEXTURE_OLD_AGE < gpuStats.totals.numFlips) { if (tex->second->lastFrame + DEPAL_TEXTURE_OLD_AGE < gpuStats.totals.numFlips) {
tex->second->texture->Release(); if (tex->second->texture) {
tex->second->texture->Release();
}
delete tex->second; delete tex->second;
texCache_.erase(tex++); texCache_.erase(tex++);
} else { } else {
+8 -1
View File
@@ -170,8 +170,9 @@ void GetIndexBounds(const void *inds, int count, u32 vertType, u16 *indexLowerBo
bool oob = false; bool oob = false;
const u32_le *ind32 = (const u32_le *)inds; const u32_le *ind32 = (const u32_le *)inds;
for (int i = 0; i < count; i++) { for (int i = 0; i < count; i++) {
// The PSP ignores the upper 16 bits, so only the low ones count.
const u16 value = (u16)ind32[i]; const u16 value = (u16)ind32[i];
// These aren't documented and should be rare. Let's bounds check each one. // Games setting the upper bits should be rare, so report them.
if (ind32[i] != value) { if (ind32[i] != value) {
oob = true; oob = true;
} }
@@ -1373,6 +1374,12 @@ void VertexDecoder::SetVertexType(u32 fmt, const VertexDecoderOptions &options,
// Attempt to JIT as well. But only do that if the main CPU JIT is enabled, in order to aid // Attempt to JIT as well. But only do that if the main CPU JIT is enabled, in order to aid
// debugging attempts - if the main JIT doesn't work, this one won't do any better, probably. // debugging attempts - if the main JIT doesn't work, this one won't do any better, probably.
if (jitCache) { if (jitCache) {
// Compile doesn't check for space. We can't clear the cache here since other decoders point into it,
// so when it's full (only seen with garbage display lists), new decoders use the interpreter.
if (jitCache->GetSpaceLeft() < 4096) {
WARN_LOG(Log::G3D, "Vertex decoder JIT cache full, using the interpreter for %08x", fmt_);
return;
}
jitted_ = jitCache->Compile(*this, &jittedSize_); jitted_ = jitCache->Compile(*this, &jittedSize_);
if (!jitted_) { if (!jitted_) {
WARN_LOG(Log::G3D, "Vertex decoder JIT failed! fmt = %08x (%s)", fmt_, GetString(SHADER_STRING_SHORT_DESC).c_str()); WARN_LOG(Log::G3D, "Vertex decoder JIT failed! fmt = %08x (%s)", fmt_, GetString(SHADER_STRING_SHORT_DESC).c_str());
+2 -1
View File
@@ -115,7 +115,8 @@ public:
case GE_VTYPE_IDX_16BIT: case GE_VTYPE_IDX_16BIT:
return indices16[index]; return indices16[index];
case GE_VTYPE_IDX_32BIT: case GE_VTYPE_IDX_32BIT:
return indices32[index]; // The PSP supports 32-bit indices in name only: it ignores the upper 16 bits.
return indices32[index] & 0xFFFF;
default: default:
return index; return index;
} }
+3 -1
View File
@@ -63,7 +63,9 @@ public:
void Flush() override; void Flush() override;
void FinishDeferred() { void FinishDeferred() {
DecodeVerts(dec_, decoded_); // Decoding only the vertices isn't enough: the indices are still read from PSP memory at flush
// time, and the game may change them once it regains control (#10095).
Flush();
} }
void NotifyConfigChanged() override; void NotifyConfigChanged() override;
+1 -1
View File
@@ -90,7 +90,7 @@ HRESULT SamplerCacheD3D11::GetOrCreateSampler(ID3D11Device *device, const Sample
samp.AddressV = key.tClamp ? D3D11_TEXTURE_ADDRESS_CLAMP : D3D11_TEXTURE_ADDRESS_WRAP; samp.AddressV = key.tClamp ? D3D11_TEXTURE_ADDRESS_CLAMP : D3D11_TEXTURE_ADDRESS_WRAP;
samp.AddressW = samp.AddressU; // Mali benefits from all clamps being the same, and this one is irrelevant. samp.AddressW = samp.AddressU; // Mali benefits from all clamps being the same, and this one is irrelevant.
if (key.aniso) { if (key.aniso) {
samp.MaxAnisotropy = (float)(1 << g_Config.iAnisotropyLevel); samp.MaxAnisotropy = (float)(1 << key.anisoLevel);
} else { } else {
samp.MaxAnisotropy = 1.0f; samp.MaxAnisotropy = 1.0f;
} }
+31 -13
View File
@@ -126,13 +126,16 @@ void Recorder::DirtyDrawnVRAM() {
} }
bool Recorder::BeginRecording() { bool Recorder::BeginRecording() {
std::unique_lock<std::mutex> guard(callbackLock_);
nextFrame = false;
if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) { if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) {
// Can't record a GE dump. // Can't record a GE dump. RecordNextFrame refuses this too.
writeCallback = nullptr;
return false; return false;
} }
active = true; active = true;
nextFrame = false; guard.unlock();
lastTextures.clear(); lastTextures.clear();
lastRenderTargets.clear(); lastRenderTargets.clear();
flipLastAction = gpuStats.totals.numFlips; flipLastAction = gpuStats.totals.numFlips;
@@ -180,6 +183,10 @@ Path Recorder::WriteRecording() {
NOTICE_LOG(Log::G3D, "Recording filename: %s", filename.c_str()); NOTICE_LOG(Log::G3D, "Recording filename: %s", filename.c_str());
FILE *fp = File::OpenCFile(filename, "wb"); FILE *fp = File::OpenCFile(filename, "wb");
if (!fp) {
ERROR_LOG(Log::G3D, "Failed to open '%s' for writing the recording", filename.c_str());
return Path();
}
Header header{}; Header header{};
memcpy(header.magic, HEADER_MAGIC, sizeof(header.magic)); memcpy(header.magic, HEADER_MAGIC, sizeof(header.magic));
header.version = VERSION; header.version = VERSION;
@@ -574,14 +581,19 @@ void Recorder::EmitBezierSpline(u32 op) {
} }
bool Recorder::RecordNextFrame(const std::function<void(const Path &)> callback) { bool Recorder::RecordNextFrame(const std::function<void(const Path &)> callback) {
if (!nextFrame) { if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) {
flipLastAction = gpuStats.totals.numFlips; return false;
flipFinishAt = -1;
writeCallback = callback;
nextFrame = true;
return true;
} }
return false; std::lock_guard<std::mutex> guard(callbackLock_);
// Don't take over a recording in progress, it would get the wrong callback and end point.
if (nextFrame || active) {
return false;
}
flipLastAction = gpuStats.totals.numFlips;
flipFinishAt = -1;
writeCallback = callback;
nextFrame = true;
return true;
} }
void Recorder::FinishRecording() { void Recorder::FinishRecording() {
@@ -596,15 +608,21 @@ void Recorder::FinishRecording() {
lastVRAM.clear(); lastVRAM.clear();
NOTICE_LOG(Log::System, "Recording finished"); NOTICE_LOG(Log::System, "Recording finished");
active = false;
flipLastAction = gpuStats.totals.numFlips; flipLastAction = gpuStats.totals.numFlips;
flipFinishAt = -1; flipFinishAt = -1;
lastEdramTrans = 0x400; lastEdramTrans = 0x400;
if (writeCallback) { std::function<void(const Path &)> callback;
writeCallback(filename); {
std::lock_guard<std::mutex> guard(callbackLock_);
callback = std::move(writeCallback);
writeCallback = nullptr;
active = false;
}
if (callback && !filename.empty()) {
callback(filename);
} }
writeCallback = nullptr;
} }
void Recorder::CheckEdramTrans() { void Recorder::CheckEdramTrans() {
+4 -1
View File
@@ -19,6 +19,7 @@
#include <functional> #include <functional>
#include <atomic> #include <atomic>
#include <mutex>
#include <vector> #include <vector>
#include <set> #include <set>
@@ -50,7 +51,7 @@ public:
} }
bool RecordNextFrame(const std::function<void(const Path &)> callback); bool RecordNextFrame(const std::function<void(const Path &)> callback);
void ClearCallback() { void ClearCallback() {
// Not super thread safe.. std::lock_guard<std::mutex> guard(callbackLock_);
writeCallback = nullptr; writeCallback = nullptr;
} }
@@ -95,6 +96,8 @@ private:
int flipFinishAt = -1; int flipFinishAt = -1;
uint32_t lastEdramTrans = 0x400; uint32_t lastEdramTrans = 0x400;
std::function<void(const Path &)> writeCallback; std::function<void(const Path &)> writeCallback;
// RecordNextFrame is called from other threads. Guards writeCallback, nextFrame and the writes to active.
std::mutex callbackLock_;
std::vector<u8> pushbuf; std::vector<u8> pushbuf;
std::vector<Command> commands; std::vector<Command> commands;
+1
View File
@@ -229,6 +229,7 @@ void DrawEngineGLES::Flush() {
numDrawVerts_ = 0; numDrawVerts_ = 0;
numDrawInds_ = 0; numDrawInds_ = 0;
vertexCountInDrawCalls_ = 0; vertexCountInDrawCalls_ = 0;
numVertsToDecode_ = 0;
decodeVertsCounter_ = 0; decodeVertsCounter_ = 0;
decodeIndsCounter_ = 0; decodeIndsCounter_ = 0;
return; return;
+1 -1
View File
@@ -144,7 +144,7 @@ GLRTexture *FragmentTestCacheGLES::CreateTestTexture(const GEComparison funcs[4]
} }
void FragmentTestCacheGLES::Clear(bool deleteThem) { void FragmentTestCacheGLES::Clear(bool deleteThem) {
if (deleteThem) { if (deleteThem && render_) {
for (const auto &[_, v] : cache_) { for (const auto &[_, v] : cache_) {
render_->DeleteTexture(v.texture); render_->DeleteTexture(v.texture);
} }
+2 -1
View File
@@ -68,7 +68,8 @@ public:
void BindTestTexture(int slot); void BindTestTexture(int slot);
void DeviceLost() { void DeviceLost() {
Clear(false); // Queue the deletes anyway, the deleter frees the GLRTexture objects and skips the GL calls if needed.
Clear(true);
render_ = nullptr; render_ = nullptr;
} }
void DeviceRestore(Draw::DrawContext *draw); void DeviceRestore(Draw::DrawContext *draw);
+4
View File
@@ -37,6 +37,8 @@ class IOFile;
class LinkedShader { class LinkedShader {
public: public:
LinkedShader(const LinkedShader &) = delete;
LinkedShader &operator=(const LinkedShader &) = delete;
LinkedShader(GLRenderManager *render, VShaderID VSID, Shader *vs, FShaderID FSID, Shader *fs, bool useHWTransform, bool preloading = false); LinkedShader(GLRenderManager *render, VShaderID VSID, Shader *vs, FShaderID FSID, Shader *fs, bool useHWTransform, bool preloading = false);
~LinkedShader(); ~LinkedShader();
@@ -139,6 +141,8 @@ struct ShaderDescGLES {
class Shader { class Shader {
public: public:
Shader(const Shader &) = delete;
Shader &operator=(const Shader &) = delete;
Shader(GLRenderManager *render, const char *code, const std::string &desc, const ShaderDescGLES &params); Shader(GLRenderManager *render, const char *code, const std::string &desc, const ShaderDescGLES &params);
~Shader(); ~Shader();
GLRShader *shader; GLRShader *shader;
+4 -4
View File
@@ -51,10 +51,10 @@ void TextureCacheGLES::SetFramebufferManager(FramebufferManagerGLES *fbManager)
} }
void TextureCacheGLES::ReleaseTexture(TexCacheEntry *entry, bool delete_them) { void TextureCacheGLES::ReleaseTexture(TexCacheEntry *entry, bool delete_them) {
if (delete_them) { // Delete even when !delete_them (device lost): the GLRTexture is a heap object that only the deleter
if (entry->textureName) { // frees, and the deleter skips the GL calls itself once the context is gone.
render_->DeleteTexture(entry->textureName); if (entry->textureName && render_) {
} render_->DeleteTexture(entry->textureName);
} }
entry->textureName = nullptr; entry->textureName = nullptr;
} }
+24 -2
View File
@@ -957,6 +957,7 @@ DLResult GPUCommon::ProcessDLQueue() {
// Nothing to execute here, and leaving it at the head of the queue would block everything // Nothing to execute here, and leaving it at the head of the queue would block everything
// behind it for good. Treat it like a list that ran into an error. // behind it for good. Treat it like a list that ran into an error.
ERROR_LOG(Log::G3D, "Display list %d has a bad pc %08x (state %d), dropping it", listIndex, list.pc, (int)list.state); ERROR_LOG(Log::G3D, "Display list %d has a bad pc %08x (state %d), dropping it", listIndex, list.pc, (int)list.state);
CompleteFailedList(list);
dlQueue.erase(std::remove(dlQueue.begin(), dlQueue.end(), listIndex), dlQueue.end()); dlQueue.erase(std::remove(dlQueue.begin(), dlQueue.end(), listIndex), dlQueue.end());
continue; continue;
} }
@@ -1045,7 +1046,8 @@ DLResult GPUCommon::ProcessDLQueue() {
} }
break; break;
case GPUSTATE_ERROR: case GPUSTATE_ERROR:
// don't do anything - though dunno about error... // The list can't continue. It's removed from the queue below.
CompleteFailedList(list);
break; break;
case GPUSTATE_STALL: case GPUSTATE_STALL:
// Resume work on this same display list later. The GE is still busy with what it has // Resume work on this same display list later. The GE is still busy with what it has
@@ -1390,6 +1392,20 @@ void GPUCommon::Execute_End(u32 op, u32 diff) {
} }
} }
// A list dropped on an error still has to complete like a finished one, or its ID is never freed and
// sceGeListSync on it never returns.
void GPUCommon::CompleteFailedList(DisplayList &list) {
if (list.started && list.context.IsValid()) {
gstate.Restore(list.context);
ReapplyGfxState();
list.started = false;
}
list.state = PSP_GE_DL_STATE_COMPLETED;
list.waitUntilTicks = startingTicks + cyclesExecuted;
busyTicks = std::max(busyTicks, list.waitUntilTicks);
__GeTriggerSync(GPU_SYNC_LIST, list.id, list.waitUntilTicks);
}
void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) { void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) {
// Just resetting, nothing to check bounds for. // Just resetting, nothing to check bounds for.
const u32 count = op & 0xFFFF; const u32 count = op & 0xFFFF;
@@ -1552,8 +1568,10 @@ void GPUCommon::FlushImm() {
bool changed = texturing != prevTexturing || cullEnable != prevCullEnable || dither != prevDither; bool changed = texturing != prevTexturing || cullEnable != prevCullEnable || dither != prevDither;
changed = changed || prevShading != shading || prevFog != fog; changed = changed || prevShading != shading || prevFog != fog;
// Always flush, even if the flags match: DispatchSubmitImm switches to through mode and a different
// vertex decoder, which would otherwise apply to the draws already queued.
Flush();
if (changed) { if (changed) {
Flush();
gstate.antiAliasEnable = (GE_CMD_ANTIALIASENABLE << 24) | (int)antialias; gstate.antiAliasEnable = (GE_CMD_ANTIALIASENABLE << 24) | (int)antialias;
gstate.shademodel = (GE_CMD_SHADEMODE << 24) | (int)shading; gstate.shademodel = (GE_CMD_SHADEMODE << 24) | (int)shading;
gstate.cullfaceEnable = (GE_CMD_CULLFACEENABLE << 24) | (int)cullEnable; gstate.cullfaceEnable = (GE_CMD_CULLFACEENABLE << 24) | (int)cullEnable;
@@ -2290,12 +2308,16 @@ bool GPUCommon::NeedsSlowInterpreter() const {
void GPUCommon::ClearBreakNext() { void GPUCommon::ClearBreakNext() {
breakNext_ = GPUDebug::BreakNext::NONE; breakNext_ = GPUDebug::BreakNext::NONE;
breakAtCount_ = -1; breakAtCount_ = -1;
// A step that never reached its target leaves these behind, and they'd trip unexpectedly later.
breakpoints_.ClearTempBreakpoints();
GPUStepping::ResumeFromStepping(); GPUStepping::ResumeFromStepping();
} }
void GPUCommon::SetBreakNext(GPUDebug::BreakNext next) { void GPUCommon::SetBreakNext(GPUDebug::BreakNext next) {
breakNext_ = next; breakNext_ = next;
breakAtCount_ = -1; breakAtCount_ = -1;
// Drop the ones from a previous step that didn't get there, before adding this one's.
breakpoints_.ClearTempBreakpoints();
switch (next) { switch (next) {
case GPUDebug::BreakNext::TEX: case GPUDebug::BreakNext::TEX:
breakpoints_.AddTextureChangeTempBreakpoint(); breakpoints_.AddTextureChangeTempBreakpoint();
+3
View File
@@ -56,6 +56,8 @@ inline bool IsTrianglePrim(GEPrimitiveType prim) {
struct TransformStats; struct TransformStats;
class GPUCommon { class GPUCommon {
public: public:
GPUCommon(const GPUCommon &) = delete;
GPUCommon &operator=(const GPUCommon &) = delete;
// The constructor might run on the loader thread. // The constructor might run on the loader thread.
GPUCommon(GraphicsContext *gfxCtx, Draw::DrawContext *draw); GPUCommon(GraphicsContext *gfxCtx, Draw::DrawContext *draw);
virtual ~GPUCommon() = default; virtual ~GPUCommon() = default;
@@ -452,4 +454,5 @@ protected:
private: private:
void DoExecuteCall(u32 target); void DoExecuteCall(u32 target);
void PopDLQueue(); void PopDLQueue();
void CompleteFailedList(DisplayList &list);
}; };
+13 -2
View File
@@ -246,21 +246,32 @@ void BinManager::UpdateState() {
if (newMaxTasks > MAX_POSSIBLE_TASKS) if (newMaxTasks > MAX_POSSIBLE_TASKS)
newMaxTasks = MAX_POSSIBLE_TASKS; newMaxTasks = MAX_POSSIBLE_TASKS;
// We don't want to overlap wrong, so flush any pending. // We don't want to overlap wrong, so flush any pending.
bool flushed = false;
if (maxTasks_ != newMaxTasks) { if (maxTasks_ != newMaxTasks) {
maxTasks_ = newMaxTasks; maxTasks_ = newMaxTasks;
Flush("selfrender"); Flush("selfrender");
flushed = true;
} }
pendingOverlap_ = pendingOverlap_ || selfRender;
// Lastly, we have to check if we're newly writing depth we were texturing before. // Lastly, we have to check if we're newly writing depth we were texturing before.
// This happens in Call of Duty (depth clear after depth texture), for example. // This happens in Call of Duty (depth clear after depth texture), for example.
if (!hadDepth && state.pixelID.depthWrite) { if (!flushed && !hadDepth && state.pixelID.depthWrite) {
for (size_t i = 0; i < states_.Size(); ++i) { for (size_t i = 0; i < states_.Size(); ++i) {
if (HasTextureWrite(states_.Peek(i))) { if (HasTextureWrite(states_.Peek(i))) {
Flush("selfdepth"); Flush("selfdepth");
flushed = true;
break;
} }
} }
} }
if (flushed) {
// The flush forgot what this draw writes and reads, so record it again.
MarkPendingWrites(state);
MarkPendingReads(state);
ClearDirty(SoftDirty::BINNER_RANGE);
}
pendingOverlap_ = pendingOverlap_ || selfRender;
ClearDirty(SoftDirty::BINNER_OVERLAP); ClearDirty(SoftDirty::BINNER_OVERLAP);
} }
} }
+4
View File
@@ -57,6 +57,8 @@ struct BinItem {
template <typename T, size_t N> template <typename T, size_t N>
struct BinQueue { struct BinQueue {
BinQueue(const BinQueue &) = delete;
BinQueue &operator=(const BinQueue &) = delete;
BinQueue() { BinQueue() {
Reset(); Reset();
} }
@@ -185,6 +187,8 @@ struct BinDirtyRange {
class StringWriter; class StringWriter;
class BinManager { class BinManager {
public: public:
BinManager(const BinManager &) = delete;
BinManager &operator=(const BinManager &) = delete;
BinManager(); BinManager();
~BinManager(); ~BinManager();
+2 -2
View File
@@ -826,8 +826,8 @@ void SoftGPU::Execute_BlockTransferStart(u32 op, u32 diff) {
// Need to flush both source and target, so we overwrite properly. // Need to flush both source and target, so we overwrite properly.
if (Memory::IsValidRange(src, srcSize) && Memory::IsValidRange(dst, dstSize)) { if (Memory::IsValidRange(src, srcSize) && Memory::IsValidRange(dst, dstSize)) {
drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", false, src, srcStride, width * bpp, height); drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", false, src, srcStride * bpp, width * bpp, height);
drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", true, dst, dstStride, width * bpp, height); drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", true, dst, dstStride * bpp, width * bpp, height);
} else { } else {
drawEngine_->transformUnit.Flush(this, "blockxfer_wrap"); drawEngine_->transformUnit.Flush(this, "blockxfer_wrap");
} }
+2
View File
@@ -111,6 +111,8 @@ class StringWriter;
class TransformUnit { class TransformUnit {
public: public:
TransformUnit(const TransformUnit &) = delete;
TransformUnit &operator=(const TransformUnit &) = delete;
TransformUnit(); TransformUnit();
~TransformUnit(); ~TransformUnit();
+1
View File
@@ -550,6 +550,7 @@ void DrawEngineVulkan::ResetAfterSkippedDraw() {
numDrawVerts_ = 0; numDrawVerts_ = 0;
numDrawInds_ = 0; numDrawInds_ = 0;
vertexCountInDrawCalls_ = 0; vertexCountInDrawCalls_ = 0;
numVertsToDecode_ = 0;
decodeIndsCounter_ = 0; decodeIndsCounter_ = 0;
decodeVertsCounter_ = 0; decodeVertsCounter_ = 0;
gstate_c.vertexFullAlpha = true; gstate_c.vertexFullAlpha = true;
+17 -9
View File
@@ -90,7 +90,7 @@ GPU_Vulkan::GPU_Vulkan(GraphicsContext *gfxCtx, Draw::DrawContext *draw)
if (discID.size()) { if (discID.size()) {
File::CreateFullPath(GetSysDirectory(DIRECTORY_APP_CACHE)); File::CreateFullPath(GetSysDirectory(DIRECTORY_APP_CACHE));
shaderCachePath_ = GetSysDirectory(DIRECTORY_APP_CACHE) / (discID + ".vkshadercache"); shaderCachePath_ = GetSysDirectory(DIRECTORY_APP_CACHE) / (discID + ".vkshadercache");
LoadCache(shaderCachePath_); LoadCache(shaderCachePath_, true);
} }
InitDeviceObjects(); InitDeviceObjects();
@@ -101,7 +101,7 @@ void GPU_Vulkan::FinishInitOnMainThread() {
framebufferManagerVulkan_->Init(msaaLevel_); framebufferManagerVulkan_->Init(msaaLevel_);
} }
void GPU_Vulkan::LoadCache(const Path &filename) { void GPU_Vulkan::LoadCache(const Path &filename, bool waitForPipelines) {
_dbg_assert_(draw_); _dbg_assert_(draw_);
if (!g_Config.bShaderCache) { if (!g_Config.bShaderCache) {
WARN_LOG(Log::G3D, "Shader cache disabled. Not loading."); WARN_LOG(Log::G3D, "Shader cache disabled. Not loading.");
@@ -134,13 +134,15 @@ void GPU_Vulkan::LoadCache(const Path &filename) {
} }
fclose(f); fclose(f);
// Now, since we're on the loader thread, we can just block here until all pipelines are actually created. if (waitForPipelines) {
// This makes it so that the on-screen spinner keeps spinning until we are done. // Now, since we're on the loader thread, we can just block here until all pipelines are actually created.
double start = time_now_d(); // This makes it so that the on-screen spinner keeps spinning until we are done.
VulkanRenderManager *rm = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER); double start = time_now_d();
int maxTasksSeen = rm->WaitForPipelines(); VulkanRenderManager *rm = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER);
double seconds = time_now_d() - start; int maxTasksSeen = rm->WaitForPipelines();
INFO_LOG(Log::G3D, "Waited %0.1fms for at least %d pipeline tasks to finish compiling.", seconds * 1000.0, maxTasksSeen); double seconds = time_now_d() - start;
INFO_LOG(Log::G3D, "Waited %0.1fms for at least %d pipeline tasks to finish compiling.", seconds * 1000.0, maxTasksSeen);
}
if (!result) { if (!result) {
WARN_LOG(Log::G3D, "Incompatible Vulkan pipeline cache - rebuilding."); WARN_LOG(Log::G3D, "Incompatible Vulkan pipeline cache - rebuilding.");
@@ -402,6 +404,12 @@ void GPU_Vulkan::DeviceRestore(Draw::DrawContext *draw) {
pipelineManager_->DeviceRestore(vulkan); pipelineManager_->DeviceRestore(vulkan);
InitDeviceObjects(); InitDeviceObjects();
// DeviceLost saved the cache and then threw everything away. Load it again, or the next save would
// overwrite the file with only what gets drawn from now on.
if (shaderCachePath_.Valid()) {
LoadCache(shaderCachePath_, false);
}
} }
void GPU_Vulkan::GetStats(StringWriter &w) { void GPU_Vulkan::GetStats(StringWriter &w) {
+1 -1
View File
@@ -69,7 +69,7 @@ private:
void InitDeviceObjects(); void InitDeviceObjects();
void DestroyDeviceObjects(); void DestroyDeviceObjects();
void LoadCache(const Path &filename); void LoadCache(const Path &filename, bool waitForPipelines);
void SaveCache(const Path &filename); void SaveCache(const Path &filename);
FramebufferManagerVulkan *framebufferManagerVulkan_; FramebufferManagerVulkan *framebufferManagerVulkan_;
+3
View File
@@ -63,6 +63,9 @@ private:
// Simply wraps a Vulkan pipeline, providing some metadata. // Simply wraps a Vulkan pipeline, providing some metadata.
struct VulkanPipeline { struct VulkanPipeline {
VulkanPipeline() = default;
VulkanPipeline(const VulkanPipeline &) = delete;
VulkanPipeline &operator=(const VulkanPipeline &) = delete;
~VulkanPipeline() { ~VulkanPipeline() {
desc->Release(); desc->Release();
} }
+4
View File
@@ -40,6 +40,8 @@ class VulkanPushPool;
class VulkanFragmentShader { class VulkanFragmentShader {
public: public:
VulkanFragmentShader(const VulkanFragmentShader &) = delete;
VulkanFragmentShader &operator=(const VulkanFragmentShader &) = delete;
VulkanFragmentShader(VulkanContext *vulkan, FShaderID id, FragmentShaderFlags flags, const char *code, SPIRVCache *cache); VulkanFragmentShader(VulkanContext *vulkan, FShaderID id, FragmentShaderFlags flags, const char *code, SPIRVCache *cache);
~VulkanFragmentShader(); ~VulkanFragmentShader();
@@ -63,6 +65,8 @@ protected:
class VulkanVertexShader { class VulkanVertexShader {
public: public:
VulkanVertexShader(const VulkanVertexShader &) = delete;
VulkanVertexShader &operator=(const VulkanVertexShader &) = delete;
VulkanVertexShader(VulkanContext *vulkan, VShaderID id, VertexShaderFlags flags, const char *code, bool useHWTransform, SPIRVCache *cache); VulkanVertexShader(VulkanContext *vulkan, VShaderID id, VertexShaderFlags flags, const char *code, bool useHWTransform, SPIRVCache *cache);
~VulkanVertexShader(); ~VulkanVertexShader();
+1 -1
View File
@@ -143,7 +143,7 @@ VkSampler SamplerCache::GetOrCreateSampler(const SamplerCacheKey &key) {
if (key.aniso) { if (key.aniso) {
// Docs say the min of this value and the supported max are used. // Docs say the min of this value and the supported max are used.
samp.maxAnisotropy = 1 << g_Config.iAnisotropyLevel; samp.maxAnisotropy = 1 << key.anisoLevel;
samp.anisotropyEnable = true; samp.anisotropyEnable = true;
} else { } else {
samp.maxAnisotropy = 1.0f; samp.maxAnisotropy = 1.0f;
+2
View File
@@ -37,6 +37,8 @@ class StringWriter;
class SamplerCache { class SamplerCache {
public: public:
SamplerCache(const SamplerCache &) = delete;
SamplerCache &operator=(const SamplerCache &) = delete;
SamplerCache(VulkanContext *vulkan) : vulkan_(vulkan), cache_(16) {} SamplerCache(VulkanContext *vulkan) : vulkan_(vulkan), cache_(16) {}
~SamplerCache(); ~SamplerCache();
VkSampler GetOrCreateSampler(const SamplerCacheKey &key); VkSampler GetOrCreateSampler(const SamplerCacheKey &key);
+1 -1
View File
@@ -330,7 +330,7 @@ enum GEVertexType : uint32_t {
GE_VTYPE_IDX_NONE = (0<<11), GE_VTYPE_IDX_NONE = (0<<11),
GE_VTYPE_IDX_8BIT = (1<<11), GE_VTYPE_IDX_8BIT = (1<<11),
GE_VTYPE_IDX_16BIT = (2<<11), GE_VTYPE_IDX_16BIT = (2<<11),
GE_VTYPE_IDX_32BIT = (3<<11), GE_VTYPE_IDX_32BIT = (3<<11), // In name only: the hardware ignores the upper 16 bits of each index.
GE_VTYPE_IDX_MASK = (3<<11), GE_VTYPE_IDX_MASK = (3<<11),
#define GE_VTYPE_IDX_SHIFT 11 #define GE_VTYPE_IDX_SHIFT 11
}; };
+9 -3
View File
@@ -96,9 +96,14 @@ void DrawFramebuffersWindow(ImConfig &cfg, FramebufferManagerCommon *framebuffer
ImGui::SliderFloat("Scale", &cfg.fbViewerZoom, 0.5f, 16.0f, "%.2f", ImGuiSliderFlags_Logarithmic); ImGui::SliderFloat("Scale", &cfg.fbViewerZoom, 0.5f, 16.0f, "%.2f", ImGuiSliderFlags_Logarithmic);
// Now, draw the image of the selected framebuffer. // Now, draw the image of the selected framebuffer.
// Null without buffered rendering, or if creating it failed.
Draw::Framebuffer *fb = vfbs[cfg.selectedFramebuffer]->fbo; Draw::Framebuffer *fb = vfbs[cfg.selectedFramebuffer]->fbo;
ImTextureID texId = ImGui_ImplThin3d_AddFBAsTextureTemp(fb, Draw::Aspect::COLOR_BIT, ImGuiPipeline::TexturedOpaque); if (fb) {
ImGui::Image(texId, ImVec2(fb->Width() * cfg.fbViewerZoom, fb->Height() * cfg.fbViewerZoom)); ImTextureID texId = ImGui_ImplThin3d_AddFBAsTextureTemp(fb, Draw::Aspect::COLOR_BIT, ImGuiPipeline::TexturedOpaque);
ImGui::Image(texId, ImVec2(fb->Width() * cfg.fbViewerZoom, fb->Height() * cfg.fbViewerZoom));
} else {
ImGui::TextUnformatted("(no framebuffer object)");
}
} }
ImGui::End(); ImGui::End();
@@ -602,7 +607,8 @@ ImGeReadbackViewer::~ImGeReadbackViewer() {
} }
VirtualFramebuffer *ImGeReadbackViewer::GetVFB(FramebufferManagerCommon *fbMan) const { VirtualFramebuffer *ImGeReadbackViewer::GetVFB(FramebufferManagerCommon *fbMan) const {
return fbMan->GetExactVFB(gstate.getFrameBufAddress(), gstate.FrameBufStride(), gstate.FrameBufFormat()); // fbMan is null with the software renderer.
return fbMan ? fbMan->GetExactVFB(gstate.getFrameBufAddress(), gstate.FrameBufStride(), gstate.FrameBufFormat()) : nullptr;
} }
void ImGeReadbackViewer::DeviceLost() { void ImGeReadbackViewer::DeviceLost() {