mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #22382 from hrydgard/gpu-lifecycle-fixes-2
Claude code review: GPU fixes 2
This commit is contained in:
54 files changed
+297
-77
No files matched your search
@@ -30,6 +30,8 @@ static inline int operator &(const GLBufferStrategy &lhs, const GLBufferStrategy
|
||||
|
||||
class GLRBuffer {
|
||||
public:
|
||||
GLRBuffer(const GLRBuffer &) = delete;
|
||||
GLRBuffer &operator=(const GLRBuffer &) = delete;
|
||||
GLRBuffer(GLuint target, size_t size) : target_(target), size_((int)size) {}
|
||||
~GLRBuffer() {
|
||||
if (buffer_) {
|
||||
|
||||
@@ -147,6 +147,7 @@ void GLQueueRunner::RunInitSteps(const FastVec<GLRInitStep> &steps, bool skipGLC
|
||||
case GLRInitStepType::CREATE_SHADER:
|
||||
{
|
||||
WARN_LOG(Log::G3D, "CREATE_SHADER found with skipGLCalls, not good");
|
||||
delete[] step.create_shader.code;
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -678,6 +679,9 @@ void GLQueueRunner::RunSteps(const std::vector<GLRStep *> &steps, GLFrameData &f
|
||||
}
|
||||
}
|
||||
break;
|
||||
case GLRRenderCommand::UNIFORMSTEREOMATRIX:
|
||||
delete[] c.uniformStereoMatrix4.mData;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -110,11 +110,11 @@ void GLRenderManager::ThreadEnd() {
|
||||
frameData_[i].deleter_prev.Perform(this, skipGLCalls_);
|
||||
}
|
||||
deleter_.Perform(this, skipGLCalls_);
|
||||
for (int i = 0; i < (int)steps_.size(); i++) {
|
||||
delete steps_[i];
|
||||
}
|
||||
steps_.clear();
|
||||
// Steps that never got submitted. A dry run frees the data they own (texture uploads etc), and the steps.
|
||||
queueRunner_.RunInitSteps(initSteps_, true);
|
||||
initSteps_.clear();
|
||||
queueRunner_.RunSteps(steps_, frameData_[0], true, false, false);
|
||||
steps_.clear();
|
||||
INFO_LOG(Log::G3D, "GLRenderManager::ThreadEnd end");
|
||||
}
|
||||
|
||||
|
||||
@@ -30,6 +30,8 @@ constexpr int MAX_GL_TEXTURE_SLOTS = 8;
|
||||
|
||||
class GLRTexture {
|
||||
public:
|
||||
GLRTexture(const GLRTexture &) = delete;
|
||||
GLRTexture &operator=(const GLRTexture &) = delete;
|
||||
GLRTexture(const Draw::DeviceCaps &caps, int width, int height, int depth, int numMips);
|
||||
~GLRTexture();
|
||||
|
||||
@@ -53,6 +55,8 @@ public:
|
||||
|
||||
class GLRFramebuffer {
|
||||
public:
|
||||
GLRFramebuffer(const GLRFramebuffer &) = delete;
|
||||
GLRFramebuffer &operator=(const GLRFramebuffer &) = delete;
|
||||
GLRFramebuffer(const Draw::DeviceCaps &caps, int _width, int _height, bool z_stencil, const char *tag)
|
||||
: color_texture(caps, _width, _height, 1, 1), z_stencil_texture(caps, _width, _height, 1, 1),
|
||||
width(_width), height(_height), z_stencil_(z_stencil) {
|
||||
@@ -83,6 +87,8 @@ private:
|
||||
|
||||
class GLRShader {
|
||||
public:
|
||||
GLRShader(const GLRShader &) = delete;
|
||||
GLRShader &operator=(const GLRShader &) = delete;
|
||||
explicit GLRShader(std::string_view _desc) : desc(_desc) {}
|
||||
~GLRShader() {
|
||||
if (shader) {
|
||||
@@ -116,6 +122,9 @@ public:
|
||||
|
||||
class GLRProgram {
|
||||
public:
|
||||
GLRProgram() = default;
|
||||
GLRProgram(const GLRProgram &) = delete;
|
||||
GLRProgram &operator=(const GLRProgram &) = delete;
|
||||
~GLRProgram() {
|
||||
if (deleteCallback_) {
|
||||
deleteCallback_(deleteParam_);
|
||||
|
||||
@@ -12,6 +12,8 @@ struct VKRImage;
|
||||
|
||||
class VulkanBarrierBatch {
|
||||
public:
|
||||
VulkanBarrierBatch(const VulkanBarrierBatch &) = delete;
|
||||
VulkanBarrierBatch &operator=(const VulkanBarrierBatch &) = delete;
|
||||
VulkanBarrierBatch() : imageBarriers_(4) {}
|
||||
~VulkanBarrierBatch();
|
||||
|
||||
|
||||
@@ -18,6 +18,8 @@ enum class BindingType {
|
||||
// Only appropriate for use in a per-frame pool.
|
||||
class VulkanDescSetPool {
|
||||
public:
|
||||
VulkanDescSetPool(const VulkanDescSetPool &) = delete;
|
||||
VulkanDescSetPool &operator=(const VulkanDescSetPool &) = delete;
|
||||
VulkanDescSetPool(const char *tag, bool grow = true) : tag_(tag), grow_(grow) {}
|
||||
~VulkanDescSetPool();
|
||||
|
||||
|
||||
@@ -60,6 +60,8 @@ struct VKRImage {
|
||||
|
||||
class VKRFramebuffer {
|
||||
public:
|
||||
VKRFramebuffer(const VKRFramebuffer &) = delete;
|
||||
VKRFramebuffer &operator=(const VKRFramebuffer &) = delete;
|
||||
VKRFramebuffer(VulkanContext *vk, VulkanBarrierBatch *barriers, int _width, int _height, int _numLayers, int _multiSampleLevel, bool createDepthStencilBuffer, const char *tag);
|
||||
~VKRFramebuffer();
|
||||
|
||||
|
||||
@@ -22,6 +22,8 @@ struct TextureCopyBatch {
|
||||
// ALWAYS use an allocator when calling CreateDirect.
|
||||
class VulkanTexture {
|
||||
public:
|
||||
VulkanTexture(const VulkanTexture &) = delete;
|
||||
VulkanTexture &operator=(const VulkanTexture &) = delete;
|
||||
VulkanTexture(VulkanContext *vulkan, const char *tag);
|
||||
~VulkanTexture() {
|
||||
Destroy();
|
||||
|
||||
@@ -115,6 +115,8 @@ public:
|
||||
|
||||
// Wrapped pipeline. Does own desc!
|
||||
struct VKRGraphicsPipeline {
|
||||
VKRGraphicsPipeline(const VKRGraphicsPipeline &) = delete;
|
||||
VKRGraphicsPipeline &operator=(const VKRGraphicsPipeline &) = delete;
|
||||
VKRGraphicsPipeline(PipelineFlags flags, const char *tag) : flags_(flags), tag_(tag) {}
|
||||
~VKRGraphicsPipeline();
|
||||
|
||||
@@ -195,6 +197,9 @@ static_assert(sizeof(PackedDescriptor::buffer) == 16, "PackedDescriptor should b
|
||||
// Note that we only support a single descriptor set due to compatibility with some ancient devices.
|
||||
// We should probably eventually give that up eventually.
|
||||
struct VKRPipelineLayout {
|
||||
VKRPipelineLayout() = default;
|
||||
VKRPipelineLayout(const VKRPipelineLayout &) = delete;
|
||||
VKRPipelineLayout &operator=(const VKRPipelineLayout &) = delete;
|
||||
~VKRPipelineLayout();
|
||||
|
||||
enum { MAX_DESC_SET_BINDINGS = 5 };
|
||||
|
||||
@@ -51,6 +51,8 @@ class TextDrawer;
|
||||
|
||||
class DrawBuffer {
|
||||
public:
|
||||
DrawBuffer(const DrawBuffer &) = delete;
|
||||
DrawBuffer &operator=(const DrawBuffer &) = delete;
|
||||
DrawBuffer();
|
||||
~DrawBuffer();
|
||||
|
||||
|
||||
@@ -96,6 +96,9 @@ struct AtlasFontHeader {
|
||||
};
|
||||
|
||||
struct AtlasFont {
|
||||
AtlasFont() = default;
|
||||
AtlasFont(const AtlasFont &) = delete;
|
||||
AtlasFont &operator=(const AtlasFont &) = delete;
|
||||
~AtlasFont();
|
||||
|
||||
float padding;
|
||||
@@ -126,6 +129,9 @@ struct AtlasHeader {
|
||||
};
|
||||
|
||||
struct Atlas {
|
||||
Atlas() = default;
|
||||
Atlas(const Atlas &) = delete;
|
||||
Atlas &operator=(const Atlas &) = delete;
|
||||
~Atlas();
|
||||
bool LoadMeta(const uint8_t *data, size_t data_size);
|
||||
bool IsMetadataLoaded() const {
|
||||
|
||||
@@ -199,7 +199,9 @@ bool FramebufferManagerCommon::ReadbackDepthbuffer(Draw::Framebuffer *fbo, int x
|
||||
|
||||
auto *blitFBO = GetTempFBO(TempFBO::Z_COPY, fbo->Width() * scaleX, fbo->Height() * scaleY);
|
||||
draw_->BindFramebufferAsRenderTarget(blitFBO, { RPAction::DONT_CARE, RPAction::DONT_CARE, RPAction::DONT_CARE }, "ReadbackDepthbufferSync");
|
||||
Draw::Viewport viewport = { 0.0f, 0.0f, (float)destW, (float)destH, 0.0f, 1.0f };
|
||||
// The whole fbo is drawn, so the viewport has to cover all of it at the destination scale. Not just
|
||||
// destW x destH, which would squeeze it whenever the read rectangle is smaller than the fbo.
|
||||
Draw::Viewport viewport = { 0.0f, 0.0f, fbo->Width() * scaleX, fbo->Height() * scaleY, 0.0f, 1.0f };
|
||||
draw_->SetViewport(viewport);
|
||||
draw_->SetScissorRect(0, 0, fbo->Width() * scaleX, fbo->Height() * scaleY);
|
||||
|
||||
|
||||
@@ -51,11 +51,13 @@ DrawEngineCommon::DrawEngineCommon() : decoderMap_(32) {
|
||||
transformedExpanded_ = (TransformedVertex *)AllocateMemoryPages(3 * TRANSFORMED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
|
||||
decoded_ = (u8 *)AllocateMemoryPages(DECODED_VERTEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
|
||||
decIndex_ = (u16 *)AllocateMemoryPages(DECODED_INDEX_BUFFER_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
|
||||
bboxScratch_ = (u8 *)AllocateMemoryPages(BBOX_SCRATCH_SIZE, MEM_PROT_READ | MEM_PROT_WRITE);
|
||||
|
||||
_dbg_assert_(transformed_);
|
||||
_dbg_assert_(transformedExpanded_);
|
||||
_dbg_assert_(decoded_);
|
||||
_dbg_assert_(decIndex_);
|
||||
_dbg_assert_(bboxScratch_);
|
||||
|
||||
indexGen.Setup(decIndex_);
|
||||
|
||||
@@ -65,6 +67,7 @@ DrawEngineCommon::DrawEngineCommon() : decoderMap_(32) {
|
||||
DrawEngineCommon::~DrawEngineCommon() {
|
||||
FreeMemoryPages(decoded_, DECODED_VERTEX_BUFFER_SIZE);
|
||||
FreeMemoryPages(decIndex_, DECODED_INDEX_BUFFER_SIZE);
|
||||
FreeMemoryPages(bboxScratch_, BBOX_SCRATCH_SIZE);
|
||||
FreeMemoryPages(transformed_, TRANSFORMED_VERTEX_BUFFER_SIZE);
|
||||
FreeMemoryPages(transformedExpanded_, 3 * TRANSFORMED_VERTEX_BUFFER_SIZE);
|
||||
ShutdownDepthRaster();
|
||||
@@ -183,15 +186,14 @@ void DrawEngineCommon::DispatchSubmitImm(GEPrimitiveType prim, TransformedVertex
|
||||
// - Less accurate, but..
|
||||
// - Only requires six plane evaluations then.
|
||||
bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int vertexCount, const VertexDecoder *dec, u32 vertType) {
|
||||
// Grab temp buffer space from large offsets in decoded_. Not exactly safe for large draws.
|
||||
// Although this may lead to drawing that shouldn't happen, the viewport is more complex on VR.
|
||||
// Let's always say objects are within bounds.
|
||||
// The scratch buffer is sized for 1024 vertices. Although this may lead to drawing that shouldn't happen,
|
||||
// the viewport is more complex on VR. Let's always say objects are within bounds.
|
||||
if (vertexCount > 1024 || gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
SimpleVertex *corners = (SimpleVertex *)(decoded_ + 65536 * 12);
|
||||
float *verts = (float *)(decoded_ + 65536 * 18);
|
||||
SimpleVertex *corners = (SimpleVertex *)(bboxScratch_ + BBOX_SCRATCH_CORNERS_OFFSET);
|
||||
float *verts = (float *)(bboxScratch_ + BBOX_SCRATCH_VERTS_OFFSET);
|
||||
|
||||
// Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder
|
||||
// and a large vertex format.
|
||||
@@ -218,7 +220,7 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
|
||||
}
|
||||
} else {
|
||||
// Simplify away indices, bones, and morph before proceeding.
|
||||
u8 *temp_buffer = decoded_ + 65536 * 24;
|
||||
u8 *temp_buffer = bboxScratch_ + BBOX_SCRATCH_TEMP_OFFSET;
|
||||
|
||||
if ((inds || (vertType & (GE_VTYPE_WEIGHT_MASK | GE_VTYPE_MORPHCOUNT_MASK)))) {
|
||||
// Need for Speed Carbon ends up on this path! With a single bone weight.
|
||||
@@ -392,7 +394,8 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, cons
|
||||
}
|
||||
case GE_VTYPE_IDX_32BIT:
|
||||
{
|
||||
u32 idx = ((u32 *)idata)[i];
|
||||
// The PSP ignores the upper 16 bits.
|
||||
u16 idx = (u16)((u32 *)idata)[i];
|
||||
data = (const s8 *)srcdata + idx * stride;
|
||||
break;
|
||||
}
|
||||
@@ -759,7 +762,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *
|
||||
if (IsTrianglePrim(newPrim) != isTriangle)
|
||||
break;
|
||||
int vertexCount = data & 0xFFFF;
|
||||
if (numDrawInds >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + offset + vertexCount > VERTEX_BUFFER_MAX) {
|
||||
if (numDrawInds >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + offset + vertexCount > VERTEX_BUFFER_MAX || numVertsToDecode_ + (offset - dv.vertexCount) + vertexCount > VERTEX_BUFFER_MAX) {
|
||||
break;
|
||||
}
|
||||
DeferredInds &di = drawInds_[numDrawInds++];
|
||||
@@ -780,6 +783,7 @@ int DrawEngineCommon::ExtendNonIndexedPrim(const uint32_t *cmd, const uint32_t *
|
||||
dv.vertexCount = offset;
|
||||
dv.indexUpperBound = dv.vertexCount - 1;
|
||||
vertexCountInDrawCalls_ += totalCount;
|
||||
numVertsToDecode_ += totalCount;
|
||||
*bytesRead = totalCount * dec->VertexSize();
|
||||
return cmd - start;
|
||||
}
|
||||
@@ -805,7 +809,21 @@ void DrawEngineCommon::SkipPrim(GEPrimitiveType prim, int vertexCount, const Ver
|
||||
|
||||
// vertTypeID is the vertex type but with the UVGen mode smashed into the top bits.
|
||||
bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, const VertexDecoder *dec, u32 vertTypeID, bool clockwise, int *bytesRead, ClipInfoFlags clipInfoFlags) {
|
||||
if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawVerts_ >= MAX_DEFERRED_DRAW_VERTS || numDrawInds_ >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + vertexCount > VERTEX_BUFFER_MAX) {
|
||||
// The index count doesn't bound how many vertices DecodeVerts will produce (the index range can be
|
||||
// sparse), so track that separately, as the growth of the range to decode.
|
||||
u16 lowerBound = 0;
|
||||
u16 upperBound = 0;
|
||||
int decodeGrowth = 0;
|
||||
if (vertexCount > 0) {
|
||||
GetIndexBounds(inds, vertexCount, vertTypeID, &lowerBound, &upperBound);
|
||||
decodeGrowth = upperBound - lowerBound + 1;
|
||||
if (CanExtendDecode(verts, inds, dec)) {
|
||||
const DeferredVerts &last = drawVerts_[numDrawVerts_ - 1];
|
||||
decodeGrowth = std::max(upperBound, last.indexUpperBound) - std::min(lowerBound, last.indexLowerBound) - (last.indexUpperBound - last.indexLowerBound);
|
||||
}
|
||||
}
|
||||
|
||||
if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawVerts_ >= MAX_DEFERRED_DRAW_VERTS || numDrawInds_ >= MAX_DEFERRED_DRAW_INDS || vertexCountInDrawCalls_ + vertexCount > VERTEX_BUFFER_MAX || numVertsToDecode_ + decodeGrowth > VERTEX_BUFFER_MAX) {
|
||||
Flush();
|
||||
}
|
||||
|
||||
@@ -855,6 +873,7 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti
|
||||
const int rem = vertexCount % 3;
|
||||
if (rem != 0) {
|
||||
vertexCount -= rem;
|
||||
GetIndexBounds(inds, vertexCount, vertTypeID, &lowerBound, &upperBound);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -881,17 +900,16 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti
|
||||
|
||||
_dbg_assert_(numDrawVerts <= MAX_DEFERRED_DRAW_VERTS);
|
||||
|
||||
if (inds && numDrawVerts > decodeVertsCounter_ && drawVerts_[numDrawVerts - 1].verts == verts && !applySkin) {
|
||||
if (CanExtendDecode(verts, inds, dec_)) {
|
||||
// Same vertex pointer as a previous un-decoded draw call - let's just extend the decode!
|
||||
di.vertDecodeIndex = numDrawVerts - 1;
|
||||
u16 lb;
|
||||
u16 ub;
|
||||
GetIndexBounds(inds, vertexCount, vertTypeID, &lb, &ub);
|
||||
DeferredVerts &dv = drawVerts_[numDrawVerts - 1];
|
||||
if (lb < dv.indexLowerBound)
|
||||
dv.indexLowerBound = lb;
|
||||
if (ub > dv.indexUpperBound)
|
||||
dv.indexUpperBound = ub;
|
||||
const int oldCount = dv.indexUpperBound - dv.indexLowerBound + 1;
|
||||
if (lowerBound < dv.indexLowerBound)
|
||||
dv.indexLowerBound = lowerBound;
|
||||
if (upperBound > dv.indexUpperBound)
|
||||
dv.indexUpperBound = upperBound;
|
||||
numVertsToDecode_ += dv.indexUpperBound - dv.indexLowerBound + 1 - oldCount;
|
||||
} else {
|
||||
// Record a new draw, and a new index gen.
|
||||
DeferredVerts &dv = drawVerts_[numDrawVerts];
|
||||
@@ -899,8 +917,9 @@ bool DrawEngineCommon::SubmitPrim(const void *verts, const void *inds, GEPrimiti
|
||||
dv.verts = verts;
|
||||
dv.vertexCount = vertexCount;
|
||||
dv.uvScale = LoadUVScaleOffset(gstate);
|
||||
// Does handle the unindexed case.
|
||||
GetIndexBounds(inds, vertexCount, vertTypeID, &dv.indexLowerBound, &dv.indexUpperBound);
|
||||
dv.indexLowerBound = lowerBound;
|
||||
dv.indexUpperBound = upperBound;
|
||||
numVertsToDecode_ += upperBound - lowerBound + 1;
|
||||
}
|
||||
|
||||
vertexCountInDrawCalls_ += vertexCount;
|
||||
@@ -930,8 +949,9 @@ void DrawEngineCommon::DecodeVerts(const VertexDecoder *dec, u8 *dest) {
|
||||
drawVertexOffsets_[i] = numDecodedVerts - indexLowerBound;
|
||||
const int indexUpperBound = dv.indexUpperBound;
|
||||
const int count = indexUpperBound - indexLowerBound + 1;
|
||||
if (count + numDecodedVerts >= VERTEX_BUFFER_MAX) {
|
||||
// Hit our limit! Stop decoding in this draw.
|
||||
if (count + numDecodedVerts > VERTEX_BUFFER_MAX) {
|
||||
// SubmitPrim flushes before this can happen.
|
||||
_dbg_assert_(false);
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -38,6 +38,11 @@ enum {
|
||||
VERTEX_BUFFER_MAX = 65536,
|
||||
DECODED_VERTEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 2 * 36, // 36 == sizeof(SimpleVertex)
|
||||
DECODED_INDEX_BUFFER_SIZE = VERTEX_BUFFER_MAX * 6 * 6 * 2, // * 6 for spline tessellation, then * 6 again for converting into points/lines, and * 2 for 2 bytes per index
|
||||
// TestBoundingBox handles up to 1025 vertices: corners (SimpleVertex), then positions, then decoded vertices.
|
||||
BBOX_SCRATCH_CORNERS_OFFSET = 0,
|
||||
BBOX_SCRATCH_VERTS_OFFSET = 64 * 1024,
|
||||
BBOX_SCRATCH_TEMP_OFFSET = 128 * 1024,
|
||||
BBOX_SCRATCH_SIZE = 256 * 1024,
|
||||
};
|
||||
|
||||
enum {
|
||||
@@ -68,6 +73,8 @@ struct alignas(16) Plane8 {
|
||||
|
||||
class DrawEngineCommon {
|
||||
public:
|
||||
DrawEngineCommon(const DrawEngineCommon &) = delete;
|
||||
DrawEngineCommon &operator=(const DrawEngineCommon &) = delete;
|
||||
DrawEngineCommon();
|
||||
virtual ~DrawEngineCommon();
|
||||
|
||||
@@ -161,6 +168,11 @@ protected:
|
||||
void DecodeVerts(const VertexDecoder *dec, u8 *dest);
|
||||
int DecodeInds();
|
||||
|
||||
// Whether an indexed draw can share the previous draw's vertex decode, by widening its index range.
|
||||
bool CanExtendDecode(const void *verts, const void *inds, const VertexDecoder *dec) const {
|
||||
return inds && numDrawVerts_ > decodeVertsCounter_ && drawVerts_[numDrawVerts_ - 1].verts == verts && !dec->skinInDecode;
|
||||
}
|
||||
|
||||
int ComputeNumVertsToDecode() const;
|
||||
|
||||
void ApplyFramebufferRead(FBOTexState *fboTexState);
|
||||
@@ -211,6 +223,7 @@ protected:
|
||||
numDrawVerts_ = 0;
|
||||
numDrawInds_ = 0;
|
||||
vertexCountInDrawCalls_ = 0;
|
||||
numVertsToDecode_ = 0;
|
||||
decodeIndsCounter_ = 0;
|
||||
decodeVertsCounter_ = 0;
|
||||
seenPrims_ = 0;
|
||||
@@ -273,6 +286,8 @@ protected:
|
||||
// Vertex collector buffers
|
||||
u8 *decoded_ = nullptr;
|
||||
u16 *decIndex_ = nullptr;
|
||||
// Separate from decoded_, which can hold decoded vertices that haven't been flushed yet.
|
||||
u8 *bboxScratch_ = nullptr;
|
||||
|
||||
// Cached vertex decoders
|
||||
DenseHashMap<u32, VertexDecoder *> decoderMap_;
|
||||
@@ -312,6 +327,8 @@ protected:
|
||||
int numDrawVerts_ = 0;
|
||||
int numDrawInds_ = 0;
|
||||
int vertexCountInDrawCalls_ = 0;
|
||||
// How many vertices DecodeVerts will produce for the queued draws. Must stay <= VERTEX_BUFFER_MAX.
|
||||
int numVertsToDecode_ = 0;
|
||||
|
||||
int decodeVertsCounter_ = 0;
|
||||
int decodeIndsCounter_ = 0;
|
||||
|
||||
@@ -3354,7 +3354,8 @@ void FramebufferManagerCommon::FlushBeforeCopy() {
|
||||
// TODO: Replace with with depal, reading the palette from the texture on the GPU directly.
|
||||
void FramebufferManagerCommon::DownloadFramebufferForClut(u32 fb_address, u32 loadBytes) {
|
||||
VirtualFramebuffer *vfb = GetVFBAt(fb_address);
|
||||
if (vfb && vfb->fb_stride != 0) {
|
||||
// Without an fbo there's nothing to read back (ReadbackFramebuffer would read the backbuffer instead).
|
||||
if (vfb && vfb->fb_stride != 0 && vfb->fbo) {
|
||||
const u32 bpp = BufferFormatBytesPerPixel(vfb->fb_format);
|
||||
int x = 0;
|
||||
int y = 0;
|
||||
|
||||
@@ -291,6 +291,8 @@ struct DisplayLayoutConfig;
|
||||
|
||||
class FramebufferManagerCommon {
|
||||
public:
|
||||
FramebufferManagerCommon(const FramebufferManagerCommon &) = delete;
|
||||
FramebufferManagerCommon &operator=(const FramebufferManagerCommon &) = delete;
|
||||
FramebufferManagerCommon(Draw::DrawContext *draw);
|
||||
virtual ~FramebufferManagerCommon();
|
||||
|
||||
|
||||
@@ -342,6 +342,7 @@ void IndexGenerator::TranslatePrim(int prim, int numInds, const u16_le *inds, in
|
||||
}
|
||||
}
|
||||
|
||||
// The PSP ignores the upper 16 bits of 32-bit indices. The u16 output drops them the same way.
|
||||
void IndexGenerator::TranslatePrim(int prim, int numInds, const u32_le *inds, int indexOffset, bool clockwise) {
|
||||
switch (prim) {
|
||||
case GE_PRIM_POINTS: TranslatePoints<u32_le>(numInds, inds, indexOffset); break;
|
||||
|
||||
@@ -555,8 +555,12 @@ static void DoRelease(T *&obj) {
|
||||
|
||||
template <typename T>
|
||||
static void DoReleaseVector(std::vector<T *> &list) {
|
||||
for (auto &obj : list)
|
||||
obj->Release();
|
||||
for (auto &obj : list) {
|
||||
// Can be null when a creation failed partway.
|
||||
if (obj) {
|
||||
obj->Release();
|
||||
}
|
||||
}
|
||||
list.clear();
|
||||
}
|
||||
|
||||
|
||||
@@ -80,6 +80,8 @@ ENUM_CLASS_BITOPS(OutputFlags);
|
||||
|
||||
class PresentationCommon {
|
||||
public:
|
||||
PresentationCommon(const PresentationCommon &) = delete;
|
||||
PresentationCommon &operator=(const PresentationCommon &) = delete;
|
||||
PresentationCommon(Draw::DrawContext *draw);
|
||||
~PresentationCommon();
|
||||
|
||||
|
||||
@@ -278,6 +278,14 @@ void ReplacedTexture::Prepare(VFSBackend *vfs) {
|
||||
}
|
||||
|
||||
result = LoadLevelData(fileRef, desc_.filenames[i], i, &pixelFormat);
|
||||
// A level that got loaded owns the reference. Otherwise (error, or nothing to load) it's ours to free.
|
||||
bool kept = false;
|
||||
for (const ReplacedTextureLevel &level : levels_) {
|
||||
kept = kept || level.fileRef == fileRef;
|
||||
}
|
||||
if (!kept) {
|
||||
vfs_->ReleaseFile(fileRef);
|
||||
}
|
||||
if (result == LoadLevelResult::DONE) {
|
||||
// Loaded all the levels we're gonna get.
|
||||
fmt = pixelFormat;
|
||||
@@ -730,6 +738,7 @@ ReplacedTexture::LoadLevelResult ReplacedTexture::LoadLevelData(VFSFileReference
|
||||
}
|
||||
if (png.width > (uint32_t)level.w || png.height > (uint32_t)level.h) {
|
||||
ERROR_LOG(Log::TexReplacement, "Texture replacement changed since header read: %s", filename.c_str());
|
||||
png_image_free(&png);
|
||||
return LoadLevelResult::LOAD_ERROR;
|
||||
}
|
||||
|
||||
|
||||
@@ -140,6 +140,8 @@ struct ReplacedTextureLevel {
|
||||
|
||||
class ReplacedTexture {
|
||||
public:
|
||||
ReplacedTexture(const ReplacedTexture &) = delete;
|
||||
ReplacedTexture &operator=(const ReplacedTexture &) = delete;
|
||||
ReplacedTexture(VFSBackend *vfs, const ReplacementDesc &desc);
|
||||
~ReplacedTexture();
|
||||
|
||||
|
||||
@@ -123,6 +123,8 @@ enum : uint64_t {
|
||||
|
||||
class ShaderManagerCommon {
|
||||
public:
|
||||
ShaderManagerCommon(const ShaderManagerCommon &) = delete;
|
||||
ShaderManagerCommon &operator=(const ShaderManagerCommon &) = delete;
|
||||
ShaderManagerCommon(Draw::DrawContext *draw) : draw_(draw) {}
|
||||
virtual ~ShaderManagerCommon() {}
|
||||
|
||||
|
||||
@@ -198,7 +198,7 @@ SoftwareTransformAction RunSoftwareTransform(SoftwareTransformParams ¶ms, in
|
||||
float depth = std::clamp(transformed[1].z, 0.0f, 65535.0f) / 65535.0f;
|
||||
// Non-zero depth clears are unusual, but some drivers don't match drawn depth values to cleared values.
|
||||
// Games sometimes expect exact matches (see #12626, for example) for equal comparisons.
|
||||
if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && result->depth > 0.0f && result->depth < 1.0f)) {
|
||||
if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && depth > 0.0f && depth < 1.0f)) {
|
||||
result->color = transformed[1].color0_32;
|
||||
result->depth = depth;
|
||||
gpuStats.perFrame.numClears++;
|
||||
|
||||
@@ -114,6 +114,8 @@ TextureCacheCommon::TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D)
|
||||
}
|
||||
|
||||
TextureCacheCommon::~TextureCacheCommon() {
|
||||
// Only DeviceLost cleared these, which the GLES and D3D11 backends don't go through at shutdown.
|
||||
clutTextureCache_.Clear();
|
||||
FreeAlignedMemory(clutBufConverted_);
|
||||
FreeAlignedMemory(clutBufRaw_);
|
||||
FreeAlignedMemory(expandClut_);
|
||||
@@ -342,6 +344,9 @@ SamplerCacheKey TextureCacheCommon::GetSamplingParams(int maxLevel, const TexCac
|
||||
break;
|
||||
}
|
||||
|
||||
if (key.aniso) {
|
||||
key.anisoLevel = (uint8_t)g_Config.iAnisotropyLevel;
|
||||
}
|
||||
return key;
|
||||
}
|
||||
|
||||
|
||||
@@ -86,6 +86,8 @@ struct SamplerCacheKey {
|
||||
bool tClamp : 1;
|
||||
bool aniso : 1;
|
||||
bool texture3d : 1;
|
||||
// Baked into the sampler objects, so it has to be part of the key, or a change wouldn't apply.
|
||||
uint8_t anisoLevel;
|
||||
};
|
||||
};
|
||||
bool operator < (const SamplerCacheKey &other) const {
|
||||
@@ -158,6 +160,9 @@ ENUM_CLASS_BITOPS(TexStatus);
|
||||
|
||||
// TODO: Shrink this struct. There is some fluff.
|
||||
struct TexCacheEntry {
|
||||
TexCacheEntry() = default;
|
||||
TexCacheEntry(const TexCacheEntry &) = delete;
|
||||
TexCacheEntry &operator=(const TexCacheEntry &) = delete;
|
||||
~TexCacheEntry() {
|
||||
#ifdef _DEBUG
|
||||
if (texturePtr || textureName || vkTex)
|
||||
@@ -336,6 +341,8 @@ struct TextureApplyResult {
|
||||
|
||||
class TextureCacheCommon {
|
||||
public:
|
||||
TextureCacheCommon(const TextureCacheCommon &) = delete;
|
||||
TextureCacheCommon &operator=(const TextureCacheCommon &) = delete;
|
||||
TextureCacheCommon(Draw::DrawContext *draw, Draw2D *draw2D);
|
||||
virtual ~TextureCacheCommon();
|
||||
|
||||
|
||||
@@ -230,6 +230,7 @@ bool TextureReplacer::LoadIni(std::string *error, bool notify) {
|
||||
|
||||
if (filenameMap.empty()) {
|
||||
WARN_LOG(Log::TexReplacement, "No replacement textures found.");
|
||||
delete dir;
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -76,6 +76,8 @@ enum class ReplacerDecimateMode {
|
||||
|
||||
class TextureReplacer {
|
||||
public:
|
||||
TextureReplacer(const TextureReplacer &) = delete;
|
||||
TextureReplacer &operator=(const TextureReplacer &) = delete;
|
||||
// The draw context is checked for supported texture formats.
|
||||
TextureReplacer(Draw::DrawContext *draw);
|
||||
~TextureReplacer();
|
||||
|
||||
@@ -252,7 +252,9 @@ ClutTexture ClutTextureCache::GetClutTexture(GEPaletteFormat clutFormat, const u
|
||||
|
||||
void ClutTextureCache::Clear() {
|
||||
for (auto tex = texCache_.begin(); tex != texCache_.end(); ++tex) {
|
||||
tex->second->texture->Release();
|
||||
if (tex->second->texture) {
|
||||
tex->second->texture->Release();
|
||||
}
|
||||
delete tex->second;
|
||||
}
|
||||
texCache_.clear();
|
||||
@@ -261,7 +263,9 @@ void ClutTextureCache::Clear() {
|
||||
void ClutTextureCache::Decimate() {
|
||||
for (auto tex = texCache_.begin(); tex != texCache_.end(); ) {
|
||||
if (tex->second->lastFrame + DEPAL_TEXTURE_OLD_AGE < gpuStats.totals.numFlips) {
|
||||
tex->second->texture->Release();
|
||||
if (tex->second->texture) {
|
||||
tex->second->texture->Release();
|
||||
}
|
||||
delete tex->second;
|
||||
texCache_.erase(tex++);
|
||||
} else {
|
||||
|
||||
@@ -170,8 +170,9 @@ void GetIndexBounds(const void *inds, int count, u32 vertType, u16 *indexLowerBo
|
||||
bool oob = false;
|
||||
const u32_le *ind32 = (const u32_le *)inds;
|
||||
for (int i = 0; i < count; i++) {
|
||||
// The PSP ignores the upper 16 bits, so only the low ones count.
|
||||
const u16 value = (u16)ind32[i];
|
||||
// These aren't documented and should be rare. Let's bounds check each one.
|
||||
// Games setting the upper bits should be rare, so report them.
|
||||
if (ind32[i] != value) {
|
||||
oob = true;
|
||||
}
|
||||
@@ -1373,6 +1374,12 @@ void VertexDecoder::SetVertexType(u32 fmt, const VertexDecoderOptions &options,
|
||||
// Attempt to JIT as well. But only do that if the main CPU JIT is enabled, in order to aid
|
||||
// debugging attempts - if the main JIT doesn't work, this one won't do any better, probably.
|
||||
if (jitCache) {
|
||||
// Compile doesn't check for space. We can't clear the cache here since other decoders point into it,
|
||||
// so when it's full (only seen with garbage display lists), new decoders use the interpreter.
|
||||
if (jitCache->GetSpaceLeft() < 4096) {
|
||||
WARN_LOG(Log::G3D, "Vertex decoder JIT cache full, using the interpreter for %08x", fmt_);
|
||||
return;
|
||||
}
|
||||
jitted_ = jitCache->Compile(*this, &jittedSize_);
|
||||
if (!jitted_) {
|
||||
WARN_LOG(Log::G3D, "Vertex decoder JIT failed! fmt = %08x (%s)", fmt_, GetString(SHADER_STRING_SHORT_DESC).c_str());
|
||||
|
||||
@@ -115,7 +115,8 @@ public:
|
||||
case GE_VTYPE_IDX_16BIT:
|
||||
return indices16[index];
|
||||
case GE_VTYPE_IDX_32BIT:
|
||||
return indices32[index];
|
||||
// The PSP supports 32-bit indices in name only: it ignores the upper 16 bits.
|
||||
return indices32[index] & 0xFFFF;
|
||||
default:
|
||||
return index;
|
||||
}
|
||||
|
||||
@@ -63,7 +63,9 @@ public:
|
||||
void Flush() override;
|
||||
|
||||
void FinishDeferred() {
|
||||
DecodeVerts(dec_, decoded_);
|
||||
// Decoding only the vertices isn't enough: the indices are still read from PSP memory at flush
|
||||
// time, and the game may change them once it regains control (#10095).
|
||||
Flush();
|
||||
}
|
||||
|
||||
void NotifyConfigChanged() override;
|
||||
|
||||
@@ -90,7 +90,7 @@ HRESULT SamplerCacheD3D11::GetOrCreateSampler(ID3D11Device *device, const Sample
|
||||
samp.AddressV = key.tClamp ? D3D11_TEXTURE_ADDRESS_CLAMP : D3D11_TEXTURE_ADDRESS_WRAP;
|
||||
samp.AddressW = samp.AddressU; // Mali benefits from all clamps being the same, and this one is irrelevant.
|
||||
if (key.aniso) {
|
||||
samp.MaxAnisotropy = (float)(1 << g_Config.iAnisotropyLevel);
|
||||
samp.MaxAnisotropy = (float)(1 << key.anisoLevel);
|
||||
} else {
|
||||
samp.MaxAnisotropy = 1.0f;
|
||||
}
|
||||
|
||||
+31
-13
@@ -126,13 +126,16 @@ void Recorder::DirtyDrawnVRAM() {
|
||||
}
|
||||
|
||||
bool Recorder::BeginRecording() {
|
||||
std::unique_lock<std::mutex> guard(callbackLock_);
|
||||
nextFrame = false;
|
||||
if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) {
|
||||
// Can't record a GE dump.
|
||||
// Can't record a GE dump. RecordNextFrame refuses this too.
|
||||
writeCallback = nullptr;
|
||||
return false;
|
||||
}
|
||||
|
||||
active = true;
|
||||
nextFrame = false;
|
||||
guard.unlock();
|
||||
lastTextures.clear();
|
||||
lastRenderTargets.clear();
|
||||
flipLastAction = gpuStats.totals.numFlips;
|
||||
@@ -180,6 +183,10 @@ Path Recorder::WriteRecording() {
|
||||
NOTICE_LOG(Log::G3D, "Recording filename: %s", filename.c_str());
|
||||
|
||||
FILE *fp = File::OpenCFile(filename, "wb");
|
||||
if (!fp) {
|
||||
ERROR_LOG(Log::G3D, "Failed to open '%s' for writing the recording", filename.c_str());
|
||||
return Path();
|
||||
}
|
||||
Header header{};
|
||||
memcpy(header.magic, HEADER_MAGIC, sizeof(header.magic));
|
||||
header.version = VERSION;
|
||||
@@ -574,14 +581,19 @@ void Recorder::EmitBezierSpline(u32 op) {
|
||||
}
|
||||
|
||||
bool Recorder::RecordNextFrame(const std::function<void(const Path &)> callback) {
|
||||
if (!nextFrame) {
|
||||
flipLastAction = gpuStats.totals.numFlips;
|
||||
flipFinishAt = -1;
|
||||
writeCallback = callback;
|
||||
nextFrame = true;
|
||||
return true;
|
||||
if (PSP_CoreParameter().fileType == IdentifiedFileType::PPSSPP_GE_DUMP) {
|
||||
return false;
|
||||
}
|
||||
return false;
|
||||
std::lock_guard<std::mutex> guard(callbackLock_);
|
||||
// Don't take over a recording in progress, it would get the wrong callback and end point.
|
||||
if (nextFrame || active) {
|
||||
return false;
|
||||
}
|
||||
flipLastAction = gpuStats.totals.numFlips;
|
||||
flipFinishAt = -1;
|
||||
writeCallback = callback;
|
||||
nextFrame = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
void Recorder::FinishRecording() {
|
||||
@@ -596,15 +608,21 @@ void Recorder::FinishRecording() {
|
||||
lastVRAM.clear();
|
||||
|
||||
NOTICE_LOG(Log::System, "Recording finished");
|
||||
active = false;
|
||||
flipLastAction = gpuStats.totals.numFlips;
|
||||
flipFinishAt = -1;
|
||||
lastEdramTrans = 0x400;
|
||||
|
||||
if (writeCallback) {
|
||||
writeCallback(filename);
|
||||
std::function<void(const Path &)> callback;
|
||||
{
|
||||
std::lock_guard<std::mutex> guard(callbackLock_);
|
||||
callback = std::move(writeCallback);
|
||||
writeCallback = nullptr;
|
||||
active = false;
|
||||
}
|
||||
|
||||
if (callback && !filename.empty()) {
|
||||
callback(filename);
|
||||
}
|
||||
writeCallback = nullptr;
|
||||
}
|
||||
|
||||
void Recorder::CheckEdramTrans() {
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
|
||||
#include <functional>
|
||||
#include <atomic>
|
||||
#include <mutex>
|
||||
#include <vector>
|
||||
#include <set>
|
||||
|
||||
@@ -50,7 +51,7 @@ public:
|
||||
}
|
||||
bool RecordNextFrame(const std::function<void(const Path &)> callback);
|
||||
void ClearCallback() {
|
||||
// Not super thread safe..
|
||||
std::lock_guard<std::mutex> guard(callbackLock_);
|
||||
writeCallback = nullptr;
|
||||
}
|
||||
|
||||
@@ -95,6 +96,8 @@ private:
|
||||
int flipFinishAt = -1;
|
||||
uint32_t lastEdramTrans = 0x400;
|
||||
std::function<void(const Path &)> writeCallback;
|
||||
// RecordNextFrame is called from other threads. Guards writeCallback, nextFrame and the writes to active.
|
||||
std::mutex callbackLock_;
|
||||
|
||||
std::vector<u8> pushbuf;
|
||||
std::vector<Command> commands;
|
||||
|
||||
@@ -229,6 +229,7 @@ void DrawEngineGLES::Flush() {
|
||||
numDrawVerts_ = 0;
|
||||
numDrawInds_ = 0;
|
||||
vertexCountInDrawCalls_ = 0;
|
||||
numVertsToDecode_ = 0;
|
||||
decodeVertsCounter_ = 0;
|
||||
decodeIndsCounter_ = 0;
|
||||
return;
|
||||
|
||||
@@ -144,7 +144,7 @@ GLRTexture *FragmentTestCacheGLES::CreateTestTexture(const GEComparison funcs[4]
|
||||
}
|
||||
|
||||
void FragmentTestCacheGLES::Clear(bool deleteThem) {
|
||||
if (deleteThem) {
|
||||
if (deleteThem && render_) {
|
||||
for (const auto &[_, v] : cache_) {
|
||||
render_->DeleteTexture(v.texture);
|
||||
}
|
||||
|
||||
@@ -68,7 +68,8 @@ public:
|
||||
void BindTestTexture(int slot);
|
||||
|
||||
void DeviceLost() {
|
||||
Clear(false);
|
||||
// Queue the deletes anyway, the deleter frees the GLRTexture objects and skips the GL calls if needed.
|
||||
Clear(true);
|
||||
render_ = nullptr;
|
||||
}
|
||||
void DeviceRestore(Draw::DrawContext *draw);
|
||||
|
||||
@@ -37,6 +37,8 @@ class IOFile;
|
||||
|
||||
class LinkedShader {
|
||||
public:
|
||||
LinkedShader(const LinkedShader &) = delete;
|
||||
LinkedShader &operator=(const LinkedShader &) = delete;
|
||||
LinkedShader(GLRenderManager *render, VShaderID VSID, Shader *vs, FShaderID FSID, Shader *fs, bool useHWTransform, bool preloading = false);
|
||||
~LinkedShader();
|
||||
|
||||
@@ -139,6 +141,8 @@ struct ShaderDescGLES {
|
||||
|
||||
class Shader {
|
||||
public:
|
||||
Shader(const Shader &) = delete;
|
||||
Shader &operator=(const Shader &) = delete;
|
||||
Shader(GLRenderManager *render, const char *code, const std::string &desc, const ShaderDescGLES ¶ms);
|
||||
~Shader();
|
||||
GLRShader *shader;
|
||||
|
||||
@@ -51,10 +51,10 @@ void TextureCacheGLES::SetFramebufferManager(FramebufferManagerGLES *fbManager)
|
||||
}
|
||||
|
||||
void TextureCacheGLES::ReleaseTexture(TexCacheEntry *entry, bool delete_them) {
|
||||
if (delete_them) {
|
||||
if (entry->textureName) {
|
||||
render_->DeleteTexture(entry->textureName);
|
||||
}
|
||||
// Delete even when !delete_them (device lost): the GLRTexture is a heap object that only the deleter
|
||||
// frees, and the deleter skips the GL calls itself once the context is gone.
|
||||
if (entry->textureName && render_) {
|
||||
render_->DeleteTexture(entry->textureName);
|
||||
}
|
||||
entry->textureName = nullptr;
|
||||
}
|
||||
|
||||
+24
-2
@@ -957,6 +957,7 @@ DLResult GPUCommon::ProcessDLQueue() {
|
||||
// Nothing to execute here, and leaving it at the head of the queue would block everything
|
||||
// behind it for good. Treat it like a list that ran into an error.
|
||||
ERROR_LOG(Log::G3D, "Display list %d has a bad pc %08x (state %d), dropping it", listIndex, list.pc, (int)list.state);
|
||||
CompleteFailedList(list);
|
||||
dlQueue.erase(std::remove(dlQueue.begin(), dlQueue.end(), listIndex), dlQueue.end());
|
||||
continue;
|
||||
}
|
||||
@@ -1045,7 +1046,8 @@ DLResult GPUCommon::ProcessDLQueue() {
|
||||
}
|
||||
break;
|
||||
case GPUSTATE_ERROR:
|
||||
// don't do anything - though dunno about error...
|
||||
// The list can't continue. It's removed from the queue below.
|
||||
CompleteFailedList(list);
|
||||
break;
|
||||
case GPUSTATE_STALL:
|
||||
// Resume work on this same display list later. The GE is still busy with what it has
|
||||
@@ -1390,6 +1392,20 @@ void GPUCommon::Execute_End(u32 op, u32 diff) {
|
||||
}
|
||||
}
|
||||
|
||||
// A list dropped on an error still has to complete like a finished one, or its ID is never freed and
|
||||
// sceGeListSync on it never returns.
|
||||
void GPUCommon::CompleteFailedList(DisplayList &list) {
|
||||
if (list.started && list.context.IsValid()) {
|
||||
gstate.Restore(list.context);
|
||||
ReapplyGfxState();
|
||||
list.started = false;
|
||||
}
|
||||
list.state = PSP_GE_DL_STATE_COMPLETED;
|
||||
list.waitUntilTicks = startingTicks + cyclesExecuted;
|
||||
busyTicks = std::max(busyTicks, list.waitUntilTicks);
|
||||
__GeTriggerSync(GPU_SYNC_LIST, list.id, list.waitUntilTicks);
|
||||
}
|
||||
|
||||
void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) {
|
||||
// Just resetting, nothing to check bounds for.
|
||||
const u32 count = op & 0xFFFF;
|
||||
@@ -1552,8 +1568,10 @@ void GPUCommon::FlushImm() {
|
||||
|
||||
bool changed = texturing != prevTexturing || cullEnable != prevCullEnable || dither != prevDither;
|
||||
changed = changed || prevShading != shading || prevFog != fog;
|
||||
// Always flush, even if the flags match: DispatchSubmitImm switches to through mode and a different
|
||||
// vertex decoder, which would otherwise apply to the draws already queued.
|
||||
Flush();
|
||||
if (changed) {
|
||||
Flush();
|
||||
gstate.antiAliasEnable = (GE_CMD_ANTIALIASENABLE << 24) | (int)antialias;
|
||||
gstate.shademodel = (GE_CMD_SHADEMODE << 24) | (int)shading;
|
||||
gstate.cullfaceEnable = (GE_CMD_CULLFACEENABLE << 24) | (int)cullEnable;
|
||||
@@ -2290,12 +2308,16 @@ bool GPUCommon::NeedsSlowInterpreter() const {
|
||||
void GPUCommon::ClearBreakNext() {
|
||||
breakNext_ = GPUDebug::BreakNext::NONE;
|
||||
breakAtCount_ = -1;
|
||||
// A step that never reached its target leaves these behind, and they'd trip unexpectedly later.
|
||||
breakpoints_.ClearTempBreakpoints();
|
||||
GPUStepping::ResumeFromStepping();
|
||||
}
|
||||
|
||||
void GPUCommon::SetBreakNext(GPUDebug::BreakNext next) {
|
||||
breakNext_ = next;
|
||||
breakAtCount_ = -1;
|
||||
// Drop the ones from a previous step that didn't get there, before adding this one's.
|
||||
breakpoints_.ClearTempBreakpoints();
|
||||
switch (next) {
|
||||
case GPUDebug::BreakNext::TEX:
|
||||
breakpoints_.AddTextureChangeTempBreakpoint();
|
||||
|
||||
@@ -56,6 +56,8 @@ inline bool IsTrianglePrim(GEPrimitiveType prim) {
|
||||
struct TransformStats;
|
||||
class GPUCommon {
|
||||
public:
|
||||
GPUCommon(const GPUCommon &) = delete;
|
||||
GPUCommon &operator=(const GPUCommon &) = delete;
|
||||
// The constructor might run on the loader thread.
|
||||
GPUCommon(GraphicsContext *gfxCtx, Draw::DrawContext *draw);
|
||||
virtual ~GPUCommon() = default;
|
||||
@@ -452,4 +454,5 @@ protected:
|
||||
private:
|
||||
void DoExecuteCall(u32 target);
|
||||
void PopDLQueue();
|
||||
void CompleteFailedList(DisplayList &list);
|
||||
};
|
||||
@@ -246,21 +246,32 @@ void BinManager::UpdateState() {
|
||||
if (newMaxTasks > MAX_POSSIBLE_TASKS)
|
||||
newMaxTasks = MAX_POSSIBLE_TASKS;
|
||||
// We don't want to overlap wrong, so flush any pending.
|
||||
bool flushed = false;
|
||||
if (maxTasks_ != newMaxTasks) {
|
||||
maxTasks_ = newMaxTasks;
|
||||
Flush("selfrender");
|
||||
flushed = true;
|
||||
}
|
||||
pendingOverlap_ = pendingOverlap_ || selfRender;
|
||||
|
||||
// Lastly, we have to check if we're newly writing depth we were texturing before.
|
||||
// This happens in Call of Duty (depth clear after depth texture), for example.
|
||||
if (!hadDepth && state.pixelID.depthWrite) {
|
||||
if (!flushed && !hadDepth && state.pixelID.depthWrite) {
|
||||
for (size_t i = 0; i < states_.Size(); ++i) {
|
||||
if (HasTextureWrite(states_.Peek(i))) {
|
||||
Flush("selfdepth");
|
||||
flushed = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (flushed) {
|
||||
// The flush forgot what this draw writes and reads, so record it again.
|
||||
MarkPendingWrites(state);
|
||||
MarkPendingReads(state);
|
||||
ClearDirty(SoftDirty::BINNER_RANGE);
|
||||
}
|
||||
pendingOverlap_ = pendingOverlap_ || selfRender;
|
||||
ClearDirty(SoftDirty::BINNER_OVERLAP);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,6 +57,8 @@ struct BinItem {
|
||||
|
||||
template <typename T, size_t N>
|
||||
struct BinQueue {
|
||||
BinQueue(const BinQueue &) = delete;
|
||||
BinQueue &operator=(const BinQueue &) = delete;
|
||||
BinQueue() {
|
||||
Reset();
|
||||
}
|
||||
@@ -185,6 +187,8 @@ struct BinDirtyRange {
|
||||
class StringWriter;
|
||||
class BinManager {
|
||||
public:
|
||||
BinManager(const BinManager &) = delete;
|
||||
BinManager &operator=(const BinManager &) = delete;
|
||||
BinManager();
|
||||
~BinManager();
|
||||
|
||||
|
||||
@@ -826,8 +826,8 @@ void SoftGPU::Execute_BlockTransferStart(u32 op, u32 diff) {
|
||||
|
||||
// Need to flush both source and target, so we overwrite properly.
|
||||
if (Memory::IsValidRange(src, srcSize) && Memory::IsValidRange(dst, dstSize)) {
|
||||
drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", false, src, srcStride, width * bpp, height);
|
||||
drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", true, dst, dstStride, width * bpp, height);
|
||||
drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", false, src, srcStride * bpp, width * bpp, height);
|
||||
drawEngine_->transformUnit.FlushIfOverlap(this, "blockxfer", true, dst, dstStride * bpp, width * bpp, height);
|
||||
} else {
|
||||
drawEngine_->transformUnit.Flush(this, "blockxfer_wrap");
|
||||
}
|
||||
|
||||
@@ -111,6 +111,8 @@ class StringWriter;
|
||||
|
||||
class TransformUnit {
|
||||
public:
|
||||
TransformUnit(const TransformUnit &) = delete;
|
||||
TransformUnit &operator=(const TransformUnit &) = delete;
|
||||
TransformUnit();
|
||||
~TransformUnit();
|
||||
|
||||
|
||||
@@ -550,6 +550,7 @@ void DrawEngineVulkan::ResetAfterSkippedDraw() {
|
||||
numDrawVerts_ = 0;
|
||||
numDrawInds_ = 0;
|
||||
vertexCountInDrawCalls_ = 0;
|
||||
numVertsToDecode_ = 0;
|
||||
decodeIndsCounter_ = 0;
|
||||
decodeVertsCounter_ = 0;
|
||||
gstate_c.vertexFullAlpha = true;
|
||||
|
||||
@@ -90,7 +90,7 @@ GPU_Vulkan::GPU_Vulkan(GraphicsContext *gfxCtx, Draw::DrawContext *draw)
|
||||
if (discID.size()) {
|
||||
File::CreateFullPath(GetSysDirectory(DIRECTORY_APP_CACHE));
|
||||
shaderCachePath_ = GetSysDirectory(DIRECTORY_APP_CACHE) / (discID + ".vkshadercache");
|
||||
LoadCache(shaderCachePath_);
|
||||
LoadCache(shaderCachePath_, true);
|
||||
}
|
||||
|
||||
InitDeviceObjects();
|
||||
@@ -101,7 +101,7 @@ void GPU_Vulkan::FinishInitOnMainThread() {
|
||||
framebufferManagerVulkan_->Init(msaaLevel_);
|
||||
}
|
||||
|
||||
void GPU_Vulkan::LoadCache(const Path &filename) {
|
||||
void GPU_Vulkan::LoadCache(const Path &filename, bool waitForPipelines) {
|
||||
_dbg_assert_(draw_);
|
||||
if (!g_Config.bShaderCache) {
|
||||
WARN_LOG(Log::G3D, "Shader cache disabled. Not loading.");
|
||||
@@ -134,13 +134,15 @@ void GPU_Vulkan::LoadCache(const Path &filename) {
|
||||
}
|
||||
fclose(f);
|
||||
|
||||
// Now, since we're on the loader thread, we can just block here until all pipelines are actually created.
|
||||
// This makes it so that the on-screen spinner keeps spinning until we are done.
|
||||
double start = time_now_d();
|
||||
VulkanRenderManager *rm = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER);
|
||||
int maxTasksSeen = rm->WaitForPipelines();
|
||||
double seconds = time_now_d() - start;
|
||||
INFO_LOG(Log::G3D, "Waited %0.1fms for at least %d pipeline tasks to finish compiling.", seconds * 1000.0, maxTasksSeen);
|
||||
if (waitForPipelines) {
|
||||
// Now, since we're on the loader thread, we can just block here until all pipelines are actually created.
|
||||
// This makes it so that the on-screen spinner keeps spinning until we are done.
|
||||
double start = time_now_d();
|
||||
VulkanRenderManager *rm = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER);
|
||||
int maxTasksSeen = rm->WaitForPipelines();
|
||||
double seconds = time_now_d() - start;
|
||||
INFO_LOG(Log::G3D, "Waited %0.1fms for at least %d pipeline tasks to finish compiling.", seconds * 1000.0, maxTasksSeen);
|
||||
}
|
||||
|
||||
if (!result) {
|
||||
WARN_LOG(Log::G3D, "Incompatible Vulkan pipeline cache - rebuilding.");
|
||||
@@ -402,6 +404,12 @@ void GPU_Vulkan::DeviceRestore(Draw::DrawContext *draw) {
|
||||
pipelineManager_->DeviceRestore(vulkan);
|
||||
|
||||
InitDeviceObjects();
|
||||
|
||||
// DeviceLost saved the cache and then threw everything away. Load it again, or the next save would
|
||||
// overwrite the file with only what gets drawn from now on.
|
||||
if (shaderCachePath_.Valid()) {
|
||||
LoadCache(shaderCachePath_, false);
|
||||
}
|
||||
}
|
||||
|
||||
void GPU_Vulkan::GetStats(StringWriter &w) {
|
||||
|
||||
@@ -69,7 +69,7 @@ private:
|
||||
void InitDeviceObjects();
|
||||
void DestroyDeviceObjects();
|
||||
|
||||
void LoadCache(const Path &filename);
|
||||
void LoadCache(const Path &filename, bool waitForPipelines);
|
||||
void SaveCache(const Path &filename);
|
||||
|
||||
FramebufferManagerVulkan *framebufferManagerVulkan_;
|
||||
|
||||
@@ -63,6 +63,9 @@ private:
|
||||
|
||||
// Simply wraps a Vulkan pipeline, providing some metadata.
|
||||
struct VulkanPipeline {
|
||||
VulkanPipeline() = default;
|
||||
VulkanPipeline(const VulkanPipeline &) = delete;
|
||||
VulkanPipeline &operator=(const VulkanPipeline &) = delete;
|
||||
~VulkanPipeline() {
|
||||
desc->Release();
|
||||
}
|
||||
|
||||
@@ -40,6 +40,8 @@ class VulkanPushPool;
|
||||
|
||||
class VulkanFragmentShader {
|
||||
public:
|
||||
VulkanFragmentShader(const VulkanFragmentShader &) = delete;
|
||||
VulkanFragmentShader &operator=(const VulkanFragmentShader &) = delete;
|
||||
VulkanFragmentShader(VulkanContext *vulkan, FShaderID id, FragmentShaderFlags flags, const char *code, SPIRVCache *cache);
|
||||
~VulkanFragmentShader();
|
||||
|
||||
@@ -63,6 +65,8 @@ protected:
|
||||
|
||||
class VulkanVertexShader {
|
||||
public:
|
||||
VulkanVertexShader(const VulkanVertexShader &) = delete;
|
||||
VulkanVertexShader &operator=(const VulkanVertexShader &) = delete;
|
||||
VulkanVertexShader(VulkanContext *vulkan, VShaderID id, VertexShaderFlags flags, const char *code, bool useHWTransform, SPIRVCache *cache);
|
||||
~VulkanVertexShader();
|
||||
|
||||
|
||||
@@ -143,7 +143,7 @@ VkSampler SamplerCache::GetOrCreateSampler(const SamplerCacheKey &key) {
|
||||
|
||||
if (key.aniso) {
|
||||
// Docs say the min of this value and the supported max are used.
|
||||
samp.maxAnisotropy = 1 << g_Config.iAnisotropyLevel;
|
||||
samp.maxAnisotropy = 1 << key.anisoLevel;
|
||||
samp.anisotropyEnable = true;
|
||||
} else {
|
||||
samp.maxAnisotropy = 1.0f;
|
||||
|
||||
@@ -37,6 +37,8 @@ class StringWriter;
|
||||
|
||||
class SamplerCache {
|
||||
public:
|
||||
SamplerCache(const SamplerCache &) = delete;
|
||||
SamplerCache &operator=(const SamplerCache &) = delete;
|
||||
SamplerCache(VulkanContext *vulkan) : vulkan_(vulkan), cache_(16) {}
|
||||
~SamplerCache();
|
||||
VkSampler GetOrCreateSampler(const SamplerCacheKey &key);
|
||||
|
||||
+1
-1
@@ -330,7 +330,7 @@ enum GEVertexType : uint32_t {
|
||||
GE_VTYPE_IDX_NONE = (0<<11),
|
||||
GE_VTYPE_IDX_8BIT = (1<<11),
|
||||
GE_VTYPE_IDX_16BIT = (2<<11),
|
||||
GE_VTYPE_IDX_32BIT = (3<<11),
|
||||
GE_VTYPE_IDX_32BIT = (3<<11), // In name only: the hardware ignores the upper 16 bits of each index.
|
||||
GE_VTYPE_IDX_MASK = (3<<11),
|
||||
#define GE_VTYPE_IDX_SHIFT 11
|
||||
};
|
||||
|
||||
@@ -96,9 +96,14 @@ void DrawFramebuffersWindow(ImConfig &cfg, FramebufferManagerCommon *framebuffer
|
||||
ImGui::SliderFloat("Scale", &cfg.fbViewerZoom, 0.5f, 16.0f, "%.2f", ImGuiSliderFlags_Logarithmic);
|
||||
|
||||
// Now, draw the image of the selected framebuffer.
|
||||
// Null without buffered rendering, or if creating it failed.
|
||||
Draw::Framebuffer *fb = vfbs[cfg.selectedFramebuffer]->fbo;
|
||||
ImTextureID texId = ImGui_ImplThin3d_AddFBAsTextureTemp(fb, Draw::Aspect::COLOR_BIT, ImGuiPipeline::TexturedOpaque);
|
||||
ImGui::Image(texId, ImVec2(fb->Width() * cfg.fbViewerZoom, fb->Height() * cfg.fbViewerZoom));
|
||||
if (fb) {
|
||||
ImTextureID texId = ImGui_ImplThin3d_AddFBAsTextureTemp(fb, Draw::Aspect::COLOR_BIT, ImGuiPipeline::TexturedOpaque);
|
||||
ImGui::Image(texId, ImVec2(fb->Width() * cfg.fbViewerZoom, fb->Height() * cfg.fbViewerZoom));
|
||||
} else {
|
||||
ImGui::TextUnformatted("(no framebuffer object)");
|
||||
}
|
||||
}
|
||||
|
||||
ImGui::End();
|
||||
@@ -602,7 +607,8 @@ ImGeReadbackViewer::~ImGeReadbackViewer() {
|
||||
}
|
||||
|
||||
VirtualFramebuffer *ImGeReadbackViewer::GetVFB(FramebufferManagerCommon *fbMan) const {
|
||||
return fbMan->GetExactVFB(gstate.getFrameBufAddress(), gstate.FrameBufStride(), gstate.FrameBufFormat());
|
||||
// fbMan is null with the software renderer.
|
||||
return fbMan ? fbMan->GetExactVFB(gstate.getFrameBufAddress(), gstate.FrameBufStride(), gstate.FrameBufFormat()) : nullptr;
|
||||
}
|
||||
|
||||
void ImGeReadbackViewer::DeviceLost() {
|
||||
|
||||
Reference in new issue
Block a user