Merge pull request #22390 from hrydgard/gpu-review-leftovers

Claude code review of GPU: Fix leftover findings
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-09-29 15:43:31 -06:00
commit ae460c9e1a
28 files changed
+289 -58

No files matched your search

+4
View File
@@ -283,6 +283,10 @@ protected:
bool everUsedEqualDepth_ = false;
bool everUsedExactEqualDepth_ = false;
// The draw context's invalidation callback is installed from BeginFrame, on the emu thread. The draw
// engine is created on the loader thread, while the UI thread may already be rendering and calling it.
bool invalidationCallbackInstalled_ = false;
// Vertex collector buffers
u8 *decoded_ = nullptr;
u16 *decIndex_ = nullptr;
+10 -7
View File
@@ -1557,14 +1557,19 @@ bool FramebufferManagerCommon::DrawFramebufferToOutput(const DisplayLayoutConfig
if (needBackBufferYSwap_) {
flags |= OutputFlags::BACKBUFFER_FLIPPED;
}
if (!useBufferedRendering_) {
// We're inside the backbuffer pass, where nothing else will draw this image.
flags |= OutputFlags::NO_POST_SHADER;
}
constexpr float u0 = 0.0f, u1 = 1.0f;
constexpr float v0 = 0.0f, v1 = 1.0f;
if (useBufferedRendering_) {
presentation_->UpdateUniforms(gpu->VideoIsPlaying());
presentation_->SourceTexture(pixelsTex, 480, 272);
presentation_->RunPostshaderPasses(config, flags, uvRotation, u0, v0, u1, v1);
presentation_->UpdateUniforms(gpu->VideoIsPlaying());
presentation_->SourceTexture(pixelsTex, 480, 272);
presentation_->RunPostshaderPasses(config, flags, uvRotation, u0, v0, u1, v1);
if (!useBufferedRendering_) {
presentation_->CopyToOutput(config);
}
// PresentationCommon sets all kinds of state, we can't rely on anything.
@@ -2863,9 +2868,7 @@ void FramebufferManagerCommon::NotifyBlockTransferAfter(u32 dstBasePtr, int dstS
if (isPrevDisplayBuffer || isDisplayBuffer) {
FlushBeforeCopy();
// HACK
if (DrawFramebufferToOutput(displayLayoutConfigCopy_, Memory::GetPointerUnchecked(dstBasePtr), dstStride, displayFormat_)) {
presentation_->CopyToOutput(displayLayoutConfigCopy_);
}
DrawFramebufferToOutput(displayLayoutConfigCopy_, Memory::GetPointerUnchecked(dstBasePtr), dstStride, displayFormat_);
return;
}
}
+2 -2
View File
@@ -680,7 +680,7 @@ void PresentationCommon::RunPostshaderPasses(const DisplayLayoutConfig &config,
bool useNearest = flags & OutputFlags::NEAREST;
bool useStereo = gstate_c.Use(GPU_USE_SIMPLE_STEREO_PERSPECTIVE) && stereoPipeline_ != nullptr; // TODO: Also check that the backend has support for it.
const bool usePostShader = usePostShader_ && !useStereo && !(flags & OutputFlags::RB_SWIZZLE);
const bool usePostShader = usePostShader_ && !useStereo && !(flags & (OutputFlags::RB_SWIZZLE | OutputFlags::NO_POST_SHADER));
const bool isFinalAtOutputResolution = usePostShader && postShaderFramebuffers_.size() < postShaderPipelines_.size();
int lastWidth = srcWidth_;
int lastHeight = srcHeight_;
@@ -897,7 +897,7 @@ void PresentationCommon::CopyToOutput(const DisplayLayoutConfig &config) {
bool useNearest = outputFlags_ & OutputFlags::NEAREST;
bool useStereo = gstate_c.Use(GPU_USE_SIMPLE_STEREO_PERSPECTIVE) && stereoPipeline_ != nullptr; // TODO: Also check that the backend has support for it.
const bool usePostShader = usePostShader_ && !useStereo && !(outputFlags_ & OutputFlags::RB_SWIZZLE);
const bool usePostShader = usePostShader_ && !useStereo && !(outputFlags_ & (OutputFlags::RB_SWIZZLE | OutputFlags::NO_POST_SHADER));
const bool isFinalAtOutputResolution = usePostShader && postShaderFramebuffers_.size() < postShaderPipelines_.size();
int lastWidth = srcWidth_;
int lastHeight = srcHeight_;
+1
View File
@@ -75,6 +75,7 @@ enum class OutputFlags {
RB_SWIZZLE = 0x0002,
BACKBUFFER_FLIPPED = 0x0004, // Viewport/scissor coordinates are y-flipped.
PILLARBOX = 0x0010, // Squeeze the image horizontally. Used for the DarkStalkers hack.
NO_POST_SHADER = 0x0020, // Draw straight to the bound target, as post shaders would need to bind their own.
};
ENUM_CLASS_BITOPS(OutputFlags);
+1 -1
View File
@@ -119,7 +119,7 @@ class ReplacedTexture;
// replacement (texture == nullptr).
struct ReplacedTextureRef {
ReplacedTexture *texture; // shortcut
std::string hashfiles; // key into the cache
std::string hashfiles; // key into levelCache_
};
// Metadata about a given texture level.
+13 -6
View File
@@ -68,6 +68,15 @@ struct SurfaceInfo {
break;
}
}
// Only the larger factor, so a lopsided patch keeps at least one step along its short axis.
void ReduceLargerTess() {
if (tess_u >= tess_v) {
tess_u--;
} else {
tess_v--;
}
}
};
struct BezierSurface : public SurfaceInfo {
@@ -78,9 +87,8 @@ struct BezierSurface : public SurfaceInfo {
void Init(int maxVertices) {
SurfaceInfo::BaseInit();
// Downsample until it fits, in case crazy tessellation factors are sent.
while ((tess_u + 1) * (tess_v + 1) * num_patches_u * num_patches_v > maxVertices) {
tess_u--;
tess_v--;
while ((tess_u + 1) * (tess_v + 1) * num_patches_u * num_patches_v > maxVertices && (tess_u > 1 || tess_v > 1)) {
ReduceLargerTess();
}
num_verts_per_patch = (tess_u + 1) * (tess_v + 1);
}
@@ -116,9 +124,8 @@ struct SplineSurface : public SurfaceInfo {
void Init(int maxVertices) {
SurfaceInfo::BaseInit();
// Downsample until it fits, in case crazy tessellation factors are sent.
while ((num_patches_u * tess_u + 1) * (num_patches_v * tess_v + 1) > maxVertices) {
tess_u--;
tess_v--;
while ((num_patches_u * tess_u + 1) * (num_patches_v * tess_v + 1) > maxVertices && (tess_u > 1 || tess_v > 1)) {
ReduceLargerTess();
}
num_vertices_u = num_patches_u * tess_u + 1;
}
+19 -11
View File
@@ -147,6 +147,10 @@ bool TextureReplacer::LoadIni(std::string *error, bool notify) {
hashranges_.clear();
filtering_.clear();
reducehashranges_.clear();
// These hold what the old ini said about each texture, including "no replacement" markers.
// Only references go, the textures stay in levelCache_.
cache_.clear();
savedCache_.clear();
ignoreAddress_ = false;
reduceHash_ = false;
@@ -670,15 +674,18 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
desc.cacheKey = replacementKey;
desc.forceFiltering = (TextureFiltering)0; // invalid value
// Hash ranges are per address even with ignoreAddress, since ComputeHash applies them.
LookupHashRange(replacementKey.Address(), w, h, &desc.newW, &desc.newH);
// cache_ stays keyed by the full key. ignoreAddress only affects finding the files.
ReplacementCacheKey lookupKey = replacementKey;
if (ignoreAddress_) {
replacementKey.ZeroAddress();
} else {
LookupHashRange(replacementKey.Address(), w, h, &desc.newW, &desc.newH);
lookupKey.ZeroAddress();
}
bool foundAlias = false;
bool ignored = false;
std::string hashfiles = LookupHashFile(replacementKey, &foundAlias, &ignored);
std::string hashfiles = LookupHashFile(lookupKey, &foundAlias, &ignored);
// Early-out for ignored textures, let's not bother even starting a thread task.
if (ignored) {
@@ -689,7 +696,7 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
return nullptr;
}
FindFiltering(replacementKey, &desc.forceFiltering);
FindFiltering(lookupKey, &desc.forceFiltering);
if (foundAlias) {
desc.logId = hashfiles;
@@ -707,12 +714,14 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
}
_dbg_assert_(!hashfiles.empty());
// OK, we might already have a matching texture, we use hashfiles as a key. Look it up in the level cache.
auto iter = levelCache_.find(hashfiles);
// OK, we might already have a matching texture. Textures sharing files can still differ in how
// they're scaled and filtered, so those go in the level cache key too.
std::string levelKey = StringFromFormat("%s#%dx%d>%dx%d#%d", hashfiles.c_str(), desc.w, desc.h, desc.newW, desc.newH, (int)desc.forceFiltering);
auto iter = levelCache_.find(levelKey);
if (iter != levelCache_.end()) {
// Insert an entry into the cache for faster lookup next time.
ReplacedTextureRef ref;
ref.hashfiles = hashfiles;
ref.hashfiles = levelKey;
ref.texture = iter->second;
cache_.emplace(std::make_pair(replacementKey, ref));
return iter->second;
@@ -725,12 +734,11 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
ReplacedTexture *texture = new ReplacedTexture(vfs_, desc);
ReplacedTextureRef ref;
ref.hashfiles = hashfiles;
ref.hashfiles = levelKey;
ref.texture = texture;
cache_.emplace(std::make_pair(replacementKey, ref));
// Also, insert the level in the level cache so we can look up by desc_->hashfiles again.
levelCache_.emplace(std::make_pair(hashfiles, texture));
levelCache_.emplace(std::make_pair(levelKey, texture));
return texture;
}
+13 -2
View File
@@ -46,7 +46,11 @@ public:
nextMapDiscard_ = true;
}
// Returns null if the buffer can't be mapped (after device removal, for example). Skip the draw then.
uint8_t *BeginPush(ID3D11DeviceContext *context, UINT *offset, size_t size, int align = 16) {
if (!buffer_) {
return nullptr;
}
D3D11_MAPPED_SUBRESOURCE map;
pos_ = (pos_ + align - 1) & ~(align - 1);
if (pos_ + size > size_) {
@@ -55,7 +59,10 @@ public:
pos_ = 0;
nextMapDiscard_ = true;
}
context->Map(buffer_.Get(), 0, nextMapDiscard_ ? D3D11_MAP_WRITE_DISCARD : D3D11_MAP_WRITE_NO_OVERWRITE, 0, &map);
if (FAILED(context->Map(buffer_.Get(), 0, nextMapDiscard_ ? D3D11_MAP_WRITE_DISCARD : D3D11_MAP_WRITE_NO_OVERWRITE, 0, &map))) {
return nullptr;
}
mapped_ = true;
nextMapDiscard_ = false;
*offset = (UINT)pos_;
uint8_t *retval = (uint8_t *)map.pData + pos_;
@@ -63,7 +70,10 @@ public:
return retval;
}
void EndPush(ID3D11DeviceContext *context) {
context->Unmap(buffer_.Get(), 0);
if (mapped_) {
context->Unmap(buffer_.Get(), 0);
mapped_ = false;
}
}
private:
@@ -71,6 +81,7 @@ private:
size_t pos_ = 0;
size_t size_;
bool nextMapDiscard_ = false;
bool mapped_ = false;
};
std::vector<uint8_t> CompileShaderToBytecodeD3D11(const char *code, size_t codeSize, const char *target, UINT flags, std::string *errorMessage = nullptr);
+26 -2
View File
@@ -83,13 +83,12 @@ DrawEngineD3D11::~DrawEngineD3D11() {
void DrawEngineD3D11::InitDeviceObjects() {
pushVerts_ = new PushBufferD3D11(device_, VERTEX_PUSH_SIZE, D3D11_BIND_VERTEX_BUFFER);
pushInds_ = new PushBufferD3D11(device_, INDEX_PUSH_SIZE, D3D11_BIND_INDEX_BUFFER);
draw_->SetInvalidationCallback(std::bind(&DrawEngineD3D11::Invalidate, this, std::placeholders::_1));
}
void DrawEngineD3D11::DestroyDeviceObjects() {
if (draw_) {
draw_->SetInvalidationCallback(InvalidationCallback());
invalidationCallbackInstalled_ = false;
}
ClearInputLayoutMap();
@@ -129,10 +128,19 @@ void DrawEngineD3D11::DestroyDeviceObjects() {
void DrawEngineD3D11::DeviceLost() {
DestroyDeviceObjects();
draw_ = nullptr;
device_ = nullptr;
context_ = nullptr;
device1_ = nullptr;
context1_ = nullptr;
}
void DrawEngineD3D11::DeviceRestore(Draw::DrawContext *draw) {
// The restored context can be a new device, so don't keep the old pointers.
draw_ = draw;
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
device1_ = (ID3D11Device1 *)draw->GetNativeObject(Draw::NativeObject::DEVICE_EX);
context1_ = (ID3D11DeviceContext1 *)draw->GetNativeObject(Draw::NativeObject::CONTEXT_EX);
InitDeviceObjects();
}
@@ -251,6 +259,10 @@ HRESULT DrawEngineD3D11::SetupDecFmtForDraw(D3D11VertexShader *vshader, const De
void DrawEngineD3D11::BeginFrame() {
DrawEngineCommon::BeginFrame();
if (!invalidationCallbackInstalled_) {
draw_->SetInvalidationCallback(std::bind(&DrawEngineD3D11::Invalidate, this, std::placeholders::_1));
invalidationCallbackInstalled_ = true;
}
pushVerts_->Reset();
pushInds_->Reset();
@@ -355,6 +367,9 @@ void DrawEngineD3D11::Flush() {
UINT vOffset;
int vSize = numDecodedVerts_ * dec_->GetDecVtxFmt().stride;
uint8_t *vptr = pushVerts_->BeginPush(context_, &vOffset, vSize);
if (!vptr) {
goto bail;
}
memcpy(vptr, decoded_, vSize);
pushVerts_->EndPush(context_);
ID3D11Buffer *buf = pushVerts_->Buf();
@@ -363,6 +378,9 @@ void DrawEngineD3D11::Flush() {
UINT iOffset;
int iSize = 2 * vertexCount;
uint8_t *iptr = pushInds_->BeginPush(context_, &iOffset, iSize);
if (!iptr) {
goto bail;
}
memcpy(iptr, decIndex_, iSize);
pushInds_->EndPush(context_);
context_->IASetIndexBuffer(pushInds_->Buf(), DXGI_FORMAT_R16_UINT, iOffset);
@@ -474,6 +492,9 @@ void DrawEngineD3D11::Flush() {
UINT vOffset = 0;
int vSize = result.drawVertexCount * stride;
uint8_t *vptr = pushVerts_->BeginPush(context_, &vOffset, vSize);
if (!vptr) {
goto bail;
}
memcpy(vptr, result.drawBuffer, vSize);
pushVerts_->EndPush(context_);
ID3D11Buffer *buf = pushVerts_->Buf();
@@ -481,6 +502,9 @@ void DrawEngineD3D11::Flush() {
UINT iOffset;
int iSize = sizeof(uint16_t) * result.drawIndexCount;
uint8_t *iptr = pushInds_->BeginPush(context_, &iOffset, iSize);
if (!iptr) {
goto bail;
}
memcpy(iptr, inds, iSize);
pushInds_->EndPush(context_);
context_->IASetIndexBuffer(pushInds_->Buf(), DXGI_FORMAT_R16_UINT, iOffset);
+3
View File
@@ -110,6 +110,9 @@ private:
PushBufferD3D11 *pushInds_ = nullptr;
// D3D11 state object caches. Previously had smart pointers but they were harder to deal with.
// These are never trimmed, although D3D11 allows only 4096 unique state objects per type and a
// failed create is fatal. That's deliberate: games use far fewer combinations, so it isn't an
// issue in practice, and eviction would cost more than it's worth.
DenseHashMap<uint64_t, ID3D11BlendState *> blendCache_;
DenseHashMap<uint64_t, ID3D11BlendState1 *> blendCache1_;
DenseHashMap<uint64_t, ID3D11DepthStencilState *> depthStencilCache_;
+3
View File
@@ -89,6 +89,9 @@ void GPU_D3D11::DeviceLost() {
}
void GPU_D3D11::DeviceRestore(Draw::DrawContext *draw) {
// The restored context can be a new device, so don't keep the old pointers.
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
GPUCommonHW::DeviceRestore(draw);
}
+15 -6
View File
@@ -111,10 +111,16 @@ void ShaderManagerD3D11::DestroyDeviceObjects() {
void ShaderManagerD3D11::DeviceLost() {
DestroyDeviceObjects();
draw_ = nullptr;
device_ = nullptr;
context_ = nullptr;
}
void ShaderManagerD3D11::DeviceRestore(Draw::DrawContext *draw) {
// The restored context can be a new device, so don't keep the old pointers.
draw_ = draw;
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
featureLevel_ = (D3D_FEATURE_LEVEL)draw->GetNativeObject(Draw::NativeObject::FEATURE_LEVEL);
InitDeviceObjects();
}
@@ -147,15 +153,18 @@ uint64_t ShaderManagerD3D11::UpdateUniforms(bool useBufferedRendering, bool pixe
D3D11_MAPPED_SUBRESOURCE map;
if (dirty & DIRTY_BASE_UNIFORMS) {
BaseUpdateUniforms(&ub_base, dirty, useBufferedRendering, pixelMapped);
context_->Map(push_base.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map);
memcpy(map.pData, &ub_base, sizeof(ub_base));
context_->Unmap(push_base.Get(), 0);
// Map fails after device removal, and map.pData is garbage then.
if (SUCCEEDED(context_->Map(push_base.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map))) {
memcpy(map.pData, &ub_base, sizeof(ub_base));
context_->Unmap(push_base.Get(), 0);
}
}
if (dirty & DIRTY_LIGHT_UNIFORMS) {
LightUpdateUniforms(&ub_lights, dirty);
context_->Map(push_lights.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map);
memcpy(map.pData, &ub_lights, sizeof(ub_lights));
context_->Unmap(push_lights.Get(), 0);
if (SUCCEEDED(context_->Map(push_lights.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map))) {
memcpy(map.pData, &ub_lights, sizeof(ub_lights));
context_->Unmap(push_lights.Get(), 0);
}
}
}
gstate_c.CleanUniforms();
+6 -1
View File
@@ -178,9 +178,14 @@ void TextureCacheD3D11::DeviceLost() {
TextureCacheCommon::DeviceLost();
DestroyDeviceObjects();
draw_ = nullptr;
device_ = nullptr;
context_ = nullptr;
}
void TextureCacheD3D11::DeviceRestore(Draw::DrawContext *draw) {
void TextureCacheD3D11::DeviceRestore(Draw::DrawContext *draw) {
// The restored context can be a new device, so don't keep the old pointers.
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
TextureCacheCommon::DeviceRestore(draw);
InitDeviceObjects();
}
+3
View File
@@ -319,6 +319,8 @@ void GPUBreakpoints::AddAddressBreakpoint(u32 addr, bool temp) {
}
void GPUBreakpoints::AddCmdBreakpoint(u8 cmd, bool temp) {
// Debuggers call this from their own threads, racing ClearTempBreakpoints on the emu thread.
std::lock_guard<std::mutex> guard(breaksLock);
if (temp) {
if (!breakCmds[cmd]) {
breakCmdsTemp[cmd] = true;
@@ -374,6 +376,7 @@ void GPUBreakpoints::AddRenderTargetBreakpoint(u32 addr, bool temp) {
}
void GPUBreakpoints::AddTextureChangeTempBreakpoint() {
std::lock_guard<std::mutex> guard(breaksLock);
textureChangeTemp = true;
hasBreakpoints_ = true;
}
+5 -1
View File
@@ -91,7 +91,6 @@ void DrawEngineGLES::InitDeviceObjects() {
entries.push_back({ ATTR_NORMAL, 1, GL_FLOAT, GL_FALSE, offsetof(TransformedVertex, fog) });
softwareInputLayout_ = render_->CreateInputLayout(entries, stride);
draw_->SetInvalidationCallback(std::bind(&DrawEngineGLES::Invalidate, this, std::placeholders::_1));
}
void DrawEngineGLES::DestroyDeviceObjects() {
@@ -99,6 +98,7 @@ void DrawEngineGLES::DestroyDeviceObjects() {
return;
}
draw_->SetInvalidationCallback(InvalidationCallback());
invalidationCallbackInstalled_ = false;
// Beware: this could be called twice in a row, sometimes.
for (int i = 0; i < GLRenderManager::MAX_INFLIGHT_FRAMES; i++) {
@@ -129,6 +129,10 @@ void DrawEngineGLES::ClearInputLayoutMap() {
void DrawEngineGLES::BeginFrame() {
DrawEngineCommon::BeginFrame();
if (!invalidationCallbackInstalled_) {
draw_->SetInvalidationCallback(std::bind(&DrawEngineGLES::Invalidate, this, std::placeholders::_1));
invalidationCallbackInstalled_ = true;
}
FrameData &frameData = frameData_[render_->GetCurFrame()];
frameData.pushIndex->Begin();
+5
View File
@@ -75,6 +75,11 @@ void GPUCommon::BeginHostFrame(const DisplayLayoutConfig &config) {
CheckConfigChanged(config);
CheckDisplayResized();
CheckRenderResized(config);
// After the resizes, which ask for the post shaders to be rebuilt at the new size.
if (framebufferManager_) {
framebufferManager_->CheckPostShaders(config);
}
}
void GPUCommon::EndHostFrame() {
-5
View File
@@ -416,11 +416,6 @@ void GPUCommonHW::CheckConfigChanged(const DisplayLayoutConfig &config) {
BuildReportingInfo();
configChanged_ = false;
}
// Check needed when running tests.
if (framebufferManager_) {
framebufferManager_->CheckPostShaders(config);
}
}
void GPUCommonHW::CheckDisplayResized() {
+19 -4
View File
@@ -209,6 +209,7 @@ void BinManager::UpdateState() {
const bool hadDepth = pendingWrites_[1].base != 0;
if (HasDirty(SoftDirty::BINNER_RANGE)) {
drawTargetAddr_ = gstate.getFrameBufAddress();
DrawingCoords scissorTL(gstate.getScissorX1(), gstate.getScissorY1());
DrawingCoords scissorBR(std::min(gstate.getScissorX2(), gstate.getRegionX2()), std::min(gstate.getScissorY2(), gstate.getRegionY2()));
ScreenCoords screenScissorTL = TransformUnit::DrawingToScreen(scissorTL, 0);
@@ -292,14 +293,14 @@ bool BinManager::HasTextureWrite(const RasterizerState &state) {
return false;
}
bool BinManager::IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item) {
bool BinManager::IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item) const {
if (item.type != BinItemType::SPRITE && item.type != BinItemType::RECT)
return false;
if (state.textureProj || state.maxTexLevel > 0)
return false;
// Only possible if the texture is 1:1.
if ((state.texaddr[0] & 0x0F1FFFFF) != (gstate.getFrameBufAddress() & 0x0F1FFFFF))
if ((state.texaddr[0] & 0x0F1FFFFF) != (drawTargetAddr_ & 0x0F1FFFFF))
return false;
int bufferPixelWidth = BufferFormatBytesPerPixel(state.pixelID.FBFormat());
int texturePixelWidth = textureBitsPerPixel[state.samplerID.texfmt] / 8;
@@ -361,8 +362,13 @@ void BinManager::MarkPendingWrites(const Rasterizer::RasterizerState &state) {
constexpr uint32_t mirrorMask = 0x041FFFFF;
const uint32_t bpp = state.pixelID.FBFormat() == GE_FORMAT_8888 ? 4 : 2;
pendingWrites_[0].Expand(gstate.getFrameBufAddress() & mirrorMask, bpp, gstate.FrameBufStride(), scissorTL, scissorBR);
if (state.pixelID.depthWrite)
if (state.pixelID.depthWrite) {
pendingWrites_[1].Expand(gstate.getDepthBufAddress() & mirrorMask, 2, gstate.DepthBufStride(), scissorTL, scissorBR);
} else if (gstate.isDepthTestEnabled() && !gstate.isModeClear()) {
// Testing without writing still reads the depth buffer, so a transfer into it has to wait.
const uint32_t depthAddr = gstate.getDepthBufAddress() & mirrorMask;
pendingReads_[depthAddr].Expand(depthAddr, 2, gstate.DepthBufStride(), scissorTL, scissorBR);
}
}
inline void BinDirtyRange::Expand(uint32_t newBase, uint32_t bpp, uint32_t stride, const DrawingCoords &tl, const DrawingCoords &br) {
@@ -584,8 +590,17 @@ void BinManager::Drain(bool flushing) {
}
void BinManager::Flush(const char *reason) {
if (queueRange_.x1 == 0x7FFFFFFF)
if (queueRange_.x1 == 0x7FFFFFFF) {
// Nothing queued, so nothing refers to the older states and CLUTs. Trim them anyway: callers
// flush because one of these rings is full, and push into it right after.
while (states_.Size() > 1) {
states_.SkipNext();
}
while (cluts_.Size() > 1) {
cluts_.SkipNext();
}
return;
}
double st = 0.0;
const bool collectDebugStats = g_coreCollectDebugStats;
+5 -2
View File
@@ -132,7 +132,7 @@ struct BinQueue {
}
bool Full() const {
return size_ == N - 1;
return size_ >= N - 1;
}
bool NearFull() const {
@@ -278,13 +278,16 @@ private:
const char *slowestFlushReason_ = nullptr;
double slowestFlushTime_ = 0.0;
int lastFlipstats_ = 0;
// The framebuffer the queued draws render to. A framebuffer change flushes first, so it's one for all
// of them, and during that flush gstate already has the new one.
u32 drawTargetAddr_ = 0;
int enqueues_ = 0;
int mostThreads_ = 0;
void MarkPendingReads(const Rasterizer::RasterizerState &state);
void MarkPendingWrites(const Rasterizer::RasterizerState &state);
bool HasTextureWrite(const Rasterizer::RasterizerState &state);
static bool IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item);
bool IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item) const;
void OptimizePendingStates(uint16_t first, uint16_t last);
BinCoords Scissor(BinCoords range);
BinCoords Range(const VertexData &v0, const VertexData &v1, const VertexData &v2);
+5
View File
@@ -506,11 +506,16 @@ bool RectangleFastPath(const VertexData &v0, const VertexData &v1, BinManager &b
if (g_needsClearAfterDialog) {
g_needsClearAfterDialog = false;
// Afterwards, we also need to clear the actual destination. Can do a fast rectfill.
// The binner's state was computed with texturing on, so recompute it around the sprite.
const SoftDirty texDirty = SoftDirty::SAMPLER_BASIC | SoftDirty::SAMPLER_TEXLIST | SoftDirty::RAST_TEX | SoftDirty::BINNER_OVERLAP;
gstate.textureMapEnable &= ~1;
binner.SetDirty(texDirty);
binner.UpdateState();
VertexData newV1 = v1;
newV1.color0 = 0xFF000000;
binner.AddSprite(v0, newV1);
gstate.textureMapEnable |= 1;
binner.SetDirty(texDirty);
}
return true;
} else {
+5 -1
View File
@@ -89,7 +89,6 @@ void DrawEngineVulkan::InitDeviceObjects() {
res = vkCreateSampler(device, &samp, nullptr, &nullSampler_);
_dbg_assert_(VK_SUCCESS == res);
draw_->SetInvalidationCallback(std::bind(&DrawEngineVulkan::Invalidate, this, std::placeholders::_1));
}
DrawEngineVulkan::~DrawEngineVulkan() {
@@ -106,6 +105,7 @@ void DrawEngineVulkan::DestroyDeviceObjects() {
VulkanRenderManager *renderManager = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER);
draw_->SetInvalidationCallback(InvalidationCallback());
invalidationCallbackInstalled_ = false;
pushUBO_ = nullptr;
@@ -143,6 +143,10 @@ void DrawEngineVulkan::DeviceRestore(Draw::DrawContext *draw) {
void DrawEngineVulkan::BeginFrame() {
DrawEngineCommon::BeginFrame();
if (!invalidationCallbackInstalled_) {
draw_->SetInvalidationCallback(std::bind(&DrawEngineVulkan::Invalidate, this, std::placeholders::_1));
invalidationCallbackInstalled_ = true;
}
lastPipeline_ = nullptr;
+12 -1
View File
@@ -858,7 +858,18 @@ void TextureCacheVulkan::BuildTexture(TexCacheEntry *const entry) {
VkImageView view = entry->vkTex->CreateViewForMip(i);
VK_PROFILE_BEGIN(vulkan, cmdInit, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
"Compute Upload: %dx%d->%dx%d", mipUnscaledWidth, mipUnscaledHeight, mipWidth, mipHeight);
ScaleBufferToImage(vulkan, cmdInit, view, texBuf, bufferOffset, srcSize, mipUnscaledWidth, mipUnscaledHeight, mipWidth, mipHeight);
if (!ScaleBufferToImage(vulkan, cmdInit, view, texBuf, bufferOffset, srcSize, mipUnscaledWidth, mipUnscaledHeight, mipWidth, mipHeight)) {
// Nothing was written to this level. Clear it rather than leave garbage in it.
WARN_LOG_ONCE(vkscalefail, Log::G3D, "Hardware texture scaling failed, clearing the texture");
VkClearColorValue clear{};
VkImageSubresourceRange range{ VK_IMAGE_ASPECT_COLOR_BIT, (uint32_t)i, 1, 0, 1 };
vkCmdClearColorImage(cmdInit, entry->vkTex->GetImage(), VK_IMAGE_LAYOUT_GENERAL, &clear, 1, &range);
// The barrier at the end of the upload waits on compute, so order the clear before that.
VkMemoryBarrier barrier{ VK_STRUCTURE_TYPE_MEMORY_BARRIER };
barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
vkCmdPipelineBarrier(cmdInit, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, 1, &barrier, 0, nullptr, 0, nullptr);
}
VK_PROFILE_END(vulkan, cmdInit, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT);
vulkan->Delete().QueueDeleteImageView(view);
} else {