mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #22390 from hrydgard/gpu-review-leftovers
Claude code review of GPU: Fix leftover findings
This commit is contained in:
28 files changed
+289
-58
No files matched your search
@@ -283,6 +283,10 @@ protected:
|
||||
bool everUsedEqualDepth_ = false;
|
||||
bool everUsedExactEqualDepth_ = false;
|
||||
|
||||
// The draw context's invalidation callback is installed from BeginFrame, on the emu thread. The draw
|
||||
// engine is created on the loader thread, while the UI thread may already be rendering and calling it.
|
||||
bool invalidationCallbackInstalled_ = false;
|
||||
|
||||
// Vertex collector buffers
|
||||
u8 *decoded_ = nullptr;
|
||||
u16 *decIndex_ = nullptr;
|
||||
|
||||
@@ -1557,14 +1557,19 @@ bool FramebufferManagerCommon::DrawFramebufferToOutput(const DisplayLayoutConfig
|
||||
if (needBackBufferYSwap_) {
|
||||
flags |= OutputFlags::BACKBUFFER_FLIPPED;
|
||||
}
|
||||
if (!useBufferedRendering_) {
|
||||
// We're inside the backbuffer pass, where nothing else will draw this image.
|
||||
flags |= OutputFlags::NO_POST_SHADER;
|
||||
}
|
||||
|
||||
constexpr float u0 = 0.0f, u1 = 1.0f;
|
||||
constexpr float v0 = 0.0f, v1 = 1.0f;
|
||||
|
||||
if (useBufferedRendering_) {
|
||||
presentation_->UpdateUniforms(gpu->VideoIsPlaying());
|
||||
presentation_->SourceTexture(pixelsTex, 480, 272);
|
||||
presentation_->RunPostshaderPasses(config, flags, uvRotation, u0, v0, u1, v1);
|
||||
presentation_->UpdateUniforms(gpu->VideoIsPlaying());
|
||||
presentation_->SourceTexture(pixelsTex, 480, 272);
|
||||
presentation_->RunPostshaderPasses(config, flags, uvRotation, u0, v0, u1, v1);
|
||||
if (!useBufferedRendering_) {
|
||||
presentation_->CopyToOutput(config);
|
||||
}
|
||||
|
||||
// PresentationCommon sets all kinds of state, we can't rely on anything.
|
||||
@@ -2863,9 +2868,7 @@ void FramebufferManagerCommon::NotifyBlockTransferAfter(u32 dstBasePtr, int dstS
|
||||
if (isPrevDisplayBuffer || isDisplayBuffer) {
|
||||
FlushBeforeCopy();
|
||||
// HACK
|
||||
if (DrawFramebufferToOutput(displayLayoutConfigCopy_, Memory::GetPointerUnchecked(dstBasePtr), dstStride, displayFormat_)) {
|
||||
presentation_->CopyToOutput(displayLayoutConfigCopy_);
|
||||
}
|
||||
DrawFramebufferToOutput(displayLayoutConfigCopy_, Memory::GetPointerUnchecked(dstBasePtr), dstStride, displayFormat_);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -680,7 +680,7 @@ void PresentationCommon::RunPostshaderPasses(const DisplayLayoutConfig &config,
|
||||
bool useNearest = flags & OutputFlags::NEAREST;
|
||||
bool useStereo = gstate_c.Use(GPU_USE_SIMPLE_STEREO_PERSPECTIVE) && stereoPipeline_ != nullptr; // TODO: Also check that the backend has support for it.
|
||||
|
||||
const bool usePostShader = usePostShader_ && !useStereo && !(flags & OutputFlags::RB_SWIZZLE);
|
||||
const bool usePostShader = usePostShader_ && !useStereo && !(flags & (OutputFlags::RB_SWIZZLE | OutputFlags::NO_POST_SHADER));
|
||||
const bool isFinalAtOutputResolution = usePostShader && postShaderFramebuffers_.size() < postShaderPipelines_.size();
|
||||
int lastWidth = srcWidth_;
|
||||
int lastHeight = srcHeight_;
|
||||
@@ -897,7 +897,7 @@ void PresentationCommon::CopyToOutput(const DisplayLayoutConfig &config) {
|
||||
bool useNearest = outputFlags_ & OutputFlags::NEAREST;
|
||||
bool useStereo = gstate_c.Use(GPU_USE_SIMPLE_STEREO_PERSPECTIVE) && stereoPipeline_ != nullptr; // TODO: Also check that the backend has support for it.
|
||||
|
||||
const bool usePostShader = usePostShader_ && !useStereo && !(outputFlags_ & OutputFlags::RB_SWIZZLE);
|
||||
const bool usePostShader = usePostShader_ && !useStereo && !(outputFlags_ & (OutputFlags::RB_SWIZZLE | OutputFlags::NO_POST_SHADER));
|
||||
const bool isFinalAtOutputResolution = usePostShader && postShaderFramebuffers_.size() < postShaderPipelines_.size();
|
||||
int lastWidth = srcWidth_;
|
||||
int lastHeight = srcHeight_;
|
||||
|
||||
@@ -75,6 +75,7 @@ enum class OutputFlags {
|
||||
RB_SWIZZLE = 0x0002,
|
||||
BACKBUFFER_FLIPPED = 0x0004, // Viewport/scissor coordinates are y-flipped.
|
||||
PILLARBOX = 0x0010, // Squeeze the image horizontally. Used for the DarkStalkers hack.
|
||||
NO_POST_SHADER = 0x0020, // Draw straight to the bound target, as post shaders would need to bind their own.
|
||||
};
|
||||
ENUM_CLASS_BITOPS(OutputFlags);
|
||||
|
||||
|
||||
@@ -119,7 +119,7 @@ class ReplacedTexture;
|
||||
// replacement (texture == nullptr).
|
||||
struct ReplacedTextureRef {
|
||||
ReplacedTexture *texture; // shortcut
|
||||
std::string hashfiles; // key into the cache
|
||||
std::string hashfiles; // key into levelCache_
|
||||
};
|
||||
|
||||
// Metadata about a given texture level.
|
||||
|
||||
@@ -68,6 +68,15 @@ struct SurfaceInfo {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Only the larger factor, so a lopsided patch keeps at least one step along its short axis.
|
||||
void ReduceLargerTess() {
|
||||
if (tess_u >= tess_v) {
|
||||
tess_u--;
|
||||
} else {
|
||||
tess_v--;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
struct BezierSurface : public SurfaceInfo {
|
||||
@@ -78,9 +87,8 @@ struct BezierSurface : public SurfaceInfo {
|
||||
void Init(int maxVertices) {
|
||||
SurfaceInfo::BaseInit();
|
||||
// Downsample until it fits, in case crazy tessellation factors are sent.
|
||||
while ((tess_u + 1) * (tess_v + 1) * num_patches_u * num_patches_v > maxVertices) {
|
||||
tess_u--;
|
||||
tess_v--;
|
||||
while ((tess_u + 1) * (tess_v + 1) * num_patches_u * num_patches_v > maxVertices && (tess_u > 1 || tess_v > 1)) {
|
||||
ReduceLargerTess();
|
||||
}
|
||||
num_verts_per_patch = (tess_u + 1) * (tess_v + 1);
|
||||
}
|
||||
@@ -116,9 +124,8 @@ struct SplineSurface : public SurfaceInfo {
|
||||
void Init(int maxVertices) {
|
||||
SurfaceInfo::BaseInit();
|
||||
// Downsample until it fits, in case crazy tessellation factors are sent.
|
||||
while ((num_patches_u * tess_u + 1) * (num_patches_v * tess_v + 1) > maxVertices) {
|
||||
tess_u--;
|
||||
tess_v--;
|
||||
while ((num_patches_u * tess_u + 1) * (num_patches_v * tess_v + 1) > maxVertices && (tess_u > 1 || tess_v > 1)) {
|
||||
ReduceLargerTess();
|
||||
}
|
||||
num_vertices_u = num_patches_u * tess_u + 1;
|
||||
}
|
||||
|
||||
@@ -147,6 +147,10 @@ bool TextureReplacer::LoadIni(std::string *error, bool notify) {
|
||||
hashranges_.clear();
|
||||
filtering_.clear();
|
||||
reducehashranges_.clear();
|
||||
// These hold what the old ini said about each texture, including "no replacement" markers.
|
||||
// Only references go, the textures stay in levelCache_.
|
||||
cache_.clear();
|
||||
savedCache_.clear();
|
||||
|
||||
ignoreAddress_ = false;
|
||||
reduceHash_ = false;
|
||||
@@ -670,15 +674,18 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
|
||||
desc.cacheKey = replacementKey;
|
||||
desc.forceFiltering = (TextureFiltering)0; // invalid value
|
||||
|
||||
// Hash ranges are per address even with ignoreAddress, since ComputeHash applies them.
|
||||
LookupHashRange(replacementKey.Address(), w, h, &desc.newW, &desc.newH);
|
||||
|
||||
// cache_ stays keyed by the full key. ignoreAddress only affects finding the files.
|
||||
ReplacementCacheKey lookupKey = replacementKey;
|
||||
if (ignoreAddress_) {
|
||||
replacementKey.ZeroAddress();
|
||||
} else {
|
||||
LookupHashRange(replacementKey.Address(), w, h, &desc.newW, &desc.newH);
|
||||
lookupKey.ZeroAddress();
|
||||
}
|
||||
|
||||
bool foundAlias = false;
|
||||
bool ignored = false;
|
||||
std::string hashfiles = LookupHashFile(replacementKey, &foundAlias, &ignored);
|
||||
std::string hashfiles = LookupHashFile(lookupKey, &foundAlias, &ignored);
|
||||
|
||||
// Early-out for ignored textures, let's not bother even starting a thread task.
|
||||
if (ignored) {
|
||||
@@ -689,7 +696,7 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
FindFiltering(replacementKey, &desc.forceFiltering);
|
||||
FindFiltering(lookupKey, &desc.forceFiltering);
|
||||
|
||||
if (foundAlias) {
|
||||
desc.logId = hashfiles;
|
||||
@@ -707,12 +714,14 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
|
||||
}
|
||||
|
||||
_dbg_assert_(!hashfiles.empty());
|
||||
// OK, we might already have a matching texture, we use hashfiles as a key. Look it up in the level cache.
|
||||
auto iter = levelCache_.find(hashfiles);
|
||||
// OK, we might already have a matching texture. Textures sharing files can still differ in how
|
||||
// they're scaled and filtered, so those go in the level cache key too.
|
||||
std::string levelKey = StringFromFormat("%s#%dx%d>%dx%d#%d", hashfiles.c_str(), desc.w, desc.h, desc.newW, desc.newH, (int)desc.forceFiltering);
|
||||
auto iter = levelCache_.find(levelKey);
|
||||
if (iter != levelCache_.end()) {
|
||||
// Insert an entry into the cache for faster lookup next time.
|
||||
ReplacedTextureRef ref;
|
||||
ref.hashfiles = hashfiles;
|
||||
ref.hashfiles = levelKey;
|
||||
ref.texture = iter->second;
|
||||
cache_.emplace(std::make_pair(replacementKey, ref));
|
||||
return iter->second;
|
||||
@@ -725,12 +734,11 @@ ReplacedTexture *TextureReplacer::FindReplacement(ReplacementCacheKey replacemen
|
||||
ReplacedTexture *texture = new ReplacedTexture(vfs_, desc);
|
||||
|
||||
ReplacedTextureRef ref;
|
||||
ref.hashfiles = hashfiles;
|
||||
ref.hashfiles = levelKey;
|
||||
ref.texture = texture;
|
||||
cache_.emplace(std::make_pair(replacementKey, ref));
|
||||
|
||||
// Also, insert the level in the level cache so we can look up by desc_->hashfiles again.
|
||||
levelCache_.emplace(std::make_pair(hashfiles, texture));
|
||||
levelCache_.emplace(std::make_pair(levelKey, texture));
|
||||
return texture;
|
||||
}
|
||||
|
||||
|
||||
+13
-2
@@ -46,7 +46,11 @@ public:
|
||||
nextMapDiscard_ = true;
|
||||
}
|
||||
|
||||
// Returns null if the buffer can't be mapped (after device removal, for example). Skip the draw then.
|
||||
uint8_t *BeginPush(ID3D11DeviceContext *context, UINT *offset, size_t size, int align = 16) {
|
||||
if (!buffer_) {
|
||||
return nullptr;
|
||||
}
|
||||
D3D11_MAPPED_SUBRESOURCE map;
|
||||
pos_ = (pos_ + align - 1) & ~(align - 1);
|
||||
if (pos_ + size > size_) {
|
||||
@@ -55,7 +59,10 @@ public:
|
||||
pos_ = 0;
|
||||
nextMapDiscard_ = true;
|
||||
}
|
||||
context->Map(buffer_.Get(), 0, nextMapDiscard_ ? D3D11_MAP_WRITE_DISCARD : D3D11_MAP_WRITE_NO_OVERWRITE, 0, &map);
|
||||
if (FAILED(context->Map(buffer_.Get(), 0, nextMapDiscard_ ? D3D11_MAP_WRITE_DISCARD : D3D11_MAP_WRITE_NO_OVERWRITE, 0, &map))) {
|
||||
return nullptr;
|
||||
}
|
||||
mapped_ = true;
|
||||
nextMapDiscard_ = false;
|
||||
*offset = (UINT)pos_;
|
||||
uint8_t *retval = (uint8_t *)map.pData + pos_;
|
||||
@@ -63,7 +70,10 @@ public:
|
||||
return retval;
|
||||
}
|
||||
void EndPush(ID3D11DeviceContext *context) {
|
||||
context->Unmap(buffer_.Get(), 0);
|
||||
if (mapped_) {
|
||||
context->Unmap(buffer_.Get(), 0);
|
||||
mapped_ = false;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -71,6 +81,7 @@ private:
|
||||
size_t pos_ = 0;
|
||||
size_t size_;
|
||||
bool nextMapDiscard_ = false;
|
||||
bool mapped_ = false;
|
||||
};
|
||||
|
||||
std::vector<uint8_t> CompileShaderToBytecodeD3D11(const char *code, size_t codeSize, const char *target, UINT flags, std::string *errorMessage = nullptr);
|
||||
|
||||
@@ -83,13 +83,12 @@ DrawEngineD3D11::~DrawEngineD3D11() {
|
||||
void DrawEngineD3D11::InitDeviceObjects() {
|
||||
pushVerts_ = new PushBufferD3D11(device_, VERTEX_PUSH_SIZE, D3D11_BIND_VERTEX_BUFFER);
|
||||
pushInds_ = new PushBufferD3D11(device_, INDEX_PUSH_SIZE, D3D11_BIND_INDEX_BUFFER);
|
||||
|
||||
draw_->SetInvalidationCallback(std::bind(&DrawEngineD3D11::Invalidate, this, std::placeholders::_1));
|
||||
}
|
||||
|
||||
void DrawEngineD3D11::DestroyDeviceObjects() {
|
||||
if (draw_) {
|
||||
draw_->SetInvalidationCallback(InvalidationCallback());
|
||||
invalidationCallbackInstalled_ = false;
|
||||
}
|
||||
|
||||
ClearInputLayoutMap();
|
||||
@@ -129,10 +128,19 @@ void DrawEngineD3D11::DestroyDeviceObjects() {
|
||||
void DrawEngineD3D11::DeviceLost() {
|
||||
DestroyDeviceObjects();
|
||||
draw_ = nullptr;
|
||||
device_ = nullptr;
|
||||
context_ = nullptr;
|
||||
device1_ = nullptr;
|
||||
context1_ = nullptr;
|
||||
}
|
||||
|
||||
void DrawEngineD3D11::DeviceRestore(Draw::DrawContext *draw) {
|
||||
// The restored context can be a new device, so don't keep the old pointers.
|
||||
draw_ = draw;
|
||||
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
|
||||
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
|
||||
device1_ = (ID3D11Device1 *)draw->GetNativeObject(Draw::NativeObject::DEVICE_EX);
|
||||
context1_ = (ID3D11DeviceContext1 *)draw->GetNativeObject(Draw::NativeObject::CONTEXT_EX);
|
||||
InitDeviceObjects();
|
||||
}
|
||||
|
||||
@@ -251,6 +259,10 @@ HRESULT DrawEngineD3D11::SetupDecFmtForDraw(D3D11VertexShader *vshader, const De
|
||||
|
||||
void DrawEngineD3D11::BeginFrame() {
|
||||
DrawEngineCommon::BeginFrame();
|
||||
if (!invalidationCallbackInstalled_) {
|
||||
draw_->SetInvalidationCallback(std::bind(&DrawEngineD3D11::Invalidate, this, std::placeholders::_1));
|
||||
invalidationCallbackInstalled_ = true;
|
||||
}
|
||||
|
||||
pushVerts_->Reset();
|
||||
pushInds_->Reset();
|
||||
@@ -355,6 +367,9 @@ void DrawEngineD3D11::Flush() {
|
||||
UINT vOffset;
|
||||
int vSize = numDecodedVerts_ * dec_->GetDecVtxFmt().stride;
|
||||
uint8_t *vptr = pushVerts_->BeginPush(context_, &vOffset, vSize);
|
||||
if (!vptr) {
|
||||
goto bail;
|
||||
}
|
||||
memcpy(vptr, decoded_, vSize);
|
||||
pushVerts_->EndPush(context_);
|
||||
ID3D11Buffer *buf = pushVerts_->Buf();
|
||||
@@ -363,6 +378,9 @@ void DrawEngineD3D11::Flush() {
|
||||
UINT iOffset;
|
||||
int iSize = 2 * vertexCount;
|
||||
uint8_t *iptr = pushInds_->BeginPush(context_, &iOffset, iSize);
|
||||
if (!iptr) {
|
||||
goto bail;
|
||||
}
|
||||
memcpy(iptr, decIndex_, iSize);
|
||||
pushInds_->EndPush(context_);
|
||||
context_->IASetIndexBuffer(pushInds_->Buf(), DXGI_FORMAT_R16_UINT, iOffset);
|
||||
@@ -474,6 +492,9 @@ void DrawEngineD3D11::Flush() {
|
||||
UINT vOffset = 0;
|
||||
int vSize = result.drawVertexCount * stride;
|
||||
uint8_t *vptr = pushVerts_->BeginPush(context_, &vOffset, vSize);
|
||||
if (!vptr) {
|
||||
goto bail;
|
||||
}
|
||||
memcpy(vptr, result.drawBuffer, vSize);
|
||||
pushVerts_->EndPush(context_);
|
||||
ID3D11Buffer *buf = pushVerts_->Buf();
|
||||
@@ -481,6 +502,9 @@ void DrawEngineD3D11::Flush() {
|
||||
UINT iOffset;
|
||||
int iSize = sizeof(uint16_t) * result.drawIndexCount;
|
||||
uint8_t *iptr = pushInds_->BeginPush(context_, &iOffset, iSize);
|
||||
if (!iptr) {
|
||||
goto bail;
|
||||
}
|
||||
memcpy(iptr, inds, iSize);
|
||||
pushInds_->EndPush(context_);
|
||||
context_->IASetIndexBuffer(pushInds_->Buf(), DXGI_FORMAT_R16_UINT, iOffset);
|
||||
|
||||
@@ -110,6 +110,9 @@ private:
|
||||
PushBufferD3D11 *pushInds_ = nullptr;
|
||||
|
||||
// D3D11 state object caches. Previously had smart pointers but they were harder to deal with.
|
||||
// These are never trimmed, although D3D11 allows only 4096 unique state objects per type and a
|
||||
// failed create is fatal. That's deliberate: games use far fewer combinations, so it isn't an
|
||||
// issue in practice, and eviction would cost more than it's worth.
|
||||
DenseHashMap<uint64_t, ID3D11BlendState *> blendCache_;
|
||||
DenseHashMap<uint64_t, ID3D11BlendState1 *> blendCache1_;
|
||||
DenseHashMap<uint64_t, ID3D11DepthStencilState *> depthStencilCache_;
|
||||
|
||||
@@ -89,6 +89,9 @@ void GPU_D3D11::DeviceLost() {
|
||||
}
|
||||
|
||||
void GPU_D3D11::DeviceRestore(Draw::DrawContext *draw) {
|
||||
// The restored context can be a new device, so don't keep the old pointers.
|
||||
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
|
||||
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
|
||||
GPUCommonHW::DeviceRestore(draw);
|
||||
}
|
||||
|
||||
|
||||
@@ -111,10 +111,16 @@ void ShaderManagerD3D11::DestroyDeviceObjects() {
|
||||
void ShaderManagerD3D11::DeviceLost() {
|
||||
DestroyDeviceObjects();
|
||||
draw_ = nullptr;
|
||||
device_ = nullptr;
|
||||
context_ = nullptr;
|
||||
}
|
||||
|
||||
void ShaderManagerD3D11::DeviceRestore(Draw::DrawContext *draw) {
|
||||
// The restored context can be a new device, so don't keep the old pointers.
|
||||
draw_ = draw;
|
||||
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
|
||||
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
|
||||
featureLevel_ = (D3D_FEATURE_LEVEL)draw->GetNativeObject(Draw::NativeObject::FEATURE_LEVEL);
|
||||
InitDeviceObjects();
|
||||
}
|
||||
|
||||
@@ -147,15 +153,18 @@ uint64_t ShaderManagerD3D11::UpdateUniforms(bool useBufferedRendering, bool pixe
|
||||
D3D11_MAPPED_SUBRESOURCE map;
|
||||
if (dirty & DIRTY_BASE_UNIFORMS) {
|
||||
BaseUpdateUniforms(&ub_base, dirty, useBufferedRendering, pixelMapped);
|
||||
context_->Map(push_base.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map);
|
||||
memcpy(map.pData, &ub_base, sizeof(ub_base));
|
||||
context_->Unmap(push_base.Get(), 0);
|
||||
// Map fails after device removal, and map.pData is garbage then.
|
||||
if (SUCCEEDED(context_->Map(push_base.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map))) {
|
||||
memcpy(map.pData, &ub_base, sizeof(ub_base));
|
||||
context_->Unmap(push_base.Get(), 0);
|
||||
}
|
||||
}
|
||||
if (dirty & DIRTY_LIGHT_UNIFORMS) {
|
||||
LightUpdateUniforms(&ub_lights, dirty);
|
||||
context_->Map(push_lights.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map);
|
||||
memcpy(map.pData, &ub_lights, sizeof(ub_lights));
|
||||
context_->Unmap(push_lights.Get(), 0);
|
||||
if (SUCCEEDED(context_->Map(push_lights.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map))) {
|
||||
memcpy(map.pData, &ub_lights, sizeof(ub_lights));
|
||||
context_->Unmap(push_lights.Get(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
gstate_c.CleanUniforms();
|
||||
|
||||
@@ -178,9 +178,14 @@ void TextureCacheD3D11::DeviceLost() {
|
||||
TextureCacheCommon::DeviceLost();
|
||||
DestroyDeviceObjects();
|
||||
draw_ = nullptr;
|
||||
device_ = nullptr;
|
||||
context_ = nullptr;
|
||||
}
|
||||
|
||||
void TextureCacheD3D11::DeviceRestore(Draw::DrawContext *draw) {
|
||||
void TextureCacheD3D11::DeviceRestore(Draw::DrawContext *draw) {
|
||||
// The restored context can be a new device, so don't keep the old pointers.
|
||||
device_ = (ID3D11Device *)draw->GetNativeObject(Draw::NativeObject::DEVICE);
|
||||
context_ = (ID3D11DeviceContext *)draw->GetNativeObject(Draw::NativeObject::CONTEXT);
|
||||
TextureCacheCommon::DeviceRestore(draw);
|
||||
InitDeviceObjects();
|
||||
}
|
||||
|
||||
@@ -319,6 +319,8 @@ void GPUBreakpoints::AddAddressBreakpoint(u32 addr, bool temp) {
|
||||
}
|
||||
|
||||
void GPUBreakpoints::AddCmdBreakpoint(u8 cmd, bool temp) {
|
||||
// Debuggers call this from their own threads, racing ClearTempBreakpoints on the emu thread.
|
||||
std::lock_guard<std::mutex> guard(breaksLock);
|
||||
if (temp) {
|
||||
if (!breakCmds[cmd]) {
|
||||
breakCmdsTemp[cmd] = true;
|
||||
@@ -374,6 +376,7 @@ void GPUBreakpoints::AddRenderTargetBreakpoint(u32 addr, bool temp) {
|
||||
}
|
||||
|
||||
void GPUBreakpoints::AddTextureChangeTempBreakpoint() {
|
||||
std::lock_guard<std::mutex> guard(breaksLock);
|
||||
textureChangeTemp = true;
|
||||
hasBreakpoints_ = true;
|
||||
}
|
||||
|
||||
@@ -91,7 +91,6 @@ void DrawEngineGLES::InitDeviceObjects() {
|
||||
entries.push_back({ ATTR_NORMAL, 1, GL_FLOAT, GL_FALSE, offsetof(TransformedVertex, fog) });
|
||||
softwareInputLayout_ = render_->CreateInputLayout(entries, stride);
|
||||
|
||||
draw_->SetInvalidationCallback(std::bind(&DrawEngineGLES::Invalidate, this, std::placeholders::_1));
|
||||
}
|
||||
|
||||
void DrawEngineGLES::DestroyDeviceObjects() {
|
||||
@@ -99,6 +98,7 @@ void DrawEngineGLES::DestroyDeviceObjects() {
|
||||
return;
|
||||
}
|
||||
draw_->SetInvalidationCallback(InvalidationCallback());
|
||||
invalidationCallbackInstalled_ = false;
|
||||
|
||||
// Beware: this could be called twice in a row, sometimes.
|
||||
for (int i = 0; i < GLRenderManager::MAX_INFLIGHT_FRAMES; i++) {
|
||||
@@ -129,6 +129,10 @@ void DrawEngineGLES::ClearInputLayoutMap() {
|
||||
|
||||
void DrawEngineGLES::BeginFrame() {
|
||||
DrawEngineCommon::BeginFrame();
|
||||
if (!invalidationCallbackInstalled_) {
|
||||
draw_->SetInvalidationCallback(std::bind(&DrawEngineGLES::Invalidate, this, std::placeholders::_1));
|
||||
invalidationCallbackInstalled_ = true;
|
||||
}
|
||||
|
||||
FrameData &frameData = frameData_[render_->GetCurFrame()];
|
||||
frameData.pushIndex->Begin();
|
||||
|
||||
@@ -75,6 +75,11 @@ void GPUCommon::BeginHostFrame(const DisplayLayoutConfig &config) {
|
||||
CheckConfigChanged(config);
|
||||
CheckDisplayResized();
|
||||
CheckRenderResized(config);
|
||||
|
||||
// After the resizes, which ask for the post shaders to be rebuilt at the new size.
|
||||
if (framebufferManager_) {
|
||||
framebufferManager_->CheckPostShaders(config);
|
||||
}
|
||||
}
|
||||
|
||||
void GPUCommon::EndHostFrame() {
|
||||
|
||||
@@ -416,11 +416,6 @@ void GPUCommonHW::CheckConfigChanged(const DisplayLayoutConfig &config) {
|
||||
BuildReportingInfo();
|
||||
configChanged_ = false;
|
||||
}
|
||||
|
||||
// Check needed when running tests.
|
||||
if (framebufferManager_) {
|
||||
framebufferManager_->CheckPostShaders(config);
|
||||
}
|
||||
}
|
||||
|
||||
void GPUCommonHW::CheckDisplayResized() {
|
||||
|
||||
@@ -209,6 +209,7 @@ void BinManager::UpdateState() {
|
||||
const bool hadDepth = pendingWrites_[1].base != 0;
|
||||
|
||||
if (HasDirty(SoftDirty::BINNER_RANGE)) {
|
||||
drawTargetAddr_ = gstate.getFrameBufAddress();
|
||||
DrawingCoords scissorTL(gstate.getScissorX1(), gstate.getScissorY1());
|
||||
DrawingCoords scissorBR(std::min(gstate.getScissorX2(), gstate.getRegionX2()), std::min(gstate.getScissorY2(), gstate.getRegionY2()));
|
||||
ScreenCoords screenScissorTL = TransformUnit::DrawingToScreen(scissorTL, 0);
|
||||
@@ -292,14 +293,14 @@ bool BinManager::HasTextureWrite(const RasterizerState &state) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool BinManager::IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item) {
|
||||
bool BinManager::IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item) const {
|
||||
if (item.type != BinItemType::SPRITE && item.type != BinItemType::RECT)
|
||||
return false;
|
||||
if (state.textureProj || state.maxTexLevel > 0)
|
||||
return false;
|
||||
|
||||
// Only possible if the texture is 1:1.
|
||||
if ((state.texaddr[0] & 0x0F1FFFFF) != (gstate.getFrameBufAddress() & 0x0F1FFFFF))
|
||||
if ((state.texaddr[0] & 0x0F1FFFFF) != (drawTargetAddr_ & 0x0F1FFFFF))
|
||||
return false;
|
||||
int bufferPixelWidth = BufferFormatBytesPerPixel(state.pixelID.FBFormat());
|
||||
int texturePixelWidth = textureBitsPerPixel[state.samplerID.texfmt] / 8;
|
||||
@@ -361,8 +362,13 @@ void BinManager::MarkPendingWrites(const Rasterizer::RasterizerState &state) {
|
||||
constexpr uint32_t mirrorMask = 0x041FFFFF;
|
||||
const uint32_t bpp = state.pixelID.FBFormat() == GE_FORMAT_8888 ? 4 : 2;
|
||||
pendingWrites_[0].Expand(gstate.getFrameBufAddress() & mirrorMask, bpp, gstate.FrameBufStride(), scissorTL, scissorBR);
|
||||
if (state.pixelID.depthWrite)
|
||||
if (state.pixelID.depthWrite) {
|
||||
pendingWrites_[1].Expand(gstate.getDepthBufAddress() & mirrorMask, 2, gstate.DepthBufStride(), scissorTL, scissorBR);
|
||||
} else if (gstate.isDepthTestEnabled() && !gstate.isModeClear()) {
|
||||
// Testing without writing still reads the depth buffer, so a transfer into it has to wait.
|
||||
const uint32_t depthAddr = gstate.getDepthBufAddress() & mirrorMask;
|
||||
pendingReads_[depthAddr].Expand(depthAddr, 2, gstate.DepthBufStride(), scissorTL, scissorBR);
|
||||
}
|
||||
}
|
||||
|
||||
inline void BinDirtyRange::Expand(uint32_t newBase, uint32_t bpp, uint32_t stride, const DrawingCoords &tl, const DrawingCoords &br) {
|
||||
@@ -584,8 +590,17 @@ void BinManager::Drain(bool flushing) {
|
||||
}
|
||||
|
||||
void BinManager::Flush(const char *reason) {
|
||||
if (queueRange_.x1 == 0x7FFFFFFF)
|
||||
if (queueRange_.x1 == 0x7FFFFFFF) {
|
||||
// Nothing queued, so nothing refers to the older states and CLUTs. Trim them anyway: callers
|
||||
// flush because one of these rings is full, and push into it right after.
|
||||
while (states_.Size() > 1) {
|
||||
states_.SkipNext();
|
||||
}
|
||||
while (cluts_.Size() > 1) {
|
||||
cluts_.SkipNext();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
double st = 0.0;
|
||||
const bool collectDebugStats = g_coreCollectDebugStats;
|
||||
|
||||
@@ -132,7 +132,7 @@ struct BinQueue {
|
||||
}
|
||||
|
||||
bool Full() const {
|
||||
return size_ == N - 1;
|
||||
return size_ >= N - 1;
|
||||
}
|
||||
|
||||
bool NearFull() const {
|
||||
@@ -278,13 +278,16 @@ private:
|
||||
const char *slowestFlushReason_ = nullptr;
|
||||
double slowestFlushTime_ = 0.0;
|
||||
int lastFlipstats_ = 0;
|
||||
// The framebuffer the queued draws render to. A framebuffer change flushes first, so it's one for all
|
||||
// of them, and during that flush gstate already has the new one.
|
||||
u32 drawTargetAddr_ = 0;
|
||||
int enqueues_ = 0;
|
||||
int mostThreads_ = 0;
|
||||
|
||||
void MarkPendingReads(const Rasterizer::RasterizerState &state);
|
||||
void MarkPendingWrites(const Rasterizer::RasterizerState &state);
|
||||
bool HasTextureWrite(const Rasterizer::RasterizerState &state);
|
||||
static bool IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item);
|
||||
bool IsExactSelfRender(const Rasterizer::RasterizerState &state, const BinItem &item) const;
|
||||
void OptimizePendingStates(uint16_t first, uint16_t last);
|
||||
BinCoords Scissor(BinCoords range);
|
||||
BinCoords Range(const VertexData &v0, const VertexData &v1, const VertexData &v2);
|
||||
|
||||
@@ -506,11 +506,16 @@ bool RectangleFastPath(const VertexData &v0, const VertexData &v1, BinManager &b
|
||||
if (g_needsClearAfterDialog) {
|
||||
g_needsClearAfterDialog = false;
|
||||
// Afterwards, we also need to clear the actual destination. Can do a fast rectfill.
|
||||
// The binner's state was computed with texturing on, so recompute it around the sprite.
|
||||
const SoftDirty texDirty = SoftDirty::SAMPLER_BASIC | SoftDirty::SAMPLER_TEXLIST | SoftDirty::RAST_TEX | SoftDirty::BINNER_OVERLAP;
|
||||
gstate.textureMapEnable &= ~1;
|
||||
binner.SetDirty(texDirty);
|
||||
binner.UpdateState();
|
||||
VertexData newV1 = v1;
|
||||
newV1.color0 = 0xFF000000;
|
||||
binner.AddSprite(v0, newV1);
|
||||
gstate.textureMapEnable |= 1;
|
||||
binner.SetDirty(texDirty);
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
|
||||
@@ -89,7 +89,6 @@ void DrawEngineVulkan::InitDeviceObjects() {
|
||||
res = vkCreateSampler(device, &samp, nullptr, &nullSampler_);
|
||||
_dbg_assert_(VK_SUCCESS == res);
|
||||
|
||||
draw_->SetInvalidationCallback(std::bind(&DrawEngineVulkan::Invalidate, this, std::placeholders::_1));
|
||||
}
|
||||
|
||||
DrawEngineVulkan::~DrawEngineVulkan() {
|
||||
@@ -106,6 +105,7 @@ void DrawEngineVulkan::DestroyDeviceObjects() {
|
||||
VulkanRenderManager *renderManager = (VulkanRenderManager *)draw_->GetNativeObject(Draw::NativeObject::RENDER_MANAGER);
|
||||
|
||||
draw_->SetInvalidationCallback(InvalidationCallback());
|
||||
invalidationCallbackInstalled_ = false;
|
||||
|
||||
pushUBO_ = nullptr;
|
||||
|
||||
@@ -143,6 +143,10 @@ void DrawEngineVulkan::DeviceRestore(Draw::DrawContext *draw) {
|
||||
|
||||
void DrawEngineVulkan::BeginFrame() {
|
||||
DrawEngineCommon::BeginFrame();
|
||||
if (!invalidationCallbackInstalled_) {
|
||||
draw_->SetInvalidationCallback(std::bind(&DrawEngineVulkan::Invalidate, this, std::placeholders::_1));
|
||||
invalidationCallbackInstalled_ = true;
|
||||
}
|
||||
|
||||
lastPipeline_ = nullptr;
|
||||
|
||||
|
||||
@@ -858,7 +858,18 @@ void TextureCacheVulkan::BuildTexture(TexCacheEntry *const entry) {
|
||||
VkImageView view = entry->vkTex->CreateViewForMip(i);
|
||||
VK_PROFILE_BEGIN(vulkan, cmdInit, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
||||
"Compute Upload: %dx%d->%dx%d", mipUnscaledWidth, mipUnscaledHeight, mipWidth, mipHeight);
|
||||
ScaleBufferToImage(vulkan, cmdInit, view, texBuf, bufferOffset, srcSize, mipUnscaledWidth, mipUnscaledHeight, mipWidth, mipHeight);
|
||||
if (!ScaleBufferToImage(vulkan, cmdInit, view, texBuf, bufferOffset, srcSize, mipUnscaledWidth, mipUnscaledHeight, mipWidth, mipHeight)) {
|
||||
// Nothing was written to this level. Clear it rather than leave garbage in it.
|
||||
WARN_LOG_ONCE(vkscalefail, Log::G3D, "Hardware texture scaling failed, clearing the texture");
|
||||
VkClearColorValue clear{};
|
||||
VkImageSubresourceRange range{ VK_IMAGE_ASPECT_COLOR_BIT, (uint32_t)i, 1, 0, 1 };
|
||||
vkCmdClearColorImage(cmdInit, entry->vkTex->GetImage(), VK_IMAGE_LAYOUT_GENERAL, &clear, 1, &range);
|
||||
// The barrier at the end of the upload waits on compute, so order the clear before that.
|
||||
VkMemoryBarrier barrier{ VK_STRUCTURE_TYPE_MEMORY_BARRIER };
|
||||
barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
||||
barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
|
||||
vkCmdPipelineBarrier(cmdInit, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, 1, &barrier, 0, nullptr, 0, nullptr);
|
||||
}
|
||||
VK_PROFILE_END(vulkan, cmdInit, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT);
|
||||
vulkan->Delete().QueueDeleteImageView(view);
|
||||
} else {
|
||||
|
||||
Reference in new issue
Block a user