diff --git a/GPU/GPUCommon.cpp b/GPU/GPUCommon.cpp index 7338d29d3e..976392b818 100644 --- a/GPU/GPUCommon.cpp +++ b/GPU/GPUCommon.cpp @@ -50,9 +50,6 @@ #include "GPU/Debugger/Debugger.h" #include "GPU/Debugger/Record.h" -// TODO: Make class member? -GPUCommonHW::CommandInfo GPUCommon::cmdInfo_[256]; - void GPUCommon::Flush() { drawEngineCommon_->DispatchFlush(); } @@ -797,53 +794,6 @@ bool GPUCommon::InterpretList(DisplayList &list) { return gpuState == GPUSTATE_DONE || gpuState == GPUSTATE_ERROR; } -// Maybe should write this in ASM... -void GPUCommon::FastRunLoop(DisplayList &list) { - PROFILE_THIS_SCOPE("gpuloop"); - - if (!Memory::IsValidAddress(list.pc)) { - // We're having some serious problems here, just bail and try to limp along and not crash the app. - downcount = 0; - return; - } - - const CommandInfo *cmdInfo = cmdInfo_; - int dc = downcount; - for (; dc > 0; --dc) { - // We know that display list PCs have the upper nibble == 0 - no need to mask the pointer - const u32 op = *(const u32_le *)(Memory::base + list.pc); - const u32 cmd = op >> 24; - const CommandInfo &info = cmdInfo[cmd]; - const u32 diff = op ^ gstate.cmdmem[cmd]; - if (diff == 0) { - if (info.flags & FLAG_EXECUTE) { - downcount = dc; - (this->*info.func)(op, diff); - dc = downcount; - } - } else { - uint64_t flags = info.flags; - if (flags & FLAG_FLUSHBEFOREONCHANGE) { - if (drawEngineCommon_->GetNumDrawCalls()) { - drawEngineCommon_->DispatchFlush(); - } - } - gstate.cmdmem[cmd] = op; - if (flags & (FLAG_EXECUTE | FLAG_EXECUTEONCHANGE)) { - downcount = dc; - (this->*info.func)(op, diff); - dc = downcount; - } else { - uint64_t dirty = flags >> 8; - if (dirty) - gstate_c.Dirty(dirty); - } - } - list.pc += 4; - } - downcount = 0; -} - void GPUCommon::BeginFrame() { immCount_ = 0; if (dumpNextFrame_) { @@ -862,11 +812,9 @@ void GPUCommon::BeginFrame() { } } -void GPUCommon::SlowRunLoop(DisplayList &list) -{ +void GPUCommon::SlowRunLoop(DisplayList &list) { const bool dumpThisFrame = dumpThisFrame_; - while (downcount > 0) - { + while (downcount > 0) { bool process = GPUDebug::NotifyCommand(list.pc); if (process) { GPURecord::NotifyCommand(list.pc); @@ -1312,75 +1260,6 @@ void GPUCommon::Execute_End(u32 op, u32 diff) { } } -void GPUCommon::Execute_TexLevel(u32 op, u32 diff) { - // TODO: If you change the rules here, don't forget to update the inner interpreter in Execute_Prim. - if (diff == 0xFFFFFFFF) - return; - - gstate.texlevel ^= diff; - - if (diff & 0xFF0000) { - // Piggyback on this flag for 3D textures. - gstate_c.Dirty(DIRTY_MIPBIAS); - } - if (gstate.getTexLevelMode() != GE_TEXLEVEL_MODE_AUTO && (0x00FF0000 & gstate.texlevel) != 0) { - Flush(); - } - - gstate.texlevel ^= diff; - - gstate_c.Dirty(DIRTY_TEXTURE_PARAMS | DIRTY_FRAGMENTSHADER_STATE); -} - -void GPUCommon::Execute_TexSize0(u32 op, u32 diff) { - // Render to texture may have overridden the width/height. - // Don't reset it unless the size is different / the texture has changed. - if (diff || gstate_c.IsDirty(DIRTY_TEXTURE_IMAGE | DIRTY_TEXTURE_PARAMS)) { - gstate_c.curTextureWidth = gstate.getTextureWidth(0); - gstate_c.curTextureHeight = gstate.getTextureHeight(0); - gstate_c.Dirty(DIRTY_UVSCALEOFFSET); - // We will need to reset the texture now. - gstate_c.Dirty(DIRTY_TEXTURE_PARAMS); - } -} - -void GPUCommon::Execute_VertexType(u32 op, u32 diff) { - if (diff) - gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE); - if (diff & (GE_VTYPE_TC_MASK | GE_VTYPE_THROUGH_MASK)) { - gstate_c.Dirty(DIRTY_UVSCALEOFFSET); - // Switching between through and non-through, we need to invalidate a bunch of stuff. - if (diff & GE_VTYPE_THROUGH_MASK) - gstate_c.Dirty(DIRTY_RASTER_STATE | DIRTY_VIEWPORTSCISSOR_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_GEOMETRYSHADER_STATE | DIRTY_CULLRANGE | DIRTY_FOGCOEFENABLE); - } -} - -void GPUCommon::Execute_LoadClut(u32 op, u32 diff) { - gstate_c.Dirty(DIRTY_TEXTURE_PARAMS); - textureCache_->LoadClut(gstate.getClutAddress(), gstate.getClutLoadBytes()); -} - -void GPUCommon::Execute_VertexTypeSkinning(u32 op, u32 diff) { - // Don't flush when weight count changes. - if (diff & ~GE_VTYPE_WEIGHTCOUNT_MASK) { - // Restore and flush - gstate.vertType ^= diff; - Flush(); - gstate.vertType ^= diff; - if (diff & (GE_VTYPE_TC_MASK | GE_VTYPE_THROUGH_MASK)) - gstate_c.Dirty(DIRTY_UVSCALEOFFSET); - // In this case, we may be doing weights and morphs. - // Update any bone matrix uniforms so it uses them correctly. - if ((op & GE_VTYPE_MORPHCOUNT_MASK) != 0) { - gstate_c.Dirty(gstate_c.deferredVertTypeDirty); - gstate_c.deferredVertTypeDirty = 0; - } - gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE); - } - if (diff & GE_VTYPE_THROUGH_MASK) - gstate_c.Dirty(DIRTY_RASTER_STATE | DIRTY_VIEWPORTSCISSOR_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_GEOMETRYSHADER_STATE | DIRTY_CULLRANGE | DIRTY_FOGCOEFENABLE); -} - void GPUCommon::Execute_BoundingBox(u32 op, u32 diff) { // Just resetting, nothing to check bounds for. const u32 count = op & 0xFFFF; @@ -1849,19 +1728,6 @@ void GPUCommon::FlushImm() { } } -void GPUCommon::ExecuteOp(u32 op, u32 diff) { - const u8 cmd = op >> 24; - const CommandInfo info = cmdInfo_[cmd]; - const u8 cmdFlags = info.flags; - if ((cmdFlags & FLAG_EXECUTE) || (diff && (cmdFlags & FLAG_EXECUTEONCHANGE))) { - (this->*info.func)(op, diff); - } else if (diff) { - uint64_t dirty = info.flags >> 8; - if (dirty) - gstate_c.Dirty(dirty); - } -} - void GPUCommon::Execute_Unknown(u32 op, u32 diff) { if ((op & 0xFFFFFF) != 0) WARN_LOG_REPORT_ONCE(unknowncmd, G3D, "Unknown GE command : %08x ", op); @@ -2613,45 +2479,3 @@ size_t GPUCommon::FormatGPUStatsCommon(char *buffer, size_t size) { vertexAverageCycles ); } - - -u32 GPUCommon::CheckGPUFeaturesLate(u32 features) const { - // If we already have a 16-bit depth buffer, we don't need to round. - bool prefer24 = draw_->GetDeviceCaps().preferredDepthBufferFormat == Draw::DataFormat::D24_S8; - bool prefer16 = draw_->GetDeviceCaps().preferredDepthBufferFormat == Draw::DataFormat::D16; - if (!prefer16) { - if (sawExactEqualDepth_ && (features & GPU_USE_ACCURATE_DEPTH) != 0) { - // Exact equal tests tend to have issues unless we use the PSP's depth range. - // We use 24-bit depth virtually everwhere, the fallback is just for safety. - if (prefer24) - features |= GPU_SCALE_DEPTH_FROM_24BIT_TO_16BIT; - else - features |= GPU_ROUND_FRAGMENT_DEPTH_TO_16BIT; - } else if (!g_Config.bHighQualityDepth && (features & GPU_USE_ACCURATE_DEPTH) != 0) { - features |= GPU_SCALE_DEPTH_FROM_24BIT_TO_16BIT; - } else if (PSP_CoreParameter().compat.flags().PixelDepthRounding) { - if (prefer24 && (features & GPU_USE_ACCURATE_DEPTH) != 0) { - // Here we can simulate a 16 bit depth buffer by scaling. - // Note that the depth buffer is fixed point, not floating, so dividing by 256 is pretty good. - features |= GPU_SCALE_DEPTH_FROM_24BIT_TO_16BIT; - } else { - // Use fragment rounding on where available otherwise. - features |= GPU_ROUND_FRAGMENT_DEPTH_TO_16BIT; - } - } else if (PSP_CoreParameter().compat.flags().VertexDepthRounding) { - features |= GPU_ROUND_DEPTH_TO_16BIT; - } - } - - return features; -} - -void GPUCommon::CheckFlushOp(int cmd, u32 diff) { - const u8 cmdFlags = cmdInfo_[cmd].flags; - if (diff && (cmdFlags & FLAG_FLUSHBEFOREONCHANGE)) { - if (dumpThisFrame_) { - NOTICE_LOG(G3D, "================ FLUSH ================"); - } - drawEngineCommon_->DispatchFlush(); - } -} diff --git a/GPU/GPUCommon.h b/GPU/GPUCommon.h index f234f2faa3..8407c6a520 100644 --- a/GPU/GPUCommon.h +++ b/GPU/GPUCommon.h @@ -60,7 +60,6 @@ struct TransformedVertex { u32 color1_32; }; - void CopyFromWithOffset(const TransformedVertex &other, float xoff, float yoff) { this->x = other.x + xoff; this->y = other.y + yoff; @@ -108,7 +107,6 @@ public: void DumpNextFrame() override; - void ExecuteOp(u32 op, u32 diff) override; virtual void PreExecuteOp(u32 op, u32 diff) {} bool InterpretList(DisplayList &list); @@ -152,16 +150,8 @@ public: void Execute_Ret(u32 op, u32 diff); void Execute_End(u32 op, u32 diff); - void Execute_VertexType(u32 op, u32 diff); - void Execute_VertexTypeSkinning(u32 op, u32 diff); - void Execute_BoundingBox(u32 op, u32 diff); - void Execute_LoadClut(u32 op, u32 diff); - - void Execute_TexSize0(u32 op, u32 diff); - void Execute_TexLevel(u32 op, u32 diff); - void Execute_WorldMtxNum(u32 op, u32 diff); void Execute_WorldMtxData(u32 op, u32 diff); void Execute_ViewMtxNum(u32 op, u32 diff); @@ -267,9 +257,6 @@ protected: virtual void CheckRenderResized() {} - // Add additional common features dependent on other features, which may be backend-determined. - u32 CheckGPUFeaturesLate(u32 features) const; - inline bool IsTrianglePrim(GEPrimitiveType prim) const { return prim != GE_PRIM_RECTANGLES && prim > GE_PRIM_LINE_STRIP; } @@ -292,7 +279,7 @@ protected: void BeginFrame() override; void UpdateVsyncInterval(bool force); - virtual void FastRunLoop(DisplayList &list); + virtual void FastRunLoop(DisplayList &list) = 0; void SlowRunLoop(DisplayList &list); void UpdatePC(u32 currentPC, u32 newPC); @@ -304,8 +291,6 @@ protected: // TODO: Unify this. The only backend that differs is Vulkan. virtual void FinishDeferred() {} - void CheckFlushOp(int cmd, u32 diff); - void AdvanceVerts(u32 vertType, int count, int bytesRead) { if ((vertType & GE_VTYPE_IDX_MASK) != GE_VTYPE_IDX_NONE) { int indexShift = ((vertType & GE_VTYPE_IDX_MASK) >> GE_VTYPE_IDX_SHIFT) - 1; @@ -333,21 +318,6 @@ protected: GraphicsContext *gfxCtx_; Draw::DrawContext *draw_; - struct CommandInfo { - uint64_t flags; - GPUCommon::CmdFunc func; - - // Dirty flags are mashed into the regular flags by a left shift of 8. - void AddDirty(u64 dirty) { - flags |= dirty << 8; - } - void RemoveDirty(u64 dirty) { - flags &= ~(dirty << 8); - } - }; - - static CommandInfo cmdInfo_[256]; - typedef std::list DisplayListQueue; int nextListID; diff --git a/GPU/GPUCommonHW.cpp b/GPU/GPUCommonHW.cpp index eb62e1ff60..41f0a34c44 100644 --- a/GPU/GPUCommonHW.cpp +++ b/GPU/GPUCommonHW.cpp @@ -20,6 +20,21 @@ struct CommonCommandTableEntry { GPUCommonHW::CmdFunc func; }; +struct CommandInfo { + uint64_t flags; + GPUCommonHW::CmdFunc func; + + // Dirty flags are mashed into the regular flags by a left shift of 8. + void AddDirty(u64 dirty) { + flags |= dirty << 8; + } + void RemoveDirty(u64 dirty) { + flags &= ~(dirty << 8); + } +}; + +static CommandInfo cmdInfo_[256]; + const CommonCommandTableEntry commonCommandTable[] = { // From Common. No flushing but definitely need execute. { GE_CMD_OFFSETADDR, FLAG_EXECUTE, 0, &GPUCommon::Execute_OffsetAddr }, @@ -38,9 +53,9 @@ const CommonCommandTableEntry commonCommandTable[] = { { GE_CMD_SPLINE, FLAG_EXECUTE, 0, &GPUCommonHW::Execute_Spline }, // Changing the vertex type requires us to flush. - { GE_CMD_VERTEXTYPE, FLAG_FLUSHBEFOREONCHANGE | FLAG_EXECUTEONCHANGE, 0, &GPUCommon::Execute_VertexType }, + { GE_CMD_VERTEXTYPE, FLAG_FLUSHBEFOREONCHANGE | FLAG_EXECUTEONCHANGE, 0, &GPUCommonHW::Execute_VertexType }, - { GE_CMD_LOADCLUT, FLAG_FLUSHBEFOREONCHANGE | FLAG_EXECUTE, 0, &GPUCommon::Execute_LoadClut }, + { GE_CMD_LOADCLUT, FLAG_FLUSHBEFOREONCHANGE | FLAG_EXECUTE, 0, &GPUCommonHW::Execute_LoadClut}, // These two are actually processed in CMD_END. { GE_CMD_SIGNAL }, @@ -122,7 +137,7 @@ const CommonCommandTableEntry commonCommandTable[] = { { GE_CMD_TEXOFFSETU }, { GE_CMD_TEXOFFSETV }, - { GE_CMD_TEXSIZE0, FLAG_FLUSHBEFOREONCHANGE | FLAG_EXECUTE, 0, &GPUCommon::Execute_TexSize0 }, + { GE_CMD_TEXSIZE0, FLAG_FLUSHBEFOREONCHANGE | FLAG_EXECUTE, 0, &GPUCommonHW::Execute_TexSize0 }, { GE_CMD_TEXSIZE1, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, { GE_CMD_TEXSIZE2, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, { GE_CMD_TEXSIZE3, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, @@ -131,7 +146,7 @@ const CommonCommandTableEntry commonCommandTable[] = { { GE_CMD_TEXSIZE6, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, { GE_CMD_TEXSIZE7, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, { GE_CMD_TEXFORMAT, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_IMAGE }, - { GE_CMD_TEXLEVEL, FLAG_EXECUTEONCHANGE, DIRTY_TEXTURE_PARAMS, &GPUCommon::Execute_TexLevel }, + { GE_CMD_TEXLEVEL, FLAG_EXECUTEONCHANGE, DIRTY_TEXTURE_PARAMS, &GPUCommonHW::Execute_TexLevel }, { GE_CMD_TEXLODSLOPE, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, { GE_CMD_TEXADDR0, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_IMAGE | DIRTY_UVSCALEOFFSET }, { GE_CMD_TEXADDR1, FLAG_FLUSHBEFOREONCHANGE, DIRTY_TEXTURE_PARAMS }, @@ -356,7 +371,7 @@ GPUCommonHW::GPUCommonHW(GraphicsContext *gfxCtx, Draw::DrawContext *draw) : GPU dupeCheck.insert(cmd); } cmdInfo_[cmd].flags |= (uint64_t)commonCommandTable[i].flags | (commonCommandTable[i].dirty << 8); - cmdInfo_[cmd].func = (GPUCommon::CmdFunc)commonCommandTable[i].func; + cmdInfo_[cmd].func = commonCommandTable[i].func; if ((cmdInfo_[cmd].flags & (FLAG_EXECUTE | FLAG_EXECUTEONCHANGE)) && !cmdInfo_[cmd].func) { // Can't have FLAG_EXECUTE commands without a function pointer to execute. Crash(); @@ -402,10 +417,10 @@ void GPUCommonHW::DeviceLost() { void GPUCommonHW::UpdateCmdInfo() { if (g_Config.bSoftwareSkinning) { cmdInfo_[GE_CMD_VERTEXTYPE].flags &= ~FLAG_FLUSHBEFOREONCHANGE; - cmdInfo_[GE_CMD_VERTEXTYPE].func = &GPUCommon::Execute_VertexTypeSkinning; + cmdInfo_[GE_CMD_VERTEXTYPE].func = &GPUCommonHW::Execute_VertexTypeSkinning; } else { cmdInfo_[GE_CMD_VERTEXTYPE].flags |= FLAG_FLUSHBEFOREONCHANGE; - cmdInfo_[GE_CMD_VERTEXTYPE].func = &GPUCommon::Execute_VertexType; + cmdInfo_[GE_CMD_VERTEXTYPE].func = &GPUCommonHW::Execute_VertexType; } if (g_Config.bFastMemory) { @@ -442,6 +457,16 @@ void GPUCommonHW::UpdateCmdInfo() { } } +void GPUCommonHW::CheckFlushOp(int cmd, u32 diff) { + const u8 cmdFlags = cmdInfo_[cmd].flags; + if (diff && (cmdFlags & FLAG_FLUSHBEFOREONCHANGE)) { + if (dumpThisFrame_) { + NOTICE_LOG(G3D, "================ FLUSH ================"); + } + drawEngineCommon_->DispatchFlush(); + } +} + void GPUCommonHW::PreExecuteOp(u32 op, u32 diff) { CheckFlushOp(op >> 24, diff); } @@ -547,6 +572,37 @@ u32 GPUCommonHW::CheckGPUFeatures() const { return features; } +u32 GPUCommonHW::CheckGPUFeaturesLate(u32 features) const { + // If we already have a 16-bit depth buffer, we don't need to round. + bool prefer24 = draw_->GetDeviceCaps().preferredDepthBufferFormat == Draw::DataFormat::D24_S8; + bool prefer16 = draw_->GetDeviceCaps().preferredDepthBufferFormat == Draw::DataFormat::D16; + if (!prefer16) { + if (sawExactEqualDepth_ && (features & GPU_USE_ACCURATE_DEPTH) != 0) { + // Exact equal tests tend to have issues unless we use the PSP's depth range. + // We use 24-bit depth virtually everwhere, the fallback is just for safety. + if (prefer24) + features |= GPU_SCALE_DEPTH_FROM_24BIT_TO_16BIT; + else + features |= GPU_ROUND_FRAGMENT_DEPTH_TO_16BIT; + } else if (!g_Config.bHighQualityDepth && (features & GPU_USE_ACCURATE_DEPTH) != 0) { + features |= GPU_SCALE_DEPTH_FROM_24BIT_TO_16BIT; + } else if (PSP_CoreParameter().compat.flags().PixelDepthRounding) { + if (prefer24 && (features & GPU_USE_ACCURATE_DEPTH) != 0) { + // Here we can simulate a 16 bit depth buffer by scaling. + // Note that the depth buffer is fixed point, not floating, so dividing by 256 is pretty good. + features |= GPU_SCALE_DEPTH_FROM_24BIT_TO_16BIT; + } else { + // Use fragment rounding on where available otherwise. + features |= GPU_ROUND_FRAGMENT_DEPTH_TO_16BIT; + } + } else if (PSP_CoreParameter().compat.flags().VertexDepthRounding) { + features |= GPU_ROUND_DEPTH_TO_16BIT; + } + } + + return features; +} + void GPUCommonHW::UpdateMSAALevel(Draw::DrawContext *draw) { int level = g_Config.iMultiSampleLevel; if (draw && draw->GetDeviceCaps().multiSampleLevelsMask & (1 << level)) { @@ -604,6 +660,97 @@ void GPUCommonHW::CheckDepthUsage(VirtualFramebuffer *vfb) { } } +void GPUCommonHW::ExecuteOp(u32 op, u32 diff) { + const u8 cmd = op >> 24; + const CommandInfo info = cmdInfo_[cmd]; + const u8 cmdFlags = info.flags; + if ((cmdFlags & FLAG_EXECUTE) || (diff && (cmdFlags & FLAG_EXECUTEONCHANGE))) { + (this->*info.func)(op, diff); + } else if (diff) { + uint64_t dirty = info.flags >> 8; + if (dirty) + gstate_c.Dirty(dirty); + } +} + +void GPUCommonHW::FastRunLoop(DisplayList &list) { + PROFILE_THIS_SCOPE("gpuloop"); + + if (!Memory::IsValidAddress(list.pc)) { + // We're having some serious problems here, just bail and try to limp along and not crash the app. + downcount = 0; + return; + } + + const CommandInfo *cmdInfo = cmdInfo_; + int dc = downcount; + for (; dc > 0; --dc) { + // We know that display list PCs have the upper nibble == 0 - no need to mask the pointer + const u32 op = *(const u32_le *)(Memory::base + list.pc); + const u32 cmd = op >> 24; + const CommandInfo &info = cmdInfo[cmd]; + const u32 diff = op ^ gstate.cmdmem[cmd]; + if (diff == 0) { + if (info.flags & FLAG_EXECUTE) { + downcount = dc; + (this->*info.func)(op, diff); + dc = downcount; + } + } else { + uint64_t flags = info.flags; + if (flags & FLAG_FLUSHBEFOREONCHANGE) { + if (drawEngineCommon_->GetNumDrawCalls()) { + drawEngineCommon_->DispatchFlush(); + } + } + gstate.cmdmem[cmd] = op; + if (flags & (FLAG_EXECUTE | FLAG_EXECUTEONCHANGE)) { + downcount = dc; + (this->*info.func)(op, diff); + dc = downcount; + } else { + uint64_t dirty = flags >> 8; + if (dirty) + gstate_c.Dirty(dirty); + } + } + list.pc += 4; + } + downcount = 0; +} + +void GPUCommonHW::Execute_VertexType(u32 op, u32 diff) { + if (diff) + gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE); + if (diff & (GE_VTYPE_TC_MASK | GE_VTYPE_THROUGH_MASK)) { + gstate_c.Dirty(DIRTY_UVSCALEOFFSET); + // Switching between through and non-through, we need to invalidate a bunch of stuff. + if (diff & GE_VTYPE_THROUGH_MASK) + gstate_c.Dirty(DIRTY_RASTER_STATE | DIRTY_VIEWPORTSCISSOR_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_GEOMETRYSHADER_STATE | DIRTY_CULLRANGE | DIRTY_FOGCOEFENABLE); + } +} + +void GPUCommonHW::Execute_VertexTypeSkinning(u32 op, u32 diff) { + // Don't flush when weight count changes. + if (diff & ~GE_VTYPE_WEIGHTCOUNT_MASK) { + // Restore and flush + gstate.vertType ^= diff; + Flush(); + gstate.vertType ^= diff; + if (diff & (GE_VTYPE_TC_MASK | GE_VTYPE_THROUGH_MASK)) + gstate_c.Dirty(DIRTY_UVSCALEOFFSET); + // In this case, we may be doing weights and morphs. + // Update any bone matrix uniforms so it uses them correctly. + if ((op & GE_VTYPE_MORPHCOUNT_MASK) != 0) { + gstate_c.Dirty(gstate_c.deferredVertTypeDirty); + gstate_c.deferredVertTypeDirty = 0; + } + gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE); + } + if (diff & GE_VTYPE_THROUGH_MASK) + gstate_c.Dirty(DIRTY_RASTER_STATE | DIRTY_VIEWPORTSCISSOR_STATE | DIRTY_FRAGMENTSHADER_STATE | DIRTY_GEOMETRYSHADER_STATE | DIRTY_CULLRANGE | DIRTY_FOGCOEFENABLE); +} + void GPUCommonHW::Execute_Prim(u32 op, u32 diff) { // This drives all drawing. All other state we just buffer up, then we apply it only // when it's time to draw. As most PSP games set state redundantly ALL THE TIME, this is a huge optimization. @@ -1028,3 +1175,40 @@ void GPUCommonHW::Execute_BlockTransferStart(u32 op, u32 diff) { // Can we skip this on SkipDraw? DoBlockTransfer(gstate_c.skipDrawReason); } + +void GPUCommonHW::Execute_TexSize0(u32 op, u32 diff) { + // Render to texture may have overridden the width/height. + // Don't reset it unless the size is different / the texture has changed. + if (diff || gstate_c.IsDirty(DIRTY_TEXTURE_IMAGE | DIRTY_TEXTURE_PARAMS)) { + gstate_c.curTextureWidth = gstate.getTextureWidth(0); + gstate_c.curTextureHeight = gstate.getTextureHeight(0); + gstate_c.Dirty(DIRTY_UVSCALEOFFSET); + // We will need to reset the texture now. + gstate_c.Dirty(DIRTY_TEXTURE_PARAMS); + } +} + +void GPUCommonHW::Execute_TexLevel(u32 op, u32 diff) { + // TODO: If you change the rules here, don't forget to update the inner interpreter in Execute_Prim. + if (diff == 0xFFFFFFFF) + return; + + gstate.texlevel ^= diff; + + if (diff & 0xFF0000) { + // Piggyback on this flag for 3D textures. + gstate_c.Dirty(DIRTY_MIPBIAS); + } + if (gstate.getTexLevelMode() != GE_TEXLEVEL_MODE_AUTO && (0x00FF0000 & gstate.texlevel) != 0) { + Flush(); + } + + gstate.texlevel ^= diff; + + gstate_c.Dirty(DIRTY_TEXTURE_PARAMS | DIRTY_FRAGMENTSHADER_STATE); +} + +void GPUCommonHW::Execute_LoadClut(u32 op, u32 diff) { + gstate_c.Dirty(DIRTY_TEXTURE_PARAMS); + textureCache_->LoadClut(gstate.getClutAddress(), gstate.getClutLoadBytes()); +} diff --git a/GPU/GPUCommonHW.h b/GPU/GPUCommonHW.h index 11cadb0d0a..4f02ab15ee 100644 --- a/GPU/GPUCommonHW.h +++ b/GPU/GPUCommonHW.h @@ -19,17 +19,27 @@ public: std::vector DebugGetShaderIDs(DebugShaderType shader) override; std::string DebugGetShaderString(std::string id, DebugShaderType shader, DebugShaderStringType stringType) override; + void Execute_VertexType(u32 op, u32 diff); + void Execute_VertexTypeSkinning(u32 op, u32 diff); + void Execute_Prim(u32 op, u32 diff); void Execute_Bezier(u32 op, u32 diff); void Execute_Spline(u32 op, u32 diff); void Execute_BlockTransferStart(u32 op, u32 diff); + void Execute_TexSize0(u32 op, u32 diff); + void Execute_TexLevel(u32 op, u32 diff); + void Execute_LoadClut(u32 op, u32 diff); + typedef void (GPUCommonHW::*CmdFunc)(u32 op, u32 diff); + void FastRunLoop(DisplayList &list) override; + void ExecuteOp(u32 op, u32 diff) override; + protected: void UpdateCmdInfo() override; - void PreExecuteOp(u32 op, u32 diff); + void PreExecuteOp(u32 op, u32 diff) override; void ClearCacheNextFrame() override; // Needs to be called on GPU thread, not reporting thread. @@ -37,9 +47,11 @@ protected: void UpdateMSAALevel(Draw::DrawContext *draw) override; void CheckRenderResized() override; + u32 CheckGPUFeaturesLate(u32 features) const; int msaaLevel_ = 0; private: void CheckDepthUsage(VirtualFramebuffer *vfb); + void CheckFlushOp(int cmd, u32 diff); };