From cd40e2d4f0478d34ceb58ef02c304335f5930dea Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Tue, 14 Jul 2026 17:17:40 +0200 Subject: [PATCH] Delete more code related to hardware skinning --- Common/GPU/Vulkan/VulkanRenderManager.h | 2 +- GPU/Common/ShaderCommon.h | 12 ++---------- GPU/Common/ShaderUniforms.cpp | 8 -------- GPU/Common/ShaderUniforms.h | 12 ------------ GPU/Common/VertexDecoderCommon.cpp | 4 +--- GPU/Common/VertexDecoderCommon.h | 4 ++-- GPU/D3D11/ShaderManagerD3D11.cpp | 16 ++------------- GPU/D3D11/ShaderManagerD3D11.h | 2 -- GPU/GPUCommon.cpp | 7 +------ GPU/GPUCommonHW.cpp | 12 ------------ GPU/GPUState.h | 1 - GPU/Vulkan/DrawEngineVulkan.cpp | 26 +++++++------------------ GPU/Vulkan/DrawEngineVulkan.h | 1 - GPU/Vulkan/ShaderManagerVulkan.cpp | 3 --- GPU/Vulkan/ShaderManagerVulkan.h | 5 ----- 15 files changed, 16 insertions(+), 99 deletions(-) diff --git a/Common/GPU/Vulkan/VulkanRenderManager.h b/Common/GPU/Vulkan/VulkanRenderManager.h index 68680d4e33..024d81b094 100644 --- a/Common/GPU/Vulkan/VulkanRenderManager.h +++ b/Common/GPU/Vulkan/VulkanRenderManager.h @@ -197,7 +197,7 @@ static_assert(sizeof(PackedDescriptor::buffer) == 16, "PackedDescriptor should b struct VKRPipelineLayout { ~VKRPipelineLayout(); - enum { MAX_DESC_SET_BINDINGS = 6 }; + enum { MAX_DESC_SET_BINDINGS = 5 }; BindingType bindingTypes[MAX_DESC_SET_BINDINGS]; uint32_t bindingTypesCount = 0; diff --git a/GPU/Common/ShaderCommon.h b/GPU/Common/ShaderCommon.h index 79dfbf724c..9298fd45fc 100644 --- a/GPU/Common/ShaderCommon.h +++ b/GPU/Common/ShaderCommon.h @@ -75,14 +75,8 @@ enum : uint64_t { DIRTY_WORLDMATRIX = 1ULL << 21, DIRTY_VIEWMATRIX = 1ULL << 22, DIRTY_TEXMATRIX = 1ULL << 23, - DIRTY_BONEMATRIX0 = 1ULL << 24, // NOTE: These must be under 32 - DIRTY_BONEMATRIX1 = 1ULL << 25, - DIRTY_BONEMATRIX2 = 1ULL << 26, - DIRTY_BONEMATRIX3 = 1ULL << 27, - DIRTY_BONEMATRIX4 = 1ULL << 28, - DIRTY_BONEMATRIX5 = 1ULL << 29, - DIRTY_BONEMATRIX6 = 1ULL << 30, - DIRTY_BONEMATRIX7 = 1ULL << 31, + + // Free uniform bits 24-31! // Free uniform bit 32!, DIRTY_TEXCLAMP = 1ULL << 33, @@ -100,8 +94,6 @@ enum : uint64_t { // Bits 41-42 are free for new uniforms (although the mask below needs updating). Then we're really out and need to start merging. // Don't forget to update DIRTY_ALL_UNIFORMS when you start using them. - DIRTY_BONE_UNIFORMS = 0xFF000000ULL, - DIRTY_ALL_UNIFORMS = 0x1FFFFFFFFFFULL, // Other dirty elements that aren't uniforms diff --git a/GPU/Common/ShaderUniforms.cpp b/GPU/Common/ShaderUniforms.cpp index 5895ba9b28..ef39fac7af 100644 --- a/GPU/Common/ShaderUniforms.cpp +++ b/GPU/Common/ShaderUniforms.cpp @@ -244,14 +244,6 @@ void LightUpdateUniforms(UB_VS_Lights *ub, uint64_t dirtyUniforms) { } } -void BoneUpdateUniforms(UB_VS_Bones *ub, uint64_t dirtyUniforms) { - for (int i = 0; i < 8; i++) { - if (dirtyUniforms & (DIRTY_BONEMATRIX0 << i)) { - ConvertMatrix4x3To3x4Transposed(ub->bones[i], gstate.boneMatrix + 12 * i); - } - } -} - void UpdateFogCoef(const GEState &state, float fogCoef[2]) { fogCoef[0] = getFloat24(gstate.fog1); fogCoef[1] = getFloat24(gstate.fog2); diff --git a/GPU/Common/ShaderUniforms.h b/GPU/Common/ShaderUniforms.h index 68162f9969..57041f6fa9 100644 --- a/GPU/Common/ShaderUniforms.h +++ b/GPU/Common/ShaderUniforms.h @@ -104,21 +104,9 @@ R"( vec4 u_ambient; vec3 u_lightspecular[4]; )"; -// With some cleverness, we could get away with uploading just half this when only the four or five first -// bones are being used. This is 384b. -struct alignas(16) UB_VS_Bones { - float bones[8][12]; -}; -static_assert(sizeof(UB_VS_Bones) == 384); // No way to optimize this further. - -static const char * const ub_vs_bonesStr = -R"( mat3x4 u_bone0; mat3x4 u_bone1; mat3x4 u_bone2; mat3x4 u_bone3; mat3x4 u_bone4; mat3x4 u_bone5; mat3x4 u_bone6; mat3x4 u_bone7; mat3x4 u_bone8; -)"; - // useBufferedRendering is only used to determine the rotation uniform. void BaseUpdateUniforms(UB_VS_FS_Base *ub, uint64_t dirtyUniforms, bool useBufferedRendering); void LightUpdateUniforms(UB_VS_Lights *ub, uint64_t dirtyUniforms); -void BoneUpdateUniforms(UB_VS_Bones *ub, uint64_t dirtyUniforms); uint32_t PackLightControlBits(); uint32_t PackDepalBits(); diff --git a/GPU/Common/VertexDecoderCommon.cpp b/GPU/Common/VertexDecoderCommon.cpp index b1bdd64871..23fcf8741a 100644 --- a/GPU/Common/VertexDecoderCommon.cpp +++ b/GPU/Common/VertexDecoderCommon.cpp @@ -55,10 +55,10 @@ inline int align(int n, int align) { return (n + (align - 1)) & ~(align - 1); } +// Map 1-4 bones to 4 bones. works fine. int TranslateNumBones(int bones) { if (!bones) return 0; if (bones < 4) return 4; - // if (bones < 8) return 8; I get drawing problems in FF:CC with this! return bones; } @@ -1285,8 +1285,6 @@ void VertexDecoder::SetVertexType(u32 fmt, const VertexDecoderOptions &options, if (skinInDecode) { // No visible output, computes a matrix that is passed through the skinMatrix variable // to the "nrm" and "pos" steps. - // Technically we should support morphing the weights too, but I have a hard time - // imagining that any game would use that.. but you never know. steps_[numSteps_++] = wtstep_skin[weighttype]; } else { int fmtBase = DEC_FLOAT_1; diff --git a/GPU/Common/VertexDecoderCommon.h b/GPU/Common/VertexDecoderCommon.h index 4fdeacc797..21e377e538 100644 --- a/GPU/Common/VertexDecoderCommon.h +++ b/GPU/Common/VertexDecoderCommon.h @@ -146,11 +146,11 @@ struct VertexDecoderOptions { inline uint32_t GetVertTypeID(uint32_t vertType, int uvGenMode) { // As the decoder depends on the UVGenMode when we use UV prescale, we simply mash it // into the top of the verttype where there are unused bits. - return (vertType & 0xFFFFFF) | (uvGenMode << 24) | (1 << 26); + return (vertType & 0xFFFFFF) | (uvGenMode << 24); } inline bool VertTypeIDSkinInDecode(uint32_t vertType) { - return ((vertType >> 26) & 1) != 0; + return true; } inline GETexMapMode VertTypeIDUVGenMode(uint32_t vertType) { diff --git a/GPU/D3D11/ShaderManagerD3D11.cpp b/GPU/D3D11/ShaderManagerD3D11.cpp index cddd5d8285..deca00bc44 100644 --- a/GPU/D3D11/ShaderManagerD3D11.cpp +++ b/GPU/D3D11/ShaderManagerD3D11.cpp @@ -82,11 +82,9 @@ ShaderManagerD3D11::ShaderManagerD3D11(Draw::DrawContext *draw, ID3D11Device *de codeBuffer_ = new char[CODE_BUFFER_SIZE]; memset(&ub_base, 0, sizeof(ub_base)); memset(&ub_lights, 0, sizeof(ub_lights)); - memset(&ub_bones, 0, sizeof(ub_bones)); static_assert(sizeof(ub_base) <= 512, "ub_base grew too big"); static_assert(sizeof(ub_lights) <= 512, "ub_lights grew too big"); - static_assert(sizeof(ub_bones) <= 384, "ub_bones grew too big"); InitDeviceObjects(); } @@ -98,19 +96,15 @@ ShaderManagerD3D11::~ShaderManagerD3D11() { } void ShaderManagerD3D11::InitDeviceObjects() { - D3D11_BUFFER_DESC desc{sizeof(ub_base), D3D11_USAGE_DYNAMIC, D3D11_BIND_CONSTANT_BUFFER, D3D11_CPU_ACCESS_WRITE}; ASSERT_SUCCESS(device_->CreateBuffer(&desc, nullptr, &push_base)); desc.ByteWidth = sizeof(ub_lights); ASSERT_SUCCESS(device_->CreateBuffer(&desc, nullptr, &push_lights)); - desc.ByteWidth = sizeof(ub_bones); - ASSERT_SUCCESS(device_->CreateBuffer(&desc, nullptr, &push_bones)); } void ShaderManagerD3D11::DestroyDeviceObjects() { push_base.Reset(); push_lights.Reset(); - push_bones.Reset(); Clear(); } @@ -168,21 +162,15 @@ uint64_t ShaderManagerD3D11::UpdateUniforms(bool useBufferedRendering) { memcpy(map.pData, &ub_lights, sizeof(ub_lights)); context_->Unmap(push_lights.Get(), 0); } - if (dirty & DIRTY_BONE_UNIFORMS) { - BoneUpdateUniforms(&ub_bones, dirty); - context_->Map(push_bones.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &map); - memcpy(map.pData, &ub_bones, sizeof(ub_bones)); - context_->Unmap(push_bones.Get(), 0); - } } gstate_c.CleanUniforms(); return dirty; } void ShaderManagerD3D11::BindUniforms() { - ID3D11Buffer *vs_cbs[3] = { push_base.Get(), push_lights.Get(), push_bones.Get() }; + ID3D11Buffer *vs_cbs[2] = { push_base.Get(), push_lights.Get() }; ID3D11Buffer *ps_cbs[1] = { push_base.Get() }; - context_->VSSetConstantBuffers(0, 3, vs_cbs); + context_->VSSetConstantBuffers(0, 2, vs_cbs); context_->PSSetConstantBuffers(0, 1, ps_cbs); } diff --git a/GPU/D3D11/ShaderManagerD3D11.h b/GPU/D3D11/ShaderManagerD3D11.h index 16591d99b1..77ea968152 100644 --- a/GPU/D3D11/ShaderManagerD3D11.h +++ b/GPU/D3D11/ShaderManagerD3D11.h @@ -130,12 +130,10 @@ private: // Uniform block scratchpad. These (the relevant ones) are copied to the current pushbuffer at draw time. UB_VS_FS_Base ub_base; UB_VS_Lights ub_lights; - UB_VS_Bones ub_bones; // Not actual pushbuffers, requires D3D11.1, let's try to live without that first. Microsoft::WRL::ComPtr push_base; Microsoft::WRL::ComPtr push_lights; - Microsoft::WRL::ComPtr push_bones; D3D11FragmentShader *lastFShader_ = nullptr; D3D11VertexShader *lastVShader_ = nullptr; diff --git a/GPU/GPUCommon.cpp b/GPU/GPUCommon.cpp index 9666b66376..6a86194fb4 100644 --- a/GPU/GPUCommon.cpp +++ b/GPU/GPUCommon.cpp @@ -345,7 +345,7 @@ void GPUCommon::ResetMatrices() { matrixVisible.tgen[i] = toFloat24(gstate.tgenMatrix[i]); // Assume all the matrices changed, so dirty things related to them. - gstate_c.Dirty(DIRTY_WORLDMATRIX | DIRTY_VIEWMATRIX | DIRTY_PROJMATRIX | DIRTY_TEXMATRIX | DIRTY_FRAGMENTSHADER_STATE | DIRTY_BONE_UNIFORMS); + gstate_c.Dirty(DIRTY_WORLDMATRIX | DIRTY_VIEWMATRIX | DIRTY_PROJMATRIX | DIRTY_TEXMATRIX | DIRTY_FRAGMENTSHADER_STATE); } u32 GPUCommon::EnqueueList(u32 listpc, u32 stall, int subIntrBase, PSPPointer args, bool head, bool *runList) { @@ -1389,12 +1389,7 @@ void GPUCommon::FastLoadBoneMatrix(u32 target) { const u32 num = gstate.boneMatrixNumber & 0x7F; _dbg_assert_msg_(num + 12 <= 96, "FastLoadBoneMatrix would corrupt memory"); const u32 mtxNum = num / 12; - u32 uniformsToDirty = DIRTY_BONEMATRIX0 << mtxNum; - if (num != 12 * mtxNum) { - uniformsToDirty |= DIRTY_BONEMATRIX0 << ((mtxNum + 1) & 7); - } - gstate_c.deferredVertTypeDirty |= uniformsToDirty; gstate.FastLoadBoneMatrix(target); cyclesExecuted += 2 * 14; // one to reset the counter, 12 to load the matrix, and a return. diff --git a/GPU/GPUCommonHW.cpp b/GPU/GPUCommonHW.cpp index 52af490d8b..7d9e803ac9 100644 --- a/GPU/GPUCommonHW.cpp +++ b/GPU/GPUCommonHW.cpp @@ -844,12 +844,6 @@ void GPUCommonHW::Execute_VertexTypeSkinning(u32 op, u32 diff) { gstate.vertType ^= diff; Flush(); gstate.vertType ^= diff; - // In this case, we may be doing weights and morphs. - // Update any bone matrix uniforms so it uses them correctly. - if ((op & GE_VTYPE_MORPHCOUNT_MASK) != 0) { - gstate_c.Dirty(gstate_c.deferredVertTypeDirty); - gstate_c.deferredVertTypeDirty = 0; - } gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE); } @@ -1687,11 +1681,6 @@ void GPUCommonHW::Execute_BoneMtxNum(u32 op, u32 diff) { break; } } - - const unsigned int numPlusCount = (op & 0x7F) + i; - for (unsigned int num = op & 0x7F; num < numPlusCount; num += 12) { - gstate_c.deferredVertTypeDirty |= DIRTY_BONEMATRIX0 << (num / 12); - } } const int count = i; @@ -1708,7 +1697,6 @@ void GPUCommonHW::Execute_BoneMtxData(u32 op, u32 diff) { u32 newVal = op << 8; if (num < 96 && newVal != ((const u32 *)gstate.boneMatrix)[num]) { // Bone matrices should NOT flush, as we're always doing skinning in decode nowadays! - gstate_c.deferredVertTypeDirty |= DIRTY_BONEMATRIX0 << (num / 12); ((u32 *)gstate.boneMatrix)[num] = newVal; } num++; diff --git a/GPU/GPUState.h b/GPU/GPUState.h index 48959bdfe8..a9214feb23 100644 --- a/GPU/GPUState.h +++ b/GPU/GPUState.h @@ -620,7 +620,6 @@ public: bool useFlagsChanged; float morphWeights[8]; - u32 deferredVertTypeDirty; u32 curTextureWidth; u32 curTextureHeight; diff --git a/GPU/Vulkan/DrawEngineVulkan.cpp b/GPU/Vulkan/DrawEngineVulkan.cpp index 91d7e0cf44..179b903c34 100644 --- a/GPU/Vulkan/DrawEngineVulkan.cpp +++ b/GPU/Vulkan/DrawEngineVulkan.cpp @@ -63,7 +63,6 @@ void DrawEngineVulkan::InitDeviceObjects() { BindingType::COMBINED_IMAGE_SAMPLER, // palette BindingType::UNIFORM_BUFFER_DYNAMIC_ALL, // uniforms BindingType::UNIFORM_BUFFER_DYNAMIC_VERTEX, // lights - BindingType::UNIFORM_BUFFER_DYNAMIC_VERTEX, // bones }; VulkanContext *vulkan = (VulkanContext *)draw_->GetNativeObject(Draw::NativeObject::CONTEXT); @@ -175,7 +174,7 @@ void DrawEngineVulkan::DirtyAllUBOs() { baseBuf = VK_NULL_HANDLE; lightBuf = VK_NULL_HANDLE; boneBuf = VK_NULL_HANDLE; - dirtyUniforms_ = DIRTY_BASE_UNIFORMS | DIRTY_LIGHT_UNIFORMS | DIRTY_BONE_UNIFORMS; + dirtyUniforms_ = DIRTY_BASE_UNIFORMS | DIRTY_LIGHT_UNIFORMS; imageView = VK_NULL_HANDLE; sampler = VK_NULL_HANDLE; gstate_c.Dirty(DIRTY_TEXTURE_IMAGE); @@ -336,7 +335,7 @@ void DrawEngineVulkan::Flush() { dirtyUniforms_ |= shaderManager_->UpdateUniforms(framebufferManager_->UseBufferedRendering()); UpdateUBOs(); - int descCount = 6; + int descCount = 5; int descSetIndex; PackedDescriptor *descriptors = renderManager->PushDescriptorSet(descCount, &descSetIndex); descriptors[0].image.view = imageView; @@ -356,14 +355,10 @@ void DrawEngineVulkan::Flush() { descriptors[4].buffer.range = sizeof(UB_VS_Lights); descriptors[4].buffer.offset = 0; - descriptors[5].buffer.buffer = boneBuf; - descriptors[5].buffer.range = sizeof(UB_VS_Bones); - descriptors[5].buffer.offset = 0; - // TODO: Can we avoid binding all three when not needed? Same below for hardware transform. // Think this will require different descriptor set layouts. - const uint32_t dynamicUBOOffsets[3] = { - baseUBOOffset, lightUBOOffset, boneUBOOffset, + const uint32_t dynamicUBOOffsets[2] = { + baseUBOOffset, lightUBOOffset, }; if (useElements) { VkBuffer ibuf; @@ -497,7 +492,7 @@ void DrawEngineVulkan::Flush() { // Even if the first draw is through-mode, make sure we at least have one copy of these uniforms buffered UpdateUBOs(); - int descCount = 6; + int descCount = 5; int descSetIndex; PackedDescriptor *descriptors = renderManager->PushDescriptorSet(descCount, &descSetIndex); descriptors[0].image.view = imageView; @@ -512,12 +507,9 @@ void DrawEngineVulkan::Flush() { descriptors[4].buffer.buffer = lightBuf; descriptors[4].buffer.range = sizeof(UB_VS_Lights); descriptors[4].buffer.offset = 0; - descriptors[5].buffer.buffer = boneBuf; - descriptors[5].buffer.range = sizeof(UB_VS_Bones); - descriptors[5].buffer.offset = 0; - const uint32_t dynamicUBOOffsets[3] = { - baseUBOOffset, lightUBOOffset, boneUBOOffset, + const uint32_t dynamicUBOOffsets[2] = { + baseUBOOffset, lightUBOOffset, }; PROFILE_THIS_SCOPE("renderman_q"); @@ -577,8 +569,4 @@ void DrawEngineVulkan::UpdateUBOs() { lightUBOOffset = shaderManager_->PushLightBuffer(pushUBO_, &lightBuf); dirtyUniforms_ &= ~DIRTY_LIGHT_UNIFORMS; } - if ((dirtyUniforms_ & DIRTY_BONE_UNIFORMS) || boneBuf == VK_NULL_HANDLE) { - boneUBOOffset = shaderManager_->PushBoneBuffer(pushUBO_, &boneBuf); - dirtyUniforms_ &= ~DIRTY_BONE_UNIFORMS; - } } diff --git a/GPU/Vulkan/DrawEngineVulkan.h b/GPU/Vulkan/DrawEngineVulkan.h index 77619f1d57..ef8afc112f 100644 --- a/GPU/Vulkan/DrawEngineVulkan.h +++ b/GPU/Vulkan/DrawEngineVulkan.h @@ -24,7 +24,6 @@ // * binding 2: Depal palette // * binding 3: Base Uniform Buffer (includes fragment state) // * binding 4: Light uniform buffer -// * binding 5: Bone uniform buffer // // All shaders conform to this layout, so they are all compatible with the same descriptor set. // The format of the various uniform buffers may vary though - vertex shaders that don't skin diff --git a/GPU/Vulkan/ShaderManagerVulkan.cpp b/GPU/Vulkan/ShaderManagerVulkan.cpp index f77b5d4653..860673c9fc 100644 --- a/GPU/Vulkan/ShaderManagerVulkan.cpp +++ b/GPU/Vulkan/ShaderManagerVulkan.cpp @@ -177,7 +177,6 @@ ShaderManagerVulkan::ShaderManagerVulkan(Draw::DrawContext *draw) static_assert(sizeof(uniforms_->ub_base) <= 512, "ub_base grew too big"); static_assert(sizeof(uniforms_->ub_lights) <= 512, "ub_lights grew too big"); - static_assert(sizeof(uniforms_->ub_bones) <= 384, "ub_bones grew too big"); } ShaderManagerVulkan::~ShaderManagerVulkan() { @@ -231,8 +230,6 @@ uint64_t ShaderManagerVulkan::UpdateUniforms(bool useBufferedRendering) { BaseUpdateUniforms(&uniforms_->ub_base, dirty, useBufferedRendering); if (dirty & DIRTY_LIGHT_UNIFORMS) LightUpdateUniforms(&uniforms_->ub_lights, dirty); - if (dirty & DIRTY_BONE_UNIFORMS) - BoneUpdateUniforms(&uniforms_->ub_bones, dirty); } gstate_c.CleanUniforms(); return dirty; diff --git a/GPU/Vulkan/ShaderManagerVulkan.h b/GPU/Vulkan/ShaderManagerVulkan.h index 6feb73087c..dc49d9cb01 100644 --- a/GPU/Vulkan/ShaderManagerVulkan.h +++ b/GPU/Vulkan/ShaderManagerVulkan.h @@ -89,7 +89,6 @@ struct Uniforms { // Uniform block scratchpad. These (the relevant ones) are copied to the current pushbuffer at draw time. UB_VS_FS_Base ub_base{}; UB_VS_Lights ub_lights{}; - UB_VS_Bones ub_bones{}; }; enum class ClipInfoFlags; @@ -129,10 +128,6 @@ public: uint32_t PushLightBuffer(VulkanPushPool *dest, VkBuffer *buf) const { return dest->Push(&uniforms_->ub_lights, sizeof(uniforms_->ub_lights), uboAlignment_, buf); } - // TODO: Only push half the bone buffer if we only have four bones. - uint32_t PushBoneBuffer(VulkanPushPool *dest, VkBuffer *buf) const { - return dest->Push(&uniforms_->ub_bones, sizeof(uniforms_->ub_bones), uboAlignment_, buf); - } static bool LoadCacheFlags(FILE *f, DrawEngineVulkan *drawEngine); bool LoadCache(FILE *f);