From 0a29202e12554c8e47da72093f58cbc0b9599bd1 Mon Sep 17 00:00:00 2001 From: Ced2911 Date: Tue, 10 Sep 2013 22:35:38 +0200 Subject: [PATCH] sync gpu with gles --- GPU/Directx9/DisplayListInterpreter.cpp | 30 ++++-- GPU/Directx9/Framebuffer.cpp | 55 +++++----- GPU/Directx9/Framebuffer.h | 1 + GPU/Directx9/TextureCache.cpp | 138 ++++++++++++++---------- GPU/Directx9/TextureCache.h | 2 + GPU/Directx9/TransformPipeline.cpp | 23 ++-- GPU/Directx9/VertexDecoder.cpp | 6 +- GPU/Directx9/VertexDecoder.h | 31 ++++-- GPU/Directx9/VertexShaderGenerator.cpp | 2 +- 9 files changed, 179 insertions(+), 109 deletions(-) diff --git a/GPU/Directx9/DisplayListInterpreter.cpp b/GPU/Directx9/DisplayListInterpreter.cpp index 97039acf61..c91f1247a1 100644 --- a/GPU/Directx9/DisplayListInterpreter.cpp +++ b/GPU/Directx9/DisplayListInterpreter.cpp @@ -355,7 +355,7 @@ static const CommandTableEntry commandTable[] = { DIRECTX9_GPU::DIRECTX9_GPU() : resized_(false) { - lastVsync_ = g_Config.bVSync; + lastVsync_ = g_Config.bVSync ? 1 : 0; dxstate.SetVSyncInterval(g_Config.bVSync); shaderManager_ = new ShaderManager(); @@ -417,8 +417,8 @@ void DIRECTX9_GPU::InitClear() { ScheduleEvent(GPU_EVENT_INIT_CLEAR); } void DIRECTX9_GPU::InitClearInternal() { - bool useBufferedRendering = g_Config.iRenderingMode != 0 ? 1 : 0; - if (useBufferedRendering) { + bool useNonBufferedRendering = g_Config.iRenderingMode == FB_NON_BUFFERED_MODE; + if (useNonBufferedRendering) { dxstate.depthWrite.set(true); dxstate.colorMask.set(true, true, true, true); /* @@ -440,8 +440,8 @@ void DIRECTX9_GPU::BeginFrame() { void DIRECTX9_GPU::BeginFrameInternal() { // Turn off vsync when unthrottled - int desiredVSyncInterval = g_Config.bVSync; - if (PSP_CoreParameter().unthrottle) + int desiredVSyncInterval = g_Config.bVSync ? 1 : 0; + if ((PSP_CoreParameter().unthrottle) || (PSP_CoreParameter().fpsLimit == 1)) desiredVSyncInterval = 0; if (desiredVSyncInterval != lastVsync_) { dxstate.SetVSyncInterval(desiredVSyncInterval); @@ -522,8 +522,7 @@ void DIRECTX9_GPU::CopyDisplayToOutputInternal() { gstate_c.textureChanged = true; } -// Render queue - +// Maybe should write this in ASM... void DIRECTX9_GPU::FastRunLoop(DisplayList &list) { for (; downcount > 0; --downcount) { u32 op = Memory::ReadUnchecked_U32(list.pc); @@ -569,7 +568,7 @@ inline void DIRECTX9_GPU::CheckFlushOp(int cmd, u32 diff) { u8 cmdFlags = commandFlags_[cmd]; if ((cmdFlags & FLAG_FLUSHBEFORE) || (diff && (cmdFlags & FLAG_FLUSHBEFOREONCHANGE))) { if (dumpThisFrame_) { - NOTICE_LOG(HLE, "================ FLUSH ================"); + NOTICE_LOG(G3D, "================ FLUSH ================"); } transformDraw_.Flush(); } @@ -604,6 +603,9 @@ void DIRECTX9_GPU::ExecuteOp(u32 op, u32 diff) { u32 count = data & 0xFFFF; GEPrimitiveType prim = static_cast(data >> 16); + if (count == 0) + break; + // Discard AA lines as we can't do anything that makes sense with these anyway. The SW plugin might, though. // Discard AA lines in DOA @@ -1093,7 +1095,7 @@ void DIRECTX9_GPU::ExecuteOp(u32 op, u32 diff) { case GE_CMD_ALPHATEST: #ifndef USING_GLES2 if (((data >> 16) & 0xFF) != 0xFF && (data & 7) > 1) - WARN_LOG_REPORT_ONCE(alphatestmask, HLE, "Unsupported alphatest mask: %02x", (data >> 16) & 0xFF); + WARN_LOG_REPORT_ONCE(alphatestmask, G3D, "Unsupported alphatest mask: %02x", (data >> 16) & 0xFF); // Intentional fallthrough. #endif case GE_CMD_COLORREF: @@ -1302,6 +1304,16 @@ void DIRECTX9_GPU::DoBlockTransfer() { DEBUG_LOG(G3D, "Block transfer: %08x/%x -> %08x/%x, %ix%ix%i (%i,%i)->(%i,%i)", srcBasePtr, srcStride, dstBasePtr, dstStride, width, height, bpp, srcX, srcY, dstX, dstY); + if (!Memory::IsValidAddress(srcBasePtr)) { + ERROR_LOG_REPORT(G3D, "BlockTransfer: Bad source transfer address %08x!", srcBasePtr); + return; + } + + if (!Memory::IsValidAddress(dstBasePtr)) { + ERROR_LOG_REPORT(G3D, "BlockTransfer: Bad destination transfer address %08x!", dstBasePtr); + return; + } + // Do the copy! for (int y = 0; y < height; y++) { const u8 *src = Memory::GetPointer(srcBasePtr + ((y + srcY) * srcStride + srcX) * bpp); diff --git a/GPU/Directx9/Framebuffer.cpp b/GPU/Directx9/Framebuffer.cpp index 8e94170da1..c918f4f2a6 100644 --- a/GPU/Directx9/Framebuffer.cpp +++ b/GPU/Directx9/Framebuffer.cpp @@ -287,7 +287,7 @@ VirtualFramebuffer *FramebufferManager::GetDisplayFBO() { VirtualFramebuffer *v = vfbs_[i]; if (MaskedEqual(v->fb_address, displayFramebufPtr_) && v->format == displayFormat_ && v->width >= 480) { // Could check w too but whatever - if (match == NULL || match->last_frame_used < v->last_frame_used) { + if (match == NULL || match->last_frame_render < v->last_frame_render) { match = v; } } @@ -296,7 +296,7 @@ VirtualFramebuffer *FramebufferManager::GetDisplayFBO() { return match; } - DEBUG_LOG(HLE, "Finding no FBO matching address %08x", displayFramebufPtr_); + DEBUG_LOG(SCEGE, "Finding no FBO matching address %08x", displayFramebufPtr_); #if 0 // defined(_DEBUG) std::string debug = "FBOs: "; for (size_t i = 0; i < vfbs_.size(); ++i) { @@ -304,7 +304,7 @@ VirtualFramebuffer *FramebufferManager::GetDisplayFBO() { sprintf(temp, "%08x %i %i", vfbs_[i]->fb_address, vfbs_[i]->width, vfbs_[i]->height); debug += std::string(temp); } - ERROR_LOG(HLE, "FBOs: %s", debug.c_str()); + ERROR_LOG(SCEGE, "FBOs: %s", debug.c_str()); #endif return 0; } @@ -321,7 +321,7 @@ void DrawingSize(int &drawing_width, int &drawing_height) { int scissor_height = gstate.getScissorY2() + 1; int fb_width = gstate.fbwidth & 0x3C0; - DEBUG_LOG(HLE,"viewport : %ix%i, region : %ix%i , scissor: %ix%i, stride: %i, %i", viewport_width,viewport_height, region_width, region_height, scissor_width, scissor_height, fb_width, gstate.isModeThrough()); + DEBUG_LOG(SCEGE,"viewport : %ix%i, region : %ix%i , scissor: %ix%i, stride: %i, %i", viewport_width,viewport_height, region_width, region_height, scissor_width, scissor_height, fb_width, gstate.isModeThrough()); // Viewport may return 0x0 for example FF Type-0 and we set it to 480x272 if (viewport_width <= 1 && viewport_height <=1) { @@ -372,7 +372,7 @@ void FramebufferManager::DestroyFramebuf(VirtualFramebuffer *v) { void FramebufferManager::SetRenderFrameBuffer() { if (!gstate_c.framebufChanged && currentRenderVfb_) { - currentRenderVfb_->last_frame_used = gpuStats.numFlips; + currentRenderVfb_->last_frame_render = gpuStats.numFlips; currentRenderVfb_->dirtyAfterDisplay = true; if (!gstate_c.skipDrawReason) currentRenderVfb_->reallyDirtyAfterDisplay = true; @@ -381,10 +381,10 @@ void FramebufferManager::SetRenderFrameBuffer() { gstate_c.framebufChanged = false; // Get parameters - u32 fb_address = (gstate.fbptr & 0xFFE000) | ((gstate.fbwidth & 0xFF0000) << 8); + u32 fb_address = (gstate.fbptr & 0xFFFFFF) | ((gstate.fbwidth & 0xFF0000) << 8); int fb_stride = gstate.fbwidth & 0x3C0; - u32 z_address = (gstate.zbptr & 0xFFE000) | ((gstate.zbwidth & 0xFF0000) << 8); + u32 z_address = (gstate.zbptr & 0xFFFFFF) | ((gstate.zbwidth & 0xFF0000) << 8); int z_stride = gstate.zbwidth & 0x3C0; // Yeah this is not completely right. but it'll do for now. @@ -411,6 +411,7 @@ void FramebufferManager::SetRenderFrameBuffer() { vfb = v; // Update fb stride in case it changed vfb->fb_stride = fb_stride; + vfb->format = fmt; if (v->bufferWidth >= drawing_width && v->bufferHeight >= drawing_height) { v->width = drawing_width; v->height = drawing_height; @@ -475,7 +476,7 @@ void FramebufferManager::SetRenderFrameBuffer() { if (vfb->fbo) { fbo_bind_as_render_target(vfb->fbo); } else { - ERROR_LOG(HLE, "Error creating FBO! %i x %i", vfb->renderWidth, vfb->renderHeight); + ERROR_LOG(SCEGE, "Error creating FBO! %i x %i", vfb->renderWidth, vfb->renderHeight); } } else { fbo_unbind(); @@ -485,14 +486,14 @@ void FramebufferManager::SetRenderFrameBuffer() { textureCache_->NotifyFramebuffer(vfb->fb_address, vfb, NOTIFY_FB_CREATED); - vfb->last_frame_used = gpuStats.numFlips; + vfb->last_frame_render = gpuStats.numFlips; frameLastFramebufUsed = gpuStats.numFlips; vfbs_.push_back(vfb); ClearBuffer(); currentRenderVfb_ = vfb; - INFO_LOG(HLE, "Creating FBO for %08x : %i x %i x %i", vfb->fb_address, vfb->width, vfb->height, vfb->format); + INFO_LOG(SCEGE, "Creating FBO for %08x : %i x %i x %i", vfb->fb_address, vfb->width, vfb->height, vfb->format); // We already have it! } else if (vfb != currentRenderVfb_) { @@ -505,10 +506,10 @@ void FramebufferManager::SetRenderFrameBuffer() { ReadFramebufferToMemory(vfb, true); } // Use it as a render target. - DEBUG_LOG(HLE, "Switching render target to FBO for %08x: %i x %i x %i ", vfb->fb_address, vfb->width, vfb->height, vfb->format); + DEBUG_LOG(SCEGE, "Switching render target to FBO for %08x: %i x %i x %i ", vfb->fb_address, vfb->width, vfb->height, vfb->format); vfb->usageFlags |= FB_USAGE_RENDERTARGET; gstate_c.textureChanged = true; - vfb->last_frame_used = gpuStats.numFlips; + vfb->last_frame_render = gpuStats.numFlips; frameLastFramebufUsed = gpuStats.numFlips; vfb->dirtyAfterDisplay = true; if ((gstate_c.skipDrawReason & SKIPDRAW_SKIPFRAME) == 0) @@ -552,13 +553,13 @@ void FramebufferManager::SetRenderFrameBuffer() { // to it. This broke stuff before, so now it only clears on the first use of an // FBO in a frame. This means that some games won't be able to avoid the on-some-GPUs // performance-crushing framebuffer reloads from RAM, but we'll have to live with that. - if (vfb->last_frame_used != gpuStats.numFlips) { + if (vfb->last_frame_render != gpuStats.numFlips) { ClearBuffer(); } #endif currentRenderVfb_ = vfb; } else { - vfb->last_frame_used = gpuStats.numFlips; + vfb->last_frame_render = gpuStats.numFlips; frameLastFramebufUsed = gpuStats.numFlips; vfb->dirtyAfterDisplay = true; if ((gstate_c.skipDrawReason & SKIPDRAW_SKIPFRAME) == 0) @@ -603,7 +604,7 @@ void FramebufferManager::CopyDisplayToOutput() { // The game is displaying something directly from RAM. In GTA, it's decoded video. DrawPixels(Memory::GetPointer(displayFramebufPtr_), displayFormat_, displayStride_); } else { - DEBUG_LOG(HLE, "Found no FBO to display! displayFBPtr = %08x", displayFramebufPtr_); + DEBUG_LOG(SCEGE, "Found no FBO to display! displayFBPtr = %08x", displayFramebufPtr_); // No framebuffer to display! Clear to black. ClearBuffer(); } @@ -628,7 +629,7 @@ void FramebufferManager::CopyDisplayToOutput() { if (vfb->fbo) { dxstate.viewport.set(0, 0, PSP_CoreParameter().pixelWidth, PSP_CoreParameter().pixelHeight); - DEBUG_LOG(HLE, "Displaying FBO %08x", vfb->fb_address); + DEBUG_LOG(SCEGE, "Displaying FBO %08x", vfb->fb_address); DisableState(); fbo_bind_color_as_texture(vfb->fbo, 0); @@ -712,17 +713,17 @@ void FramebufferManager::ReadFramebufferToMemory(VirtualFramebuffer *vfb, bool s nvfb->fbo = fbo_create(nvfb->width, nvfb->height, 1, true, nvfb->colorDepth); if (!(nvfb->fbo)) { - ERROR_LOG(HLE, "Error creating FBO! %i x %i", nvfb->renderWidth, nvfb->renderHeight); + ERROR_LOG(SCEGE, "Error creating FBO! %i x %i", nvfb->renderWidth, nvfb->renderHeight); return; } - nvfb->last_frame_used = gpuStats.numFlips; + nvfb->last_frame_render = gpuStats.numFlips; bvfbs_.push_back(nvfb); fbo_bind_as_render_target(nvfb->fbo); ClearBuffer(); } else { nvfb->usageFlags |= FB_USAGE_RENDERTARGET; - nvfb->last_frame_used = gpuStats.numFlips; + nvfb->last_frame_render = gpuStats.numFlips; nvfb->dirtyAfterDisplay = true; #if 0 @@ -732,7 +733,7 @@ void FramebufferManager::ReadFramebufferToMemory(VirtualFramebuffer *vfb, bool s // to it. This broke stuff before, so now it only clears on the first use of an // FBO in a frame. This means that some games won't be able to avoid the on-some-GPUs // performance-crushing framebuffer reloads from RAM, but we'll have to live with that. - if (nvfb->last_frame_used != gpuStats.numFlips) { + if (nvfb->last_frame_render != gpuStats.numFlips) { ClearBuffer(); } #endif @@ -888,7 +889,7 @@ void FramebufferManager::BeginFrame() { void FramebufferManager::SetDisplayFramebuffer(u32 framebuf, u32 stride, GEBufferFormat format) { if ((framebuf & 0x04000000) == 0) { - DEBUG_LOG(HLE, "Non-VRAM display framebuffer address set: %08x", framebuf); + DEBUG_LOG(SCEGE, "Non-VRAM display framebuffer address set: %08x", framebuf); ramDisplayFramebufPtr_ = framebuf; displayStride_ = stride; displayFormat_ = format; @@ -929,7 +930,7 @@ void FramebufferManager::DecimateFBOs() { #endif for (size_t i = 0; i < vfbs_.size(); ++i) { VirtualFramebuffer *vfb = vfbs_[i]; - int age = frameLastFramebufUsed - vfb->last_frame_used; + int age = frameLastFramebufUsed - std::max(vfb->last_frame_render, vfb->last_frame_used); if(useMem && age == 0 && !vfb->memoryUpdated) { ReadFramebufferToMemory(vfb); @@ -940,7 +941,7 @@ void FramebufferManager::DecimateFBOs() { } if (age > FBO_OLD_AGE) { - INFO_LOG(HLE, "Decimating FBO for %08x (%i x %i x %i), age %i", vfb->fb_address, vfb->width, vfb->height, vfb->format, age) + INFO_LOG(SCEGE, "Decimating FBO for %08x (%i x %i x %i), age %i", vfb->fb_address, vfb->width, vfb->height, vfb->format, age) DestroyFramebuf(vfb); vfbs_.erase(vfbs_.begin() + i--); } @@ -949,9 +950,9 @@ void FramebufferManager::DecimateFBOs() { // Do the same for ReadFramebuffersToMemory's VFBs for (size_t i = 0; i < bvfbs_.size(); ++i) { VirtualFramebuffer *vfb = bvfbs_[i]; - int age = frameLastFramebufUsed - vfb->last_frame_used; + int age = frameLastFramebufUsed - vfb->last_frame_render; if (age > FBO_OLD_AGE) { - INFO_LOG(HLE, "Decimating FBO for %08x (%i x %i x %i), age %i", vfb->fb_address, vfb->width, vfb->height, vfb->format, age) + INFO_LOG(SCEGE, "Decimating FBO for %08x (%i x %i x %i), age %i", vfb->fb_address, vfb->width, vfb->height, vfb->format, age) DestroyFramebuf(vfb); bvfbs_.erase(bvfbs_.begin() + i--); } @@ -967,7 +968,7 @@ void FramebufferManager::DestroyAllFBOs() { for (size_t i = 0; i < vfbs_.size(); ++i) { VirtualFramebuffer *vfb = vfbs_[i]; - INFO_LOG(HLE, "Destroying FBO for %08x : %i x %i x %i", vfb->fb_address, vfb->width, vfb->height, vfb->format); + INFO_LOG(SCEGE, "Destroying FBO for %08x : %i x %i x %i", vfb->fb_address, vfb->width, vfb->height, vfb->format); DestroyFramebuf(vfb); } vfbs_.clear(); @@ -998,7 +999,7 @@ void FramebufferManager::UpdateFromMemory(u32 addr, int size) { needUnbind = true; DrawPixels(Memory::GetPointer(addr), vfb->format, vfb->fb_stride); } else { - INFO_LOG(HLE, "Invalidating FBO for %08x (%i x %i x %i)", vfb->fb_address, vfb->width, vfb->height, vfb->format) + INFO_LOG(SCEGE, "Invalidating FBO for %08x (%i x %i x %i)", vfb->fb_address, vfb->width, vfb->height, vfb->format) DestroyFramebuf(vfb); vfbs_.erase(vfbs_.begin() + i--); } diff --git a/GPU/Directx9/Framebuffer.h b/GPU/Directx9/Framebuffer.h index d6d39f4c48..2569d47781 100644 --- a/GPU/Directx9/Framebuffer.h +++ b/GPU/Directx9/Framebuffer.h @@ -46,6 +46,7 @@ enum { struct VirtualFramebuffer { int last_frame_used; + int last_frame_render; bool memoryUpdated; u32 fb_address; diff --git a/GPU/Directx9/TextureCache.cpp b/GPU/Directx9/TextureCache.cpp index 634021bc22..a2423a26df 100644 --- a/GPU/Directx9/TextureCache.cpp +++ b/GPU/Directx9/TextureCache.cpp @@ -177,9 +177,13 @@ void TextureCache::ClearNextFrame() { template inline void AttachFramebufferValid(T &entry, VirtualFramebuffer *framebuffer) { + const bool hasInvalidFramebuffer = entry->framebuffer == 0 || entry->invalidHint == -1; + const bool hasOlderFramebuffer = entry->framebuffer != 0 && entry->framebuffer->last_frame_render < framebuffer->last_frame_render; + if (hasInvalidFramebuffer || hasOlderFramebuffer) { entry->framebuffer = framebuffer; entry->invalidHint = 0; } +} template inline void AttachFramebufferInvalid(T &entry, VirtualFramebuffer *framebuffer) { @@ -192,11 +196,12 @@ inline void AttachFramebufferInvalid(T &entry, VirtualFramebuffer *framebuffer) inline void TextureCache::AttachFramebuffer(TexCacheEntry *entry, u32 address, VirtualFramebuffer *framebuffer, bool exactMatch) { // If they match exactly, it's non-CLUT and from the top left. if (exactMatch) { - DEBUG_LOG(HLE, "Render to texture detected at %08x!", address); - if (!entry->framebuffer) { + DEBUG_LOG(G3D, "Render to texture detected at %08x!", address); + if (!entry->framebuffer || entry->invalidHint == -1) { if (entry->format != framebuffer->format) { - WARN_LOG_REPORT_ONCE(diffFormat1, HLE, "Render to texture with different formats %d != %d", entry->format, framebuffer->format); + WARN_LOG_REPORT_ONCE(diffFormat1, G3D, "Render to texture with different formats %d != %d", entry->format, framebuffer->format); // If it already has one, let's hope that one is correct. + // If "AttachFramebufferValid" , Evangelion Jo and Kurohyou 2 will be 'blue background' in-game AttachFramebufferInvalid(entry, framebuffer); } else { AttachFramebufferValid(entry, framebuffer); @@ -212,13 +217,16 @@ inline void TextureCache::AttachFramebuffer(TexCacheEntry *entry, u32 address, V // Is it at least the right stride? if (framebuffer->fb_stride == entry->bufw && compatFormat) { if (framebuffer->format != entry->format) { - WARN_LOG_REPORT_ONCE(diffFormat2, HLE, "Render to texture with different formats %d != %d at %08x", entry->format, framebuffer->format, address); + WARN_LOG_REPORT_ONCE(diffFormat2, G3D, "Render to texture with different formats %d != %d at %08x", entry->format, framebuffer->format, address); // TODO: Use an FBO to translate the palette? + // If 'AttachFramebufferInvalid' , Kurohyou 2 will be missing battle scene in-game and FF Type-0 will have black box shadow/'blue fog' and 3rd birthday will have 'blue fog' + // If 'AttachFramebufferValid' , DBZ VS Tag will have 'burning effect' , AttachFramebufferValid(entry, framebuffer); } else if ((entry->addr - address) / entry->bufw < framebuffer->height) { - WARN_LOG_REPORT_ONCE(subarea, HLE, "Render to area containing texture at %08x", address); + WARN_LOG_REPORT_ONCE(subarea, G3D, "Render to area containing texture at %08x", address); // TODO: Keep track of the y offset. - AttachFramebufferValid(entry, framebuffer); + // If "AttachFramebufferValid" , God of War Ghost of Sparta/Chains of Olympus will be missing special effect. + AttachFramebufferInvalid(entry, framebuffer); } } } @@ -243,6 +251,10 @@ void TextureCache::NotifyFramebuffer(u32 address, VirtualFramebuffer *framebuffe switch (msg) { case NOTIFY_FB_CREATED: case NOTIFY_FB_UPDATED: + // Ensure it's in the framebuffer cache. + if (std::find(fbCache_.begin(), fbCache_.end(), framebuffer) == fbCache_.end()) { + fbCache_.push_back(framebuffer); + } for (auto it = cache.lower_bound(cacheKey), end = cache.upper_bound(cacheKeyEnd); it != end; ++it) { AttachFramebuffer(&it->second, address | 0x04000000, framebuffer, it->first == cacheKey); } @@ -992,7 +1004,7 @@ bool SetDebugTexture() { bool changed = false; if (((gpuStats.numFlips / highlightFrames) % mostTextures) == numTextures) { if (gpuStats.numFlips % highlightFrames == 0) { - NOTICE_LOG(HLE, "Highlighting texture # %d / %d", numTextures, mostTextures); + NOTICE_LOG(G3D, "Highlighting texture # %d / %d", numTextures, mostTextures); } static const u32 solidTextureData[] = {0x99AA99FF}; @@ -1013,6 +1025,34 @@ bool SetDebugTexture() { } #endif +void TextureCache::SetTextureFramebuffer(TexCacheEntry *entry) +{ + entry->framebuffer->usageFlags |= FB_USAGE_TEXTURE; + bool useBufferedRendering = g_Config.iRenderingMode != FB_NON_BUFFERED_MODE; + if (useBufferedRendering) { + // For now, let's not bind FBOs that we know are off (invalidHint will be -1.) + // But let's still not use random memory. + if (entry->framebuffer->fbo && entry->invalidHint != -1) { + fbo_bind_color_as_texture(entry->framebuffer->fbo, 0); + // Keep the framebuffer alive. + // TODO: Dangerous if it sets a new one? + entry->framebuffer->last_frame_used = gpuStats.numFlips; + } else { + pD3Ddevice->SetTexture(0, NULL); + gstate_c.skipDrawReason |= SKIPDRAW_BAD_FB_TEXTURE; + } + UpdateSamplingParams(*entry, false); + gstate_c.curTextureWidth = entry->framebuffer->width; + gstate_c.curTextureHeight = entry->framebuffer->height; + gstate_c.flipTexture = true; + gstate_c.textureFullAlpha = entry->framebuffer->format == GE_FORMAT_565; + } else { + if (entry->framebuffer->fbo) + entry->framebuffer->fbo = 0; + pD3Ddevice->SetTexture(0, NULL); + } +} + void TextureCache::SetTexture() { #ifdef DEBUG_TEXTURES if (SetDebugTexture()) { @@ -1068,37 +1108,18 @@ void TextureCache::SetTexture() { if (iter != cache.end()) { entry = &iter->second; + // Validate the texture still matches the cache entry. + int dim = gstate.texsize[0] & 0xF0F; + bool match = entry->Matches(dim, format, maxLevel); + // Check for FBO - slow! - if (entry->framebuffer) { - entry->framebuffer->usageFlags |= FB_USAGE_TEXTURE; - if (useBufferedRendering) { - // For now, let's not bind FBOs that we know are off (invalidHint will be -1.) - // But let's still not use random memory. - if (entry->framebuffer->fbo && entry->invalidHint != -1) { - fbo_bind_color_as_texture(entry->framebuffer->fbo, 0); - } else { - pD3Ddevice->SetTexture(0, NULL); - gstate_c.skipDrawReason |= SKIPDRAW_BAD_FB_TEXTURE; - } - UpdateSamplingParams(*entry, false); - gstate_c.curTextureWidth = entry->framebuffer->width; - gstate_c.curTextureHeight = entry->framebuffer->height; - gstate_c.flipTexture = true; - gstate_c.textureFullAlpha = entry->framebuffer->format == GE_FORMAT_565; - } else { - if (entry->framebuffer->fbo) - entry->framebuffer->fbo = 0; - pD3Ddevice->SetTexture(0, NULL); - } + if (entry->framebuffer && match) { + SetTextureFramebuffer(entry); lastBoundTexture = INVALID_TEX; entry->lastFrame = gpuStats.numFlips; return; } - //Validate the texture here (width, height etc) - - int dim = gstate.texsize[0] & 0xF0F; - bool match = entry->Matches(dim, format, maxLevel); bool rehash = (entry->status & TexCacheEntry::STATUS_MASK) == TexCacheEntry::STATUS_UNRELIABLE; bool doDelete = true; @@ -1176,12 +1197,12 @@ void TextureCache::SetTexture() { gstate_c.textureFullAlpha = (entry->status & TexCacheEntry::STATUS_ALPHA_MASK) == TexCacheEntry::STATUS_ALPHA_FULL; } UpdateSamplingParams(*entry, false); - DEBUG_LOG(G3D, "Texture at %08x Found in Cache, applying", texaddr); + VERBOSE_LOG(G3D, "Texture at %08x Found in Cache, applying", texaddr); return; //Done! } else { entry->numInvalidated++; gpuStats.numTextureInvalidations++; - INFO_LOG(G3D, "Texture different or overwritten, reloading at %08x", texaddr); + DEBUG_LOG(G3D, "Texture different or overwritten, reloading at %08x", texaddr); if (doDelete) { if (entry->maxLevel == maxLevel && entry->dim == (gstate.texsize[0] & 0xF0F) && entry->format == format && g_Config.iTexScalingLevel <= 1) { // Actually, if size and number of levels match, let's try to avoid deleting and recreating. @@ -1199,7 +1220,7 @@ void TextureCache::SetTexture() { } } } else { - INFO_LOG(G3D, "No texture in cache, decoding..."); + VERBOSE_LOG(G3D, "No texture in cache, decoding..."); TexCacheEntry entryNew = {0}; cache[cachekey] = entryNew; @@ -1208,7 +1229,7 @@ void TextureCache::SetTexture() { } if ((bufw == 0 || (gstate.texbufwidth[0] & 0xf800) != 0) && texaddr >= PSP_GetUserMemoryBase()) { - ERROR_LOG_REPORT(HLE, "Texture with unexpected bufw (full=%d)", gstate.texbufwidth[0] & 0xffff); + ERROR_LOG_REPORT(G3D, "Texture with unexpected bufw (full=%d)", gstate.texbufwidth[0] & 0xffff); } // We have to decode it, let's setup the cache entry first. @@ -1235,13 +1256,30 @@ void TextureCache::SetTexture() { gstate_c.curTextureWidth = w; gstate_c.curTextureHeight = h; - /* - if (!replaceImages) { - pD3Ddevice->CreateTexture(w, h, 1, 0, D3DFMT(D3DFMT_A8R8G8B8), NULL, &entry->texture, NULL); + // Before we go reading the texture from memory, let's check for render-to-texture. + for (size_t i = 0, n = fbCache_.size(); i < n; ++i) { + auto framebuffer = fbCache_[i]; + // This is a rough heuristic, because sometimes our framebuffers are too tall. + static const u32 MAX_SUBAREA_Y_OFFSET = 32; + + // Must be in VRAM so | 0x04000000 it is. + const u64 cacheKeyStart = (u64)(framebuffer->fb_address | 0x04000000) << 32; + // If it has a clut, those are the low 32 bits, so it'll be inside this range. + // Also, if it's a subsample of the buffer, it'll also be within the FBO. + const u64 cacheKeyEnd = cacheKeyStart + ((u64)(framebuffer->fb_stride * MAX_SUBAREA_Y_OFFSET) << 32); + + if (cachekey >= cacheKeyStart && cachekey < cacheKeyEnd) { + AttachFramebuffer(entry, framebuffer->fb_address | 0x04000000, framebuffer, cachekey == cacheKeyStart); + } + } + + // If we ended up with a framebuffer, attach it - no texture decoding needed. + if (entry->framebuffer) { + SetTextureFramebuffer(entry); + lastBoundTexture = INVALID_TEX; + entry->lastFrame = gpuStats.numFlips; + return; } - pD3Ddevice->SetTexture(0, entry->texture); - lastBoundTexture = entry->texture; - */ // Adjust maxLevel to actually present levels.. for (int i = 0; i <= maxLevel; i++) { @@ -1264,11 +1302,6 @@ void TextureCache::SetTexture() { UpdateSamplingParams(*entry, true); - //glPixelStorei(GL_UNPACK_ROW_LENGTH, 0); - //glPixelStorei(GL_UNPACK_ALIGNMENT, 1); - //glPixelStorei(GL_PACK_ROW_LENGTH, 0); - //glPixelStorei(GL_PACK_ALIGNMENT, 1); - gstate_c.textureFullAlpha = (entry->status & TexCacheEntry::STATUS_ALPHA_MASK) == TexCacheEntry::STATUS_ALPHA_FULL; } @@ -1341,7 +1374,7 @@ void *TextureCache::DecodeTextureLevel(GETextureFormat format, GEPaletteFormat c break; default: - ERROR_LOG(G3D, "Unknown CLUT4 texture mode %d", gstate.getClutPaletteFormat()); + ERROR_LOG_REPORT(G3D, "Unknown CLUT4 texture mode %d", gstate.getClutPaletteFormat()); return NULL; } } @@ -1657,15 +1690,6 @@ void TextureCache::LoadTextureLevel(TexCacheEntry &entry, int level, bool replac gpuStats.numTexturesDecoded++; - // Can restore these and remove the fixup at the end of DecodeTextureLevel on desktop GL and GLES 3. - // glPixelStorei(GL_UNPACK_ROW_LENGTH, bufw); - // glPixelStorei(GL_PACK_ROW_LENGTH, bufw); - - //glPixelStorei(GL_UNPACK_ALIGNMENT, texByteAlign); - //glPixelStorei(GL_PACK_ALIGNMENT, texByteAlign); - - // INFO_LOG(G3D, "Creating texture level %i/%i from %08x: %i x %i (stride: %i). fmt: %i", level, entry.maxLevel, texaddr, w, h, bufw, entry.format); - u32 *pixelData = (u32 *)finalBuf; int scaleFactor = g_Config.iTexScalingLevel; diff --git a/GPU/Directx9/TextureCache.h b/GPU/Directx9/TextureCache.h index e1f215ac49..e781d4cc8a 100644 --- a/GPU/Directx9/TextureCache.h +++ b/GPU/Directx9/TextureCache.h @@ -127,12 +127,14 @@ private: void UpdateCurrentClut(); void AttachFramebuffer(TexCacheEntry *entry, u32 address, VirtualFramebuffer *framebuffer, bool exactMatch); void DetachFramebuffer(TexCacheEntry *entry, u32 address, VirtualFramebuffer *framebuffer); + void SetTextureFramebuffer(TexCacheEntry *entry); TexCacheEntry *GetEntryAt(u32 texaddr); typedef std::map TexCache; TexCache cache; TexCache secondCache; + std::vector fbCache_; bool clearCacheNextFrame_; bool lowMemoryMode_; diff --git a/GPU/Directx9/TransformPipeline.cpp b/GPU/Directx9/TransformPipeline.cpp index fe44495b79..4add469b7f 100644 --- a/GPU/Directx9/TransformPipeline.cpp +++ b/GPU/Directx9/TransformPipeline.cpp @@ -665,8 +665,7 @@ void TransformDrawEngine::SoftwareTransformAndDraw( reader.ReadUV(ruv); // Perform texture coordinate generation after the transform and lighting - one style of UV depends on lights. - switch (gstate.getUVGenMode()) - { + switch (gstate.getUVGenMode()) { case GE_TEXMAP_TEXTURE_COORDS: // UV mapping case GE_TEXMAP_UNKNOWN: // Seen in Riviera. Unsure of meaning, but this works. // Texture scale/offset is only performed in this mode. @@ -674,18 +673,20 @@ void TransformDrawEngine::SoftwareTransformAndDraw( uv[1] = vscale * (ruv[1]*gstate_c.uv.vScale + gstate_c.uv.vOff); uv[2] = 1.0f; break; + case GE_TEXMAP_TEXTURE_MATRIX: { // Projection mapping Vec3f source; - switch (gstate.getUVProjMode()) - { + switch (gstate.getUVProjMode()) { case GE_PROJMAP_POSITION: // Use model space XYZ as source source = pos; break; + case GE_PROJMAP_UV: // Use unscaled UV as source source = Vec3f(ruv[0], ruv[1], 0.0f); break; + case GE_PROJMAP_NORMALIZED_NORMAL: // Use normalized normal as source if (reader.hasNormal()) { source = Vec3f(norm).Normalized(); @@ -694,6 +695,7 @@ void TransformDrawEngine::SoftwareTransformAndDraw( source = Vec3f(0.0f, 0.0f, 1.0f); } break; + case GE_PROJMAP_NORMAL: // Use non-normalized normal as source! if (reader.hasNormal()) { source = Vec3f(norm); @@ -711,6 +713,7 @@ void TransformDrawEngine::SoftwareTransformAndDraw( uv[2] = uvw[2]; } break; + case GE_TEXMAP_ENVIRONMENT_MAP: // Shade mapping - use two light sources to generate U and V. { @@ -722,8 +725,10 @@ void TransformDrawEngine::SoftwareTransformAndDraw( uv[2] = 1.0f; } break; + default: // Illegal + ERROR_LOG_REPORT(G3D, "Impossible UV gen mode? %d", gstate.getUVGenMode()); break; } uv[0] = uv[0] * widthFactor; @@ -884,7 +889,13 @@ void TransformDrawEngine::SubmitPrim(void *verts, void *inds, GEPrimitiveType pr if (!indexGen.PrimCompatible(prevPrim_, prim) || numDrawCalls >= MAX_DEFERRED_DRAW_CALLS) Flush(); + + // TODO: Is this the right thing to do? + if (prim == GE_PRIM_KEEP_PREVIOUS) { + prim = prevPrim_; + } prevPrim_ = prim; + SetupVertexDecoder(vertType); dec_->IncrementStat(STAT_VERTSSUBMITTED, vertexCount); @@ -982,7 +993,7 @@ void TransformDrawEngine::DecodeVerts() { // Sanity check if (indexGen.Prim() < 0) { - ERROR_LOG(HLE, "DecodeVerts: Failed to deduce prim: %i", indexGen.Prim()); + ERROR_LOG_REPORT(G3D, "DecodeVerts: Failed to deduce prim: %i", indexGen.Prim()); // Force to points (0) indexGen.AddPrim(GE_PRIM_POINTS, 0); } @@ -1071,7 +1082,7 @@ void TransformDrawEngine::DecimateTrackedVertexArrays() { char *ptr = buffer; ptr += dec->second->ToString(ptr); // *ptr++ = '\n'; - NOTICE_LOG(HLE, buffer); + NOTICE_LOG(G3D, buffer); } #endif } diff --git a/GPU/Directx9/VertexDecoder.cpp b/GPU/Directx9/VertexDecoder.cpp index e8f3b49ed6..d295c65afe 100644 --- a/GPU/Directx9/VertexDecoder.cpp +++ b/GPU/Directx9/VertexDecoder.cpp @@ -929,7 +929,7 @@ void VertexDecoder::SetVertexType(u32 fmt) { decOff += DecFmtSize(decFmt.nrmfmt); } - //if (pos) - there's always a position + if (pos) // there's always a position { size = align(size, posalign[pos]); posoff = size; @@ -958,7 +958,9 @@ void VertexDecoder::SetVertexType(u32 fmt) { } decFmt.posoff = decOff; decOff += DecFmtSize(decFmt.posfmt); - } + } else + ERROR_LOG_REPORT(G3D, "Vertices without position found") + decFmt.stride = decOff; size = align(size, biggest); diff --git a/GPU/Directx9/VertexDecoder.h b/GPU/Directx9/VertexDecoder.h index acfc8ad75b..e15fc59728 100644 --- a/GPU/Directx9/VertexDecoder.h +++ b/GPU/Directx9/VertexDecoder.h @@ -95,6 +95,7 @@ public: // prim is needed knowledge for a performance hack (PrescaleUV) void SetVertexType(u32 vtype); u32 VertexType() const { return fmt_; } + const DecVtxFormat &GetDecVtxFmt() { return decFmt; } void DecodeVerts(u8 *decoded, const void *verts, int indexLowerBound, int indexUpperBound) const; @@ -219,6 +220,7 @@ public: // Integer value passed in a float. Wraps and all, required for Monster Hunter. pos[2] = (float)((u16)(s32)pos[2]) * (1.0f / 65535.0f); } + // See https://github.com/hrydgard/ppsspp/pull/3419, something is weird. } break; case DEC_S16_3: @@ -252,7 +254,8 @@ public: } break; default: - ERROR_LOG(G3D, "Reader: Unsupported Pos Format"); + ERROR_LOG_REPORT_ONCE(fmt, G3D, "Reader: Unsupported Pos Format %d", decFmt_.posfmt); + memset(pos, 0, sizeof(float) * 3); break; } } @@ -282,7 +285,8 @@ public: } break; default: - ERROR_LOG(G3D, "Reader: Unsupported Nrm Format"); + ERROR_LOG_REPORT_ONCE(fmt, G3D, "Reader: Unsupported Nrm Format %d", decFmt_.nrmfmt); + memset(nrm, 0, sizeof(float) * 3); break; } } @@ -313,6 +317,14 @@ public: } break; + case DEC_U8A_2: + { + const u8 *b = (const u8 *)(data_ + decFmt_.uvoff); + uv[0] = (float)b[0]; + uv[1] = (float)b[1]; + } + break; + case DEC_U16A_2: { const u16 *p = (const u16 *)(data_ + decFmt_.uvoff); @@ -321,7 +333,8 @@ public: } break; default: - ERROR_LOG(G3D, "Reader: Unsupported UV Format"); + ERROR_LOG_REPORT_ONCE(fmt, G3D, "Reader: Unsupported UV Format %d", decFmt_.uvfmt); + memset(uv, 0, sizeof(float) * 2); break; } } @@ -339,7 +352,8 @@ public: memcpy(color, data_ + decFmt_.c0off, 16); break; default: - ERROR_LOG(G3D, "Reader: Unsupported C0 Format"); + ERROR_LOG_REPORT_ONCE(fmt, G3D, "Reader: Unsupported C0 Format %d", decFmt_.c0fmt); + memset(color, 0, sizeof(float) * 4); break; } } @@ -357,7 +371,8 @@ public: memcpy(color, data_ + decFmt_.c1off, 12); break; default: - ERROR_LOG(G3D, "Reader: Unsupported C1 Format"); + ERROR_LOG_REPORT_ONCE(fmt, G3D, "Reader: Unsupported C1 Format %d", decFmt_.c1fmt); + memset(color, 0, sizeof(float) * 3); break; } } @@ -383,7 +398,8 @@ public: case DEC_U16_3: for (int i = 0; i < 3; i++) weights[i] = s[i] * (1.f / 32768.f); break; case DEC_U16_4: for (int i = 0; i < 4; i++) weights[i] = s[i] * (1.f / 32768.f); break; default: - ERROR_LOG(G3D, "Reader: Unsupported W0 Format"); + ERROR_LOG_REPORT_ONCE(fmt0, G3D, "Reader: Unsupported W0 Format %d", decFmt_.w0fmt); + memset(weights, 0, sizeof(float) * 4); break; } @@ -410,7 +426,8 @@ public: case DEC_U16_3: for (int i = 0; i < 3; i++) weights[i+4] = s[i] * (1.f / 32768.f); break; case DEC_U16_4: for (int i = 0; i < 4; i++) weights[i+4] = s[i] * (1.f / 32768.f); break; default: - ERROR_LOG(G3D, "Reader: Unsupported W1 Format"); + ERROR_LOG_REPORT_ONCE(fmt1, G3D, "Reader: Unsupported W1 Format %d", decFmt_.w1fmt); + memset(weights + 4, 0, sizeof(float) * 4); break; } } diff --git a/GPU/Directx9/VertexShaderGenerator.cpp b/GPU/Directx9/VertexShaderGenerator.cpp index df1f949e6e..7f1b5489a8 100644 --- a/GPU/Directx9/VertexShaderGenerator.cpp +++ b/GPU/Directx9/VertexShaderGenerator.cpp @@ -54,7 +54,7 @@ void ComputeVertexShaderID(VertexShaderID *id, int prim, bool useHWTransform) { bool hasColor = (vertType & GE_VTYPE_COL_MASK) != 0; bool hasNormal = (vertType & GE_VTYPE_NRM_MASK) != 0; - bool hasBones = gstate.getWeightMask() != 0; + bool hasBones = gstate.getWeightMask() != GE_VTYPE_WEIGHT_NONE; bool enableFog = gstate.isFogEnabled() && !gstate.isModeThrough() && !gstate.isModeClear(); bool lmode = gstate.isUsingSecondaryColor() && gstate.isLightingEnabled();