From 9f7e0978a97da14de6df89ccadb18388d50cf404 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Mon, 11 Apr 2022 20:10:22 +0200 Subject: [PATCH] AND together colors while decoding, and then check against fullAlphaMask. --- Common/Data/Convert/ColorConv.h | 4 + GPU/Common/TextureCacheCommon.cpp | 145 +++++++++++++++++++++++------- GPU/Common/TextureCacheCommon.h | 4 +- GPU/Common/TextureDecoder.cpp | 23 ++++- GPU/Common/TextureDecoder.h | 37 +++++--- GPU/D3D11/TextureCacheD3D11.cpp | 12 +-- GPU/Directx9/TextureCacheDX9.cpp | 11 +-- GPU/GLES/TextureCacheGLES.cpp | 11 +-- GPU/Vulkan/TextureCacheVulkan.cpp | 16 ++-- 9 files changed, 194 insertions(+), 69 deletions(-) diff --git a/Common/Data/Convert/ColorConv.h b/Common/Data/Convert/ColorConv.h index 7ba629c637..66bd47b9cf 100644 --- a/Common/Data/Convert/ColorConv.h +++ b/Common/Data/Convert/ColorConv.h @@ -99,6 +99,10 @@ void convert5551_dx9(u16* data, u32* out, int width, int l, int u); // TODO: Need to revisit the naming convention of these. Seems totally backwards // now that we've standardized on Draw::DataFormat. +// +// The functions that have the same bit width of input and output can generally +// tolerate being called with src == dst, which is used a lot for ReverseColors +// in the GLES backend. void ConvertBGRA8888ToRGBA8888(u32 *dst, const u32 *src, u32 numPixels); #define ConvertRGBA8888ToBGRA8888 ConvertBGRA8888ToRGBA8888 diff --git a/GPU/Common/TextureCacheCommon.cpp b/GPU/Common/TextureCacheCommon.cpp index 7397b22c96..84bbed70be 100644 --- a/GPU/Common/TextureCacheCommon.cpp +++ b/GPU/Common/TextureCacheCommon.cpp @@ -1306,6 +1306,8 @@ ReplacedTexture &TextureCacheCommon::FindReplacement(TexCacheEntry *entry, int & return replacer_.FindNone(); } +// This is only used in the GLES backend, where we don't point these to video memory. +// So we shouldn't add a check for dstBuf != srcBuf, as long as the functions we call can handle that. static void ReverseColors(void *dstBuf, const void *srcBuf, GETextureFormat fmt, int numPixels, bool useBGRA) { switch (fmt) { case GE_TFMT_4444: @@ -1353,7 +1355,10 @@ static inline void ConvertFormatToRGBA8888(GEPaletteFormat format, u32 *dst, con } template -static void DecodeDXTBlock(uint8_t *out, int outPitch, uint32_t texaddr, const uint8_t *texptr, int w, int h, int bufw, bool reverseColors, bool useBGRA) { +static void DecodeDXTBlocks(uint8_t *out, int outPitch, uint32_t texaddr, const uint8_t *texptr, + int w, int h, int bufw, bool reverseColors, bool useBGRA, + u32 *alphaSum, u32 *fullAlphaMask) { + int minw = std::min(bufw, w); uint32_t *dst = (uint32_t *)out; int outPitch32 = outPitch / sizeof(uint32_t); @@ -1366,26 +1371,88 @@ static void DecodeDXTBlock(uint8_t *out, int outPitch, uint32_t texaddr, const u h = (((int)limited / sizeof(DXTBlock)) / (bufw / 4)) * 4; } + *fullAlphaMask = 1; // we just use one bit here. + + u32 alpha = 0xFFFFFFFF; for (int y = 0; y < h; y += 4) { u32 blockIndex = (y / 4) * (bufw / 4); int blockHeight = std::min(h - y, 4); for (int x = 0; x < minw; x += 4) { - if (n == 1) - DecodeDXT1Block(dst + outPitch32 * y + x, (const DXT1Block *)src + blockIndex, outPitch32, blockHeight, false); - if (n == 3) + switch (n) { + case 1: + DecodeDXT1Block(dst + outPitch32 * y + x, (const DXT1Block *)src + blockIndex, outPitch32, blockHeight, &alpha); + break; + case 3: DecodeDXT3Block(dst + outPitch32 * y + x, (const DXT3Block *)src + blockIndex, outPitch32, blockHeight); - if (n == 5) + break; + case 5: DecodeDXT5Block(dst + outPitch32 * y + x, (const DXT5Block *)src + blockIndex, outPitch32, blockHeight); + break; + } blockIndex++; } } + + switch (n) { + case 1: + *alphaSum = alpha; + break; + case 3: + case 5: + // Just report that we don't have full alpha, since these formats are made for that. + *alphaSum = 0; + break; + } + w = (w + 3) & ~3; if (reverseColors) { ReverseColors(out, out, GE_TFMT_8888, outPitch32 * h, useBGRA); } } -void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureFormat format, GEPaletteFormat clutformat, uint32_t texaddr, int level, int bufw, bool reverseColors, bool useBGRA, bool expandTo32bit) { +inline u32 ClutFormatToFullAlpha(GEPaletteFormat fmt) { + switch (fmt) { + case GE_CMODE_16BIT_ABGR4444: return 0xF000; + case GE_CMODE_16BIT_ABGR5551: return 0x8000; + case GE_CMODE_16BIT_BGR5650: return 0x0000; + case GE_CMODE_32BIT_ABGR8888: return 0xFF000000; + } + return 0; +} + +inline u32 TfmtRawToFullAlpha(GETextureFormat fmt) { + switch (fmt) { + case GE_TFMT_4444: return 0xF000; + case GE_TFMT_5551: return 0x8000; + case GE_TFMT_5650: return 0x0000; + case GE_TFMT_8888: return 0xFF000000; + } + return 0; +} + +// TODO: SSE/SIMD +// At least on x86, compiler actually parallelizes these pretty well. +void CopyAndSumMask16(u16 *dst, const u16 *src, int width, u32 *outMask) { + u16 mask = 0xFFFF; + for (int i = 0; i < width; i++) { + u16 color = src[i]; + mask &= color; + dst[i] = color; + } + *outMask &= (u32)mask; +} + +void CopyAndSumMask32(u32 *dst, const u32 *src, int width, u32 *outMask) { + u32 mask = 0xFFFFFFFF; + for (int i = 0; i < width; i++) { + u32 color = src[i]; + mask &= color; + dst[i] = color; + } + *outMask &= (u32)mask; +} + +void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureFormat format, GEPaletteFormat clutformat, uint32_t texaddr, int level, int bufw, bool reverseColors, bool useBGRA, bool expandTo32bit, u32 *alphaSum, u32 *fullAlphaMask) { bool swizzled = gstate.isTextureSwizzled(); if ((texaddr & 0x00600000) != 0 && Memory::IsVRAMAddress(texaddr)) { // This means it's in a mirror, possibly a swizzled mirror. Let's report. @@ -1424,6 +1491,7 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm case GE_CMODE_16BIT_ABGR4444: { if (clutAlphaLinear_ && mipmapShareClut && !expandTo32bit) { + // We don't bother with fullalpha here // Here, reverseColors means the CLUT is already reversed. if (reverseColors) { for (int y = 0; y < h; ++y) { @@ -1439,12 +1507,14 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm if (expandTo32bit && !reverseColors) { // We simply expand the CLUT to 32-bit, then we deindex as usual. Probably the fastest way. ConvertFormatToRGBA8888(clutformat, expandClut_, clut, 16); + *fullAlphaMask = 0xFF000000; for (int y = 0; y < h; ++y) { - DeIndexTexture4((u32 *)(out + outPitch * y), texptr + (bufw * y) / 2, w, expandClut_); + DeIndexTexture4((u32 *)(out + outPitch * y), texptr + (bufw * y) / 2, w, expandClut_, alphaSum); } } else { + *fullAlphaMask = ClutFormatToFullAlpha(clutformat); for (int y = 0; y < h; ++y) { - DeIndexTexture4((u16 *)(out + outPitch * y), texptr + (bufw * y) / 2, w, clut); + DeIndexTexture4((u16 *)(out + outPitch * y), texptr + (bufw * y) / 2, w, clut, alphaSum); } } } @@ -1454,8 +1524,9 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm case GE_CMODE_32BIT_ABGR8888: { const u32 *clut = GetCurrentClut() + clutSharingOffset; + *fullAlphaMask = 0xFF000000; for (int y = 0; y < h; ++y) { - DeIndexTexture4((u32 *)(out + outPitch * y), texptr + (bufw * y) / 2, w, clut); + DeIndexTexture4((u32 *)(out + outPitch * y), texptr + (bufw * y) / 2, w, clut, alphaSum); } } break; @@ -1468,15 +1539,15 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm break; case GE_TFMT_CLUT8: - ReadIndexedTex(out, outPitch, level, texptr, 1, bufw, expandTo32bit); + ReadIndexedTex(out, outPitch, level, texptr, 1, bufw, expandTo32bit, alphaSum, fullAlphaMask); break; case GE_TFMT_CLUT16: - ReadIndexedTex(out, outPitch, level, texptr, 2, bufw, expandTo32bit); + ReadIndexedTex(out, outPitch, level, texptr, 2, bufw, expandTo32bit, alphaSum, fullAlphaMask); break; case GE_TFMT_CLUT32: - ReadIndexedTex(out, outPitch, level, texptr, 4, bufw, expandTo32bit); + ReadIndexedTex(out, outPitch, level, texptr, 4, bufw, expandTo32bit, alphaSum, fullAlphaMask); break; case GE_TFMT_4444: @@ -1485,41 +1556,48 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm if (!swizzled) { // Just a simple copy, we swizzle the color format. if (reverseColors) { + // TODO: Handle alpha mask for (int y = 0; y < h; ++y) { ReverseColors(out + outPitch * y, texptr + bufw * sizeof(u16) * y, format, w, useBGRA); } } else if (expandTo32bit) { + // TODO: Handle alpha mask for (int y = 0; y < h; ++y) { ConvertFormatToRGBA8888(format, (u32 *)(out + outPitch * y), (const u16 *)texptr + bufw * y, w); } } else { + *fullAlphaMask = TfmtRawToFullAlpha(format); for (int y = 0; y < h; ++y) { - memcpy(out + outPitch * y, texptr + bufw * sizeof(u16) * y, w * sizeof(u16)); + CopyAndSumMask16((u16 *)(out + outPitch * y), (u16 *)(texptr + bufw * sizeof(u16) * y), w, alphaSum); } } - } else if (h >= 8 && bufw <= w && !expandTo32bit) { + } /* else if (h >= 8 && bufw <= w && !expandTo32bit) { + // TODO: Handle alpha mask. This will require special versions of UnswizzleFromMem to keep the optimization. // Note: this is always safe since h must be a power of 2, so a multiple of 8. UnswizzleFromMem((u32 *)out, outPitch, texptr, bufw, h, 2); if (reverseColors) { ReverseColors(out, out, format, h * outPitch / 2, useBGRA); } - } else { + }*/ else { // We don't have enough space for all rows in out, so use a temp buffer. tmpTexBuf32_.resize(bufw * ((h + 7) & ~7)); UnswizzleFromMem(tmpTexBuf32_.data(), bufw * 2, texptr, bufw, h, 2); const u8 *unswizzled = (u8 *)tmpTexBuf32_.data(); if (reverseColors) { + // TODO: Handle alpha mask for (int y = 0; y < h; ++y) { ReverseColors(out + outPitch * y, unswizzled + bufw * sizeof(u16) * y, format, w, useBGRA); } } else if (expandTo32bit) { + // TODO: Handle alpha mask for (int y = 0; y < h; ++y) { ConvertFormatToRGBA8888(format, (u32 *)(out + outPitch * y), (const u16 *)unswizzled + bufw * y, w); } } else { + *fullAlphaMask = TfmtRawToFullAlpha(format); for (int y = 0; y < h; ++y) { - memcpy(out + outPitch * y, unswizzled + bufw * sizeof(u16) * y, w * sizeof(u16)); + CopyAndSumMask16((u16 *)(out + outPitch * y), (const u16 *)(unswizzled + bufw * sizeof(u16) * y), w, alphaSum); } } } @@ -1528,47 +1606,51 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm case GE_TFMT_8888: if (!swizzled) { if (reverseColors) { + *fullAlphaMask = 0; // ignore alpha optimization for now for (int y = 0; y < h; ++y) { ReverseColors(out + outPitch * y, texptr + bufw * sizeof(u32) * y, format, w, useBGRA); } } else { + *fullAlphaMask = TfmtRawToFullAlpha(format); for (int y = 0; y < h; ++y) { - memcpy(out + outPitch * y, texptr + bufw * sizeof(u32) * y, w * sizeof(u32)); + CopyAndSumMask32((u32 *)(out + outPitch * y), (const u32 *)(texptr + bufw * sizeof(u32) * y), w * sizeof(u32), alphaSum); } } - } else if (h >= 8 && bufw <= w) { + } /* else if (h >= 8 && bufw <= w) { + // TODO: Handle alpha mask UnswizzleFromMem((u32 *)out, outPitch, texptr, bufw, h, 4); if (reverseColors) { ReverseColors(out, out, format, h * outPitch / 4, useBGRA); } - } else { - // We don't have enough space for all rows in out, so use a temp buffer. + }*/ else { tmpTexBuf32_.resize(bufw * ((h + 7) & ~7)); UnswizzleFromMem(tmpTexBuf32_.data(), bufw * 4, texptr, bufw, h, 4); const u8 *unswizzled = (u8 *)tmpTexBuf32_.data(); if (reverseColors) { + // TODO: Handle alpha mask for (int y = 0; y < h; ++y) { ReverseColors(out + outPitch * y, unswizzled + bufw * sizeof(u32) * y, format, w, useBGRA); } } else { + *fullAlphaMask = TfmtRawToFullAlpha(format); for (int y = 0; y < h; ++y) { - memcpy(out + outPitch * y, unswizzled + bufw * sizeof(u32) * y, w * sizeof(u32)); + CopyAndSumMask32((u32 *)(out + outPitch * y), (const u32 *)(unswizzled + bufw * sizeof(u32) * y), w * sizeof(u32), alphaSum); } } } break; case GE_TFMT_DXT1: - DecodeDXTBlock(out, outPitch, texaddr, texptr, w, h, bufw, reverseColors, useBGRA); + DecodeDXTBlocks(out, outPitch, texaddr, texptr, w, h, bufw, reverseColors, useBGRA, alphaSum, fullAlphaMask); break; case GE_TFMT_DXT3: - DecodeDXTBlock(out, outPitch, texaddr, texptr, w, h, bufw, reverseColors, useBGRA); + DecodeDXTBlocks(out, outPitch, texaddr, texptr, w, h, bufw, reverseColors, useBGRA, alphaSum, fullAlphaMask); break; case GE_TFMT_DXT5: - DecodeDXTBlock(out, outPitch, texaddr, texptr, w, h, bufw, reverseColors, useBGRA); + DecodeDXTBlocks(out, outPitch, texaddr, texptr, w, h, bufw, reverseColors, useBGRA, alphaSum, fullAlphaMask); break; default: @@ -1577,7 +1659,7 @@ void TextureCacheCommon::DecodeTextureLevel(u8 *out, int outPitch, GETextureForm } } -void TextureCacheCommon::ReadIndexedTex(u8 *out, int outPitch, int level, const u8 *texptr, int bytesPerIndex, int bufw, bool expandTo32Bit) { +void TextureCacheCommon::ReadIndexedTex(u8 *out, int outPitch, int level, const u8 *texptr, int bytesPerIndex, int bufw, bool expandTo32Bit, u32 *alphaSum, u32 *fullAlphaMask) { int w = gstate.getTextureWidth(level); int h = gstate.getTextureHeight(level); @@ -1596,6 +1678,7 @@ void TextureCacheCommon::ReadIndexedTex(u8 *out, int outPitch, int level, const ConvertFormatToRGBA8888(GEPaletteFormat(palFormat), expandClut_, clut16, 256); clut32 = expandClut_; palFormat = GE_CMODE_32BIT_ABGR8888; + *fullAlphaMask = 0xFF000000; } switch (palFormat) { @@ -1606,19 +1689,19 @@ void TextureCacheCommon::ReadIndexedTex(u8 *out, int outPitch, int level, const switch (bytesPerIndex) { case 1: for (int y = 0; y < h; ++y) { - DeIndexTexture((u16 *)(out + outPitch * y), (const u8 *)texptr + bufw * y, w, clut16); + DeIndexTexture((u16 *)(out + outPitch * y), (const u8 *)texptr + bufw * y, w, clut16, alphaSum); } break; case 2: for (int y = 0; y < h; ++y) { - DeIndexTexture((u16 *)(out + outPitch * y), (const u16_le *)texptr + bufw * y, w, clut16); + DeIndexTexture((u16 *)(out + outPitch * y), (const u16_le *)texptr + bufw * y, w, clut16, alphaSum); } break; case 4: for (int y = 0; y < h; ++y) { - DeIndexTexture((u16 *)(out + outPitch * y), (const u32_le *)texptr + bufw * y, w, clut16); + DeIndexTexture((u16 *)(out + outPitch * y), (const u32_le *)texptr + bufw * y, w, clut16, alphaSum); } break; } @@ -1630,19 +1713,19 @@ void TextureCacheCommon::ReadIndexedTex(u8 *out, int outPitch, int level, const switch (bytesPerIndex) { case 1: for (int y = 0; y < h; ++y) { - DeIndexTexture((u32 *)(out + outPitch * y), (const u8 *)texptr + bufw * y, w, clut32); + DeIndexTexture((u32 *)(out + outPitch * y), (const u8 *)texptr + bufw * y, w, clut32, alphaSum); } break; case 2: for (int y = 0; y < h; ++y) { - DeIndexTexture((u32 *)(out + outPitch * y), (const u16_le *)texptr + bufw * y, w, clut32); + DeIndexTexture((u32 *)(out + outPitch * y), (const u16_le *)texptr + bufw * y, w, clut32, alphaSum); } break; case 4: for (int y = 0; y < h; ++y) { - DeIndexTexture((u32 *)(out + outPitch * y), (const u32_le *)texptr + bufw * y, w, clut32); + DeIndexTexture((u32 *)(out + outPitch * y), (const u32_le *)texptr + bufw * y, w, clut32, alphaSum); } break; } diff --git a/GPU/Common/TextureCacheCommon.h b/GPU/Common/TextureCacheCommon.h index efaa1cbef5..caaea2fcf9 100644 --- a/GPU/Common/TextureCacheCommon.h +++ b/GPU/Common/TextureCacheCommon.h @@ -275,9 +275,9 @@ protected: virtual void UpdateCurrentClut(GEPaletteFormat clutFormat, u32 clutBase, bool clutIndexIsSimple) = 0; bool CheckFullHash(TexCacheEntry *entry, bool &doDelete); - void DecodeTextureLevel(u8 *out, int outPitch, GETextureFormat format, GEPaletteFormat clutformat, uint32_t texaddr, int level, int bufw, bool reverseColors, bool useBGRA, bool expandTo32Bit); + void DecodeTextureLevel(u8 *out, int outPitch, GETextureFormat format, GEPaletteFormat clutformat, uint32_t texaddr, int level, int bufw, bool reverseColors, bool useBGRA, bool expandTo32Bit, u32 *alphaSum, u32 *fullAlphaMask); void UnswizzleFromMem(u32 *dest, u32 destPitch, const u8 *texptr, u32 bufw, u32 height, u32 bytesPerPixel); - void ReadIndexedTex(u8 *out, int outPitch, int level, const u8 *texptr, int bytesPerIndex, int bufw, bool expandTo32Bit); + void ReadIndexedTex(u8 *out, int outPitch, int level, const u8 *texptr, int bytesPerIndex, int bufw, bool expandTo32Bit, u32 *alphaSum, u32 *fullAlphaMask); ReplacedTexture &FindReplacement(TexCacheEntry *entry, int &w, int &h); template diff --git a/GPU/Common/TextureDecoder.cpp b/GPU/Common/TextureDecoder.cpp index a00b5ee529..08c0068702 100644 --- a/GPU/Common/TextureDecoder.cpp +++ b/GPU/Common/TextureDecoder.cpp @@ -432,9 +432,13 @@ public: inline void WriteColorsDXT3(u32 *dst, const DXT3Block *src, int pitch, int height); inline void WriteColorsDXT5(u32 *dst, const DXT5Block *src, int pitch, int height); + bool AnyNonFullAlpha() const { return anyNonFullAlpha_; } + protected: u32 colors_[4]; u8 alpha_[8]; + bool alphaMode_ = false; + bool anyNonFullAlpha_ = false; }; static inline u32 makecol(int r, int g, int b, int a) { @@ -471,6 +475,9 @@ void DXTDecoder::DecodeColors(const DXT1Block *src, bool ignore1bitAlpha) { int blue3 = (blue1 + blue2) / 2; colors_[2] = makecol(red3, green3, blue3, alpha); colors_[3] = makecol(0, 0, 0, 0); + if (alpha == 255) { + alphaMode_ = true; + } } } @@ -508,14 +515,23 @@ void DXTDecoder::DecodeAlphaDXT5(const DXT5Block *src) { } void DXTDecoder::WriteColorsDXT1(u32 *dst, const DXT1Block *src, int pitch, int height) { + bool anyColor3 = false; for (int y = 0; y < height; y++) { int colordata = src->lines[y]; for (int x = 0; x < 4; x++) { - dst[x] = colors_[colordata & 3]; + int col = colordata & 3; + if (col == 3) { + anyColor3 = true; + } + dst[x] = colors_[col]; colordata >>= 2; } dst += pitch; } + + if (alphaMode_ && anyColor3) { + anyNonFullAlpha_ = true; + } } void DXTDecoder::WriteColorsDXT3(u32 *dst, const DXT3Block *src, int pitch, int height) { @@ -610,10 +626,11 @@ uint32_t GetDXT5Texel(const DXT5Block *src, int x, int y) { } // This could probably be done faster by decoding two or four blocks at a time with SSE/NEON. -void DecodeDXT1Block(u32 *dst, const DXT1Block *src, int pitch, int height, bool ignore1bitAlpha) { +void DecodeDXT1Block(u32 *dst, const DXT1Block *src, int pitch, int height, u32 *alpha) { DXTDecoder dxt; - dxt.DecodeColors(src, ignore1bitAlpha); + dxt.DecodeColors(src, false); dxt.WriteColorsDXT1(dst, src, pitch, height); + *alpha = dxt.AnyNonFullAlpha() ? 0 : 1; } void DecodeDXT3Block(u32 *dst, const DXT3Block *src, int pitch, int height) { diff --git a/GPU/Common/TextureDecoder.h b/GPU/Common/TextureDecoder.h index 6a0e59fea3..f85adbec55 100644 --- a/GPU/Common/TextureDecoder.h +++ b/GPU/Common/TextureDecoder.h @@ -65,7 +65,7 @@ struct DXT5Block { u8 alpha1; u8 alpha2; }; -void DecodeDXT1Block(u32 *dst, const DXT1Block *src, int pitch, int height, bool ignore1bitAlpha); +void DecodeDXT1Block(u32 *dst, const DXT1Block *src, int pitch, int height, u32 *alpha); void DecodeDXT3Block(u32 *dst, const DXT3Block *src, int pitch, int height); void DecodeDXT5Block(u32 *dst, const DXT5Block *src, int pitch, int height); @@ -94,15 +94,23 @@ static const u8 textureBitsPerPixel[16] = { u32 GetTextureBufw(int level, u32 texaddr, GETextureFormat format); +inline bool AlphaSumIsFull(u32 alphaSum, u32 fullAlphaMask) { + return fullAlphaMask != 0 && (alphaSum & fullAlphaMask) == fullAlphaMask; +} + template -inline void DeIndexTexture(ClutT *dest, const IndexT *indexed, int length, const ClutT *clut) { +inline void DeIndexTexture(/*WRITEONLY*/ ClutT *dest, const IndexT *indexed, int length, const ClutT *clut, u32 *outAlphaSum) { // Usually, there is no special offset, mask, or shift. const bool nakedIndex = gstate.isClutIndexSimple(); + ClutT alphaSum = (ClutT)(-1); + if (nakedIndex) { if (sizeof(IndexT) == 1) { for (int i = 0; i < length; ++i) { - *dest++ = clut[*indexed++]; + ClutT color = clut[*indexed++]; + *dest++ = color; + } } else { for (int i = 0; i < length; ++i) { @@ -117,29 +125,38 @@ inline void DeIndexTexture(ClutT *dest, const IndexT *indexed, int length, const } template -inline void DeIndexTexture(ClutT *dest, const u32 texaddr, int length, const ClutT *clut) { +inline void DeIndexTexture(/*WRITEONLY*/ ClutT *dest, const u32 texaddr, int length, const ClutT *clut, u32 *outAlphaSum) { const IndexT *indexed = (const IndexT *) Memory::GetPointer(texaddr); - DeIndexTexture(dest, indexed, length, clut); + DeIndexTexture(dest, indexed, length, clut, outAlphaSum); } template -inline void DeIndexTexture4(ClutT *dest, const u8 *indexed, int length, const ClutT *clut) { +inline void DeIndexTexture4(/*WRITEONLY*/ ClutT *dest, const u8 *indexed, int length, const ClutT *clut, u32 *outAlphaSum) { // Usually, there is no special offset, mask, or shift. const bool nakedIndex = gstate.isClutIndexSimple(); + ClutT alphaSum = (ClutT)(-1); if (nakedIndex) { for (int i = 0; i < length; i += 2) { u8 index = *indexed++; - dest[i + 0] = clut[(index >> 0) & 0xf]; - dest[i + 1] = clut[(index >> 4) & 0xf]; + ClutT color0 = clut[index & 0xf]; + ClutT color1 = clut[index >> 4]; + dest[i + 0] = color0; + dest[i + 1] = color1; + alphaSum &= color0 & color1; } } else { for (int i = 0; i < length; i += 2) { u8 index = *indexed++; - dest[i + 0] = clut[gstate.transformClutIndex((index >> 0) & 0xf)]; - dest[i + 1] = clut[gstate.transformClutIndex((index >> 4) & 0xf)]; + ClutT color0 = clut[gstate.transformClutIndex((index >> 0) & 0xf)]; + ClutT color1 = clut[gstate.transformClutIndex((index >> 4) & 0xf)]; + dest[i + 0] = color0; + dest[i + 1] = color1; + alphaSum &= color0 & color1; } } + + *outAlphaSum &= (u32)alphaSum; } template diff --git a/GPU/D3D11/TextureCacheD3D11.cpp b/GPU/D3D11/TextureCacheD3D11.cpp index 722eae29b4..c34bb96871 100644 --- a/GPU/D3D11/TextureCacheD3D11.cpp +++ b/GPU/D3D11/TextureCacheD3D11.cpp @@ -708,16 +708,18 @@ void TextureCacheD3D11::LoadTextureLevel(TexCacheEntry &entry, ReplacedTexture & } bool expand32 = !gstate_c.Supports(GPU_SUPPORTS_16BIT_FORMATS); - DecodeTextureLevel((u8 *)pixelData, decPitch, tfmt, clutformat, texaddr, level, bufw, false, false, expand32); + u32 fullAlphaMask = 0; + u32 alphaSum = 0xFFFFFFFF; + DecodeTextureLevel((u8 *)pixelData, decPitch, tfmt, clutformat, texaddr, level, bufw, false, false, expand32, &alphaSum, &fullAlphaMask); // We check before scaling since scaling shouldn't invent alpha from a full alpha texture. - if ((entry.status & TexCacheEntry::STATUS_CHANGE_FREQUENT) == 0) { - TexCacheEntry::TexStatus alphaStatus = CheckAlpha(pixelData, dstFmt, decPitch / bpp, w, h); - entry.SetAlphaStatus(alphaStatus, level); + if (AlphaSumIsFull(alphaSum, fullAlphaMask)) { + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_FULL, level); } else { - entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN); + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN, level); } + if (scaleFactor > 1) { u32 scaleFmt = (u32)dstFmt; scaler.ScaleAlways((u32 *)mapData, pixelData, scaleFmt, w, h, scaleFactor); diff --git a/GPU/Directx9/TextureCacheDX9.cpp b/GPU/Directx9/TextureCacheDX9.cpp index 23334baf8b..37d3392102 100644 --- a/GPU/Directx9/TextureCacheDX9.cpp +++ b/GPU/Directx9/TextureCacheDX9.cpp @@ -634,14 +634,15 @@ void TextureCacheDX9::LoadTextureLevel(TexCacheEntry &entry, ReplacedTexture &re decPitch = w * bpp; } - DecodeTextureLevel((u8 *)pixelData, decPitch, tfmt, clutformat, texaddr, level, bufw, false, false, false); + u32 fullAlphaMask = 0; + u32 alphaSum = 0xFFFFFFFF; + DecodeTextureLevel((u8 *)pixelData, decPitch, tfmt, clutformat, texaddr, level, bufw, false, false, false, &alphaSum, &fullAlphaMask); // We check before scaling since scaling shouldn't invent alpha from a full alpha texture. - if ((entry.status & TexCacheEntry::STATUS_CHANGE_FREQUENT) == 0) { - TexCacheEntry::TexStatus alphaStatus = CheckAlpha(pixelData, dstFmt, decPitch / bpp, w, h); - entry.SetAlphaStatus(alphaStatus, level); + if (AlphaSumIsFull(alphaSum, fullAlphaMask)) { + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_FULL, level); } else { - entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN); + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN, level); } if (scaleFactor > 1) { diff --git a/GPU/GLES/TextureCacheGLES.cpp b/GPU/GLES/TextureCacheGLES.cpp index df9d3f88b9..0e53525689 100644 --- a/GPU/GLES/TextureCacheGLES.cpp +++ b/GPU/GLES/TextureCacheGLES.cpp @@ -675,14 +675,15 @@ void TextureCacheGLES::LoadTextureLevel(TexCacheEntry &entry, ReplacedTexture &r decPitch = std::max(w * pixelSize, 4); pixelData = (uint8_t *)AllocateAlignedMemory(decPitch * h * pixelSize, 16); - DecodeTextureLevel(pixelData, decPitch, GETextureFormat(entry.format), clutformat, texaddr, level, bufw, true, false, false); + u32 fullAlphaMask = 0; + u32 alphaSum = 0xFFFFFFFF; + DecodeTextureLevel(pixelData, decPitch, GETextureFormat(entry.format), clutformat, texaddr, level, bufw, true, false, false, &alphaSum, &fullAlphaMask); // We check before scaling since scaling shouldn't invent alpha from a full alpha texture. - if ((entry.status & TexCacheEntry::STATUS_CHANGE_FREQUENT) == 0) { - TexCacheEntry::TexStatus alphaStatus = CheckAlpha(pixelData, dstFmt, decPitch / pixelSize, w, h); - entry.SetAlphaStatus(alphaStatus, level); + if (AlphaSumIsFull(alphaSum, fullAlphaMask)) { + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_FULL, level); } else { - entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN); + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN, level); } if (scaleFactor > 1) { diff --git a/GPU/Vulkan/TextureCacheVulkan.cpp b/GPU/Vulkan/TextureCacheVulkan.cpp index afb8a5ba40..cc0c95499c 100644 --- a/GPU/Vulkan/TextureCacheVulkan.cpp +++ b/GPU/Vulkan/TextureCacheVulkan.cpp @@ -980,17 +980,17 @@ void TextureCacheVulkan::LoadTextureLevel(TexCacheEntry &entry, uint8_t *writePt } bool expand32 = !gstate_c.Supports(GPU_SUPPORTS_16BIT_FORMATS) || dstFmt == VK_FORMAT_R8G8B8A8_UNORM; - DecodeTextureLevel((u8 *)pixelData, decPitch, tfmt, clutformat, texaddr, level, bufw, false, false, expand32); + + u32 alphaSum = 0xFFFFFFFF; + u32 fullAlphaMask = 0x0; + + DecodeTextureLevel((u8 *)pixelData, decPitch, tfmt, clutformat, texaddr, level, bufw, false, false, expand32, &alphaSum, &fullAlphaMask); gpuStats.numTexturesDecoded++; - // We check before scaling since scaling shouldn't invent alpha from a full alpha texture. - if ((entry.status & TexCacheEntry::STATUS_CHANGE_FREQUENT) == 0) { - // TODO: When we decode directly, this can be more expensive (maybe not on mobile?) - // This does allow us to skip alpha testing, though. - TexCacheEntry::TexStatus alphaStatus = CheckAlpha(pixelData, dstFmt, decPitch / bpp, w, h); - entry.SetAlphaStatus(alphaStatus, level); + if (AlphaSumIsFull(alphaSum, fullAlphaMask)) { + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_FULL, level); } else { - entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN); + entry.SetAlphaStatus(TexCacheEntry::STATUS_ALPHA_UNKNOWN, level); } if (scaleFactor > 1) {