diff --git a/Common/GPU/ShaderWriter.cpp b/Common/GPU/ShaderWriter.cpp index 2445dbea18..4d6df0b476 100644 --- a/Common/GPU/ShaderWriter.cpp +++ b/Common/GPU/ShaderWriter.cpp @@ -37,6 +37,8 @@ const char * const hlsl_preamble_fs = "#define inversesqrt rsqrt\n" "#define floatBitsToUint asuint\n" "#define uintBitsToFloat asfloat\n" +"#define floatBitsToInt asint\n" +"#define intBitsToFloat asfloat\n" "\n"; static const char * const hlsl_d3d11_preamble_fs = @@ -71,6 +73,8 @@ static const char * const hlsl_preamble_vs = "#define inversesqrt rsqrt\n" "#define floatBitsToUint asuint\n" "#define uintBitsToFloat asfloat\n" +"#define floatBitsToInt asint\n" +"#define intBitsToFloat asfloat\n" "\n"; static const char * const semanticNames[] = { diff --git a/GPU/Common/ShaderId.cpp b/GPU/Common/ShaderId.cpp index b3013e9619..6231168ed2 100644 --- a/GPU/Common/ShaderId.cpp +++ b/GPU/Common/ShaderId.cpp @@ -116,8 +116,22 @@ void ComputeVertexShaderID(VShaderID *id_out, u32 vertType, bool useHWTransform, id.SetBits(VS_BIT_LS1, 2, gstate.getUVLS1()); } + if (doShadeMapping) { + // Shade mapping depends on the type of its lights and whether they do specular, even when + // they're off. The ubershader reads that from u_lightControl instead. + if (gstate_c.Use(GPU_USE_LIGHT_UBERSHADER)) { + id.SetBit(VS_BIT_LIGHT_UBERSHADER); + } else { + const int shadeLights[2] = { gstate.getUVLS0(), gstate.getUVLS1() }; + for (int l : shadeLights) { + id.SetBits(VS_BIT_LIGHT0_COMP + 4 * l, 2, gstate.getLightComputation(l)); + id.SetBits(VS_BIT_LIGHT0_TYPE + 4 * l, 2, gstate.getLightType(l)); + } + } + } + if (gstate.isLightingEnabled()) { - // doShadeMapping is stored as UVGenMode, and light type doesn't matter for shade mapping. + // doShadeMapping is stored as UVGenMode. id.SetBit(VS_BIT_LIGHTING_ENABLE); if (gstate_c.Use(GPU_USE_LIGHT_UBERSHADER)) { id.SetBit(VS_BIT_LIGHT_UBERSHADER); diff --git a/GPU/Common/ShaderUniforms.cpp b/GPU/Common/ShaderUniforms.cpp index 748aef9802..3b727720ad 100644 --- a/GPU/Common/ShaderUniforms.cpp +++ b/GPU/Common/ShaderUniforms.cpp @@ -12,6 +12,7 @@ #include "GPU/GPUState.h" #include "GPU/Common/FramebufferManagerCommon.h" #include "GPU/Common/GPUStateUtils.h" +#include "GPU/Common/TransformCommon.h" #include "GPU/Math3D.h" using namespace Lin; @@ -222,7 +223,7 @@ void LightUpdateUniforms(UB_VS_Lights *ub, uint64_t dirtyUniforms) { Uint8x3ToFloat4(ub->materialDiffuse, gstate.materialdiffuse); } if (dirtyUniforms & DIRTY_MATSPECULAR) { - Uint8x3ToFloat4_Alpha(ub->materialSpecular, gstate.materialspecular, std::max(0.0f, getFloat24(gstate.materialspecularcoef))); + Uint8x3ToFloat4_Alpha(ub->materialSpecular, gstate.materialspecular, std::max(0.0f, PSPSpecularCoef(getFloat24(gstate.materialspecularcoef)))); } if (dirtyUniforms & DIRTY_MATEMISSIVE) { // We're not touching the fourth f32 here, because we store an u32 of control bits in it. diff --git a/GPU/Common/SoftwareTransformCommon.cpp b/GPU/Common/SoftwareTransformCommon.cpp index 319f6bb07a..879282e0c0 100644 --- a/GPU/Common/SoftwareTransformCommon.cpp +++ b/GPU/Common/SoftwareTransformCommon.cpp @@ -333,23 +333,11 @@ SoftwareTransformAction RunSoftwareTransform(SoftwareTransformParams ¶ms, in case GE_TEXMAP_ENVIRONMENT_MAP: // Shade mapping - use two light sources to generate U and V. { - auto getLPosFloat = [&](int l, int i) { - return getFloat24(gstate.lpos[l * 3 + i]); - }; - auto getLPos = [&](int l) { - return Vec3f(getLPosFloat(l, 0), getLPosFloat(l, 1), getLPosFloat(l, 2)); - }; - auto calcShadingLPos = [&](int l) { - Vec3f pos = getLPos(l); - return pos.NormalizedOr001(cpu_info.bSSE4_1); - }; - // Might not have lighting enabled, so don't use lighter. - Vec3f lightpos0 = calcShadingLPos(gstate.getUVLS0()); - Vec3f lightpos1 = calcShadingLPos(gstate.getUVLS1()); - - uv[0] = (1.0f + Dot(lightpos0, worldnormal))/2.0f; - uv[1] = (1.0f + Dot(lightpos1, worldnormal))/2.0f; + const Vec3f viewDir = PSPViewDirection(gstate.viewMatrix); + const Vec3f worldpos(out[0], out[1], out[2]); + uv[0] = PSPShadeMapCoord(gstate.getUVLS0(), worldpos, worldnormal, viewDir); + uv[1] = PSPShadeMapCoord(gstate.getUVLS1(), worldpos, worldnormal, viewDir); uv[2] = 1.0f; } break; diff --git a/GPU/Common/TransformCommon.cpp b/GPU/Common/TransformCommon.cpp index 56b27a48b1..8668637569 100644 --- a/GPU/Common/TransformCommon.cpp +++ b/GPU/Common/TransformCommon.cpp @@ -40,8 +40,8 @@ Lighter::Lighter(int vertType) { materialDiffuse.a = 1.0f; materialSpecular.GetFromRGB(gstate.materialspecular); materialSpecular.a = 1.0f; - specCoef_ = getFloat24(gstate.materialspecularcoef); - // viewer_ = Vec3f(-gstate.viewMatrix[9], -gstate.viewMatrix[10], -gstate.viewMatrix[11]); + specCoef_ = PSPSpecularCoef(getFloat24(gstate.materialspecularcoef)); + viewDir_ = PSPViewDirection(gstate.viewMatrix); bool hasColor = (vertType & GE_VTYPE_COL_MASK) != 0; materialUpdate_ = hasColor ? (gstate.materialupdate & 7) : 0; @@ -117,11 +117,13 @@ void Lighter::Light(float colorOut0[4], float colorOut1[4], const float colorIn[ toLight /= distanceToLight; dot = Dot(toLight, norm); } + // Specular only applies when the light is in front of the surface. + const bool facingLight = dot >= 0.0f; // Clamp dot to zero. if (dot < 0.0f) dot = 0.0f; if (poweredDiffuse) - dot = powf(dot, specCoef_); + dot = PSPLightPow(dot, specCoef_); // Attenuation switch (type) { @@ -136,7 +138,7 @@ void Lighter::Light(float colorOut0[4], float colorOut1[4], const float colorIn[ lightDir = ldir[l]; angle = Dot(toLight.NormalizedOr001(cpu_info.bSSE4_1), lightDir.NormalizedOr001(cpu_info.bSSE4_1)); if (angle >= lcutoff[l]) - lightScale = clamp(1.0f / (latt[l].x + latt[l].y * distanceToLight + latt[l].z * distanceToLight*distanceToLight), 0.0f, 1.0f) * powf(angle, lconv[l]); + lightScale = clamp(1.0f / (latt[l].x + latt[l].y * distanceToLight + latt[l].z * distanceToLight*distanceToLight), 0.0f, 1.0f) * PSPLightPow(angle, lconv[l]); break; default: // ILLEGAL @@ -146,18 +148,13 @@ void Lighter::Light(float colorOut0[4], float colorOut1[4], const float colorIn[ Color4 lightDiff(lcolor[1][l], 0.0f); Color4 diff = (lightDiff * *diffuse) * dot; - // Real PSP specular - static const Vec3f toViewer(0, 0, 1); - // Better specular - // Vec3f toViewer = (viewer - pos).NormalizedOr001(cpu_info.bSSE4_1); - - if (doSpecular) { - Vec3f halfVec = (toLight + toViewer).NormalizedOr001(cpu_info.bSSE4_1); + if (doSpecular && facingLight) { + Vec3f halfVec = (toLight + viewDir_).NormalizedOr001(cpu_info.bSSE4_1); dot = Dot(halfVec, norm); if (dot > 0.0f) { Color4 lightSpec(lcolor[2][l], 0.0f); - lightSum1 += (lightSpec * *specular * (powf(dot, specCoef_) * lightScale)); + lightSum1 += (lightSpec * *specular * (PSPLightPow(dot, specCoef_) * lightScale)); } } diff --git a/GPU/Common/TransformCommon.h b/GPU/Common/TransformCommon.h index de4a366f23..2d03c1e999 100644 --- a/GPU/Common/TransformCommon.h +++ b/GPU/Common/TransformCommon.h @@ -17,11 +17,13 @@ #pragma once +#include #include #include "Common/CommonTypes.h" #include "GPU/Math3D.h" #include "GPU/GPU.h" +#include "GPU/GPUState.h" struct Color4 { float r, g, b, a; @@ -60,6 +62,59 @@ struct Color4 { } }; +// The GE's pow() for specular, powered diffuse and the spot exponent: 1 for e <= 0, else 0 for +// x <= 0. Otherwise exp2(e * log2(x)) with log2 and exp2 each a straight line between powers of two +// (Mitchell's approximation), which is what reading a float's bits as an integer gives: exponent +// plus mantissa, scaled by 2^23. Matches hardware within one step of 255 (gpu/lighting/specular). +inline float PSPLightPow(float x, float e) { + if (!(x > 0.0f)) { + return e > 0.0f ? 0.0f : 1.0f; + } + int32_t ix; + memcpy(&ix, &x, sizeof(ix)); + float t = (e > 0.0f ? e : 0.0f) * (float)(ix - 0x3F800000) + 1065353216.0f; + // Also turns NaN into 0, and stays below infinity's bits. + t = t >= 0.0f ? (t < 2139095039.0f ? t : 2139095039.0f) : 0.0f; + int32_t iy = (int32_t)t; + float y; + memcpy(&y, &iy, sizeof(y)); + return y; +} + +// The GE only uses the top 4 bits of the specular coefficient's mantissa. +inline float PSPSpecularCoef(float e) { + u32 bits; + memcpy(&bits, &e, sizeof(bits)); + bits &= 0xFFF80000; + memcpy(&e, &bits, sizeof(bits)); + return e; +} + +// The viewer is at infinity along view space +z, so in world space it's the view matrix's third column. +inline Vec3f PSPViewDirection(const float viewMatrix[12]) { + return Vec3f(viewMatrix[2], viewMatrix[5], viewMatrix[8]).NormalizedOr001(false); +} + +inline Vec3f NormalizedOr000(const Vec3f &v) { + float len2 = v.Length2(); + return len2 > 0.0f ? v * (1.0f / sqrtf(len2)) : Vec3f(0.0f, 0.0f, 0.0f); +} + +// Shade mapping (environment map UV gen) coordinate from light l: (N.L + 1) / 2, with L the light's +// direction as lighting sees it (a zero vector stays zero), or the half vector if the light does +// specular. Lighting and light enables don't matter (gpu/lighting/shademap). +inline float PSPShadeMapCoord(int l, const Vec3f &worldpos, const Vec3f &worldnormal, const Vec3f &viewDir) { + Vec3f L(getFloat24(gstate.lpos[l * 3]), getFloat24(gstate.lpos[l * 3 + 1]), getFloat24(gstate.lpos[l * 3 + 2])); + if (gstate.getLightType(l) != GE_LIGHTTYPE_DIRECTIONAL) { + L -= worldpos; + } + L = NormalizedOr000(L); + if (gstate.isUsingSpecularLight(l)) { + L = NormalizedOr000(L + viewDir); + } + return (Dot(L, worldnormal) + 1.0f) * 0.5f; +} + // Convenient way to do precomputation to save the parts of the lighting calculation // that's common between the many vertices of a draw call. class Lighter { @@ -81,7 +136,7 @@ private: Color4 materialDiffuse; Color4 materialSpecular; float specCoef_; - // Vec3f viewer_; + Vec3f viewDir_; bool doShadeMapping_; int materialUpdate_; diff --git a/GPU/Common/VertexShaderGenerator.cpp b/GPU/Common/VertexShaderGenerator.cpp index 7799e30bdf..b93ebd085f 100644 --- a/GPU/Common/VertexShaderGenerator.cpp +++ b/GPU/Common/VertexShaderGenerator.cpp @@ -135,7 +135,9 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag int matUpdate = id.Bits(VS_BIT_MATERIAL_UPDATE, 3); bool lightUberShader = id.Bit(VS_BIT_LIGHT_UBERSHADER) && enableLighting; // checking lighting here for the shader test's benefit, in reality if ubershader is set, lighting is set. - if (lightUberShader && !compat.bitwiseOps) { + // With the ubershader, shade mapping reads its lights' type and computation from u_lightControl. + bool shadeUberShader = id.Bit(VS_BIT_LIGHT_UBERSHADER) && doShadeMapping; + if ((lightUberShader || shadeUberShader) && !compat.bitwiseOps) { *errorString = "Light ubershader requires bitwise ops in shader language"; return false; } @@ -349,7 +351,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag WRITE(p, "uniform vec4 u_uvscaleoffset;\n"); *uniformMask |= DIRTY_UVSCALEOFFSET; - if (lightUberShader) { + if (lightUberShader || shadeUberShader) { p.C("uniform uint u_lightControl;\n"); *uniformMask |= DIRTY_LIGHT_CONTROL; } @@ -428,6 +430,22 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag WRITE(p, " float len2 = dot(v, v);\n"); WRITE(p, " return len2 == 0.0 ? vec3(0.0, 0.0, 1.0) : (v * inversesqrt(len2));\n"); WRITE(p, "}\n"); + WRITE(p, "vec3 normalizeOr000(vec3 v) {\n"); + WRITE(p, " float len2 = dot(v, v);\n"); + WRITE(p, " return len2 == 0.0 ? splat3(0.0) : (v * inversesqrt(len2));\n"); + WRITE(p, "}\n"); + // The GE's pow for lighting: 1 for e <= 0, else 0 for x <= 0. Otherwise exp2(e * log2(x)) with + // log2 and exp2 each a straight line between powers of two, which is what reading a float's + // bits as an integer gives: exponent plus mantissa, scaled by 2^23. Without integers, a true + // pow is close enough. + WRITE(p, "float pspPow(float x, float e) {\n"); + if (compat.bitwiseOps) { + WRITE(p, " float t = max(e, 0.0) * float(floatBitsToInt(max(x, 1e-30)) - 0x3F800000) + 1065353216.0;\n"); + WRITE(p, " return x > 0.0 || e <= 0.0 ? intBitsToFloat(int(max(t, 0.0))) : 0.0;\n"); + } else { + WRITE(p, " return e <= 0.0 ? 1.0 : pow(max(x, 0.0), e);\n"); + } + WRITE(p, "}\n"); } if (ShaderLanguageIsOpenGL(compat.shaderLanguage) || compat.shaderLanguage == GLSL_VULKAN) { @@ -490,6 +508,14 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag } else { WRITE(p, " mediump vec3 worldnormal = normalizeOr001(mul(vec4(0.0, 0.0, %s1.0, 0.0), u_world).xyz);\n", flipNormal ? "-" : ""); } + if (enableLighting || doShadeMapping) { + // The viewer is at infinity along view space +z: in world space, the view matrix's third column. + if (compat.shaderLanguage == HLSL_D3D11) { + WRITE(p, " mediump vec3 viewDir = normalizeOr001(vec3(u_view[0].z, u_view[1].z, u_view[2].z));\n"); + } else { + WRITE(p, " mediump vec3 viewDir = normalizeOr001(u_view[2].xyz);\n"); + } + } WRITE(p, " vec4 viewPos = vec4(mul(vec4(worldpos, 1.0), u_view).xyz, 1.0);\n"); if (useSimpleStereo) { @@ -643,7 +669,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag p.C(" } else {\n"); // type must be 0x02 - GE_LIGHTTYPE_SPOT p.F(" angle = dot(u_lightdir%s, toLight);\n", iStr); p.F(" if (angle >= u_lightangle_spotCoef%s.x) {\n", iStr); - p.F(" lightScale = attenuation * (u_lightangle_spotCoef%s.y <= 0.0 ? 1.0 : pow(angle, u_lightangle_spotCoef%s.y));\n", iStr, iStr, iStr); + p.F(" lightScale = attenuation * pspPow(angle, u_lightangle_spotCoef%s.y);\n", iStr, iStr); p.C(" } else {\n"); p.C(" lightScale = 0.0;\n"); p.C(" }\n"); @@ -653,17 +679,13 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag p.C(" }\n"); p.C(" ldot = dot(toLight, worldnormal);\n"); p.C(" if (comp == 0x2u) {\n"); // GE_LIGHTCOMP_ONLYPOWDIFFUSE - p.C(" ldot = u_matspecular.a > 0.0 ? pow(max(ldot, 0.0), u_matspecular.a) : 1.0;\n"); + p.C(" ldot = pspPow(ldot, u_matspecular.a);\n"); p.C(" }\n"); p.F(" diffuse = (u_lightdiffuse%s * diffuseColor) * max(ldot, 0.0);\n", iStr); p.C(" if (comp == 0x1u && ldot >= 0.0) {\n"); // do specular. note - must allow for the >= case, since the u_matspecular.a <= 0.0 case relies on it. - p.C(" if (u_matspecular.a > 0.0) {\n"); - p.C(" vec3 halfVec = toLight + vec3(0.0, 0.0, 1.0);\n"); - p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n"); - p.C(" ldot = pow(max(dot(halfVec, worldnormal) * halfInvLen, 0.0), u_matspecular.a);\n"); - p.C(" } else {\n"); - p.C(" ldot = 1.0;\n"); - p.C(" }\n"); + p.C(" vec3 halfVec = toLight + viewDir;\n"); + p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n"); + p.C(" ldot = pspPow(dot(halfVec, worldnormal) * halfInvLen, u_matspecular.a);\n"); p.F(" lightSum1 += u_lightspecular%s * specularColor * ldot * lightScale;\n", iStr); p.C(" }\n"); p.F(" lightSum0.rgb += (u_lightambient%s * ambientColor.rgb + diffuse) * lightScale;\n", iStr); @@ -703,11 +725,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag if (poweredDiffuse) { // pow(0.0, 0.0) may be undefined, but the PSP seems to treat it as 1.0. // Seen in Tales of the World: Radiant Mythology (#2424.) - p.C(" if (u_matspecular.a > 0.0) {\n"); - p.C(" ldot = pow(max(ldot, 0.0), u_matspecular.a);\n"); - p.C(" } else {\n"); - p.C(" ldot = 1.0;\n"); - p.C(" }\n"); + p.C(" ldot = pspPow(ldot, u_matspecular.a);\n"); } const char *timesLightScale = " * lightScale"; @@ -724,7 +742,7 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag case GE_LIGHTTYPE_UNKNOWN: p.F(" angle = dot(u_lightdir%s, toLight);\n", iStr, iStr); p.F(" if (angle >= u_lightangle_spotCoef%s.x) {\n", iStr); - p.F(" lightScale = clamp(1.0 / dot(u_lightatt%s, vec3(1.0, distance, distSq)), 0.0, 1.0) * (u_lightangle_spotCoef%s.y <= 0.0 ? 1.0 : pow(max(angle, 0.0), u_lightangle_spotCoef%s.y));\n", iStr, iStr, iStr); + p.F(" lightScale = clamp(1.0 / dot(u_lightatt%s, vec3(1.0, distance, distSq)), 0.0, 1.0) * pspPow(angle, u_lightangle_spotCoef%s.y);\n", iStr, iStr); p.C(" } else {\n"); p.C(" lightScale = 0.0;\n"); p.C(" }\n"); @@ -737,13 +755,9 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag p.F(" diffuse = (u_lightdiffuse%s * diffuseColor) * max(ldot, 0.0);\n", iStr); if (doSpecular) { p.C(" if (ldot >= 0.0) {\n"); - p.C(" if (u_matspecular.a > 0.0) {\n"); - p.C(" vec3 halfVec = toLight + vec3(0.0, 0.0, 1.0);\n"); - p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n"); - p.C(" ldot = pow(max(dot(halfVec, worldnormal) * halfInvLen, 0.0), u_matspecular.a);\n"); - p.C(" } else {\n"); - p.C(" ldot = 1.0;\n"); - p.C(" }\n"); + p.C(" vec3 halfVec = toLight + viewDir;\n"); + p.C(" float halfInvLen = inversesqrt(dot(halfVec, halfVec));\n"); + p.C(" ldot = pspPow(dot(halfVec, worldnormal) * halfInvLen, u_matspecular.a);\n"); p.C(" if (ldot > 0.0)\n"); p.F(" lightSum1 += u_lightspecular%s * specularColor * ldot %s;\n", iStr, timesLightScale); p.C(" }\n"); @@ -851,9 +865,31 @@ bool GenerateVertexShader(const VShaderID &id, char *buffer, const ShaderLanguag snprintf(ls0Str, sizeof(ls0Str), "%d", ls0); snprintf(ls1Str, sizeof(ls1Str), "%d", ls1); } - std::string lightFactor0 = StringFromFormat("(length(u_lightpos%s) == 0.0 ? worldnormal.z : dot(normalize(u_lightpos%s), worldnormal))", ls0Str, ls0Str); - std::string lightFactor1 = StringFromFormat("(length(u_lightpos%s) == 0.0 ? worldnormal.z : dot(normalize(u_lightpos%s), worldnormal))", ls1Str, ls1Str); - WRITE(p, " %sv_texcoord = vec3(u_uvscaleoffset.xy * vec2(1.0 + %s, 1.0 + %s) * 0.5, 1.0);\n", compat.vsOutPrefix, lightFactor0.c_str(), lightFactor1.c_str()); + // N.L with L the light vector as lighting sees it (zero stays zero), or the half vector + // if the light does specular. Whether lighting or the light is on doesn't matter. + auto shadeLight = [&](int ls, const char *lsStr, const char *name) { + if (shadeUberShader) { + p.F(" vec3 %s = u_lightpos%s;\n", name, lsStr); + p.F(" if (((u_lightControl >> 0x%02xu) & 0x3u) != 0x0u) %s = u_lightpos%s - worldpos;\n", 4 + 4 * ls + 2, name, lsStr); + p.F(" %s = normalizeOr000(%s);\n", name, name); + p.F(" if (((u_lightControl >> 0x%02xu) & 0x3u) == 0x1u) %s = normalizeOr000(%s + viewDir);\n", 4 + 4 * ls, name, name); + return; + } + GELightType type = static_cast(id.Bits(VS_BIT_LIGHT0_TYPE + 4 * ls, 2)); + GELightComputation comp = static_cast(id.Bits(VS_BIT_LIGHT0_COMP + 4 * ls, 2)); + if (type == GE_LIGHTTYPE_DIRECTIONAL) { + // Prenormalized. + p.F(" vec3 %s = u_lightpos%s;\n", name, lsStr); + } else { + p.F(" vec3 %s = normalizeOr000(u_lightpos%s - worldpos);\n", name, lsStr); + } + if (comp == GE_LIGHTCOMP_BOTH) { + p.F(" %s = normalizeOr000(%s + viewDir);\n", name, name); + } + }; + shadeLight(ls0, ls0Str, "shadeL0"); + shadeLight(ls1, ls1Str, "shadeL1"); + WRITE(p, " %sv_texcoord = vec3(u_uvscaleoffset.xy * vec2(1.0 + dot(shadeL0, worldnormal), 1.0 + dot(shadeL1, worldnormal)) * 0.5, 1.0);\n", compat.vsOutPrefix); } break; diff --git a/GPU/GLES/ShaderManagerGLES.cpp b/GPU/GLES/ShaderManagerGLES.cpp index d3c71d2b16..0ed21507aa 100644 --- a/GPU/GLES/ShaderManagerGLES.cpp +++ b/GPU/GLES/ShaderManagerGLES.cpp @@ -46,6 +46,7 @@ #include "GPU/GPUState.h" #include "GPU/ge_constants.h" #include "GPU/Common/ShaderUniforms.h" +#include "GPU/Common/TransformCommon.h" #include "GPU/GLES/ShaderManagerGLES.h" #include "GPU/GLES/DrawEngineGLES.h" @@ -549,7 +550,7 @@ void LinkedShader::UpdateUniforms(const ShaderID &vsid, const ShaderLanguageDesc SetColorUniform3(render_, &u_matemissive, gstate.materialemissive); } if (dirty & DIRTY_MATSPECULAR) { - SetColorUniform3ExtraFloat(render_, &u_matspecular, gstate.materialspecular, getFloat24(gstate.materialspecularcoef)); + SetColorUniform3ExtraFloat(render_, &u_matspecular, gstate.materialspecular, PSPSpecularCoef(getFloat24(gstate.materialspecularcoef))); } for (int i = 0; i < 4; i++) { @@ -855,7 +856,7 @@ enum class CacheDetectFlags { }; #define CACHE_HEADER_MAGIC 0x83277592 -#define CACHE_VERSION 43 +#define CACHE_VERSION 44 struct CacheHeader { uint32_t magic; diff --git a/GPU/Software/Lighting.cpp b/GPU/Software/Lighting.cpp index a0681da245..43d87b6367 100644 --- a/GPU/Software/Lighting.cpp +++ b/GPU/Software/Lighting.cpp @@ -21,6 +21,7 @@ #include "Common/CPUDetect.h" #include "Common/Math/SIMDHeaders.h" #include "GPU/GPUState.h" +#include "GPU/Common/TransformCommon.h" #include "GPU/Software/Lighting.h" #if PPSSPP_ARCH(SSE2) @@ -49,7 +50,7 @@ static inline float pspLightPow(float v, float e) { return 1.0f; } if (v > 0.0f) { - return pow(v, e); + return PSPLightPow(v, e); } // Negative stays negative, so let's just return the original. return v; @@ -179,7 +180,7 @@ void ComputeState(State *state, bool hasColor0) { } if (anyDiffuse || anySpecular) { - state->specularExp = gstate.getMaterialSpecularCoef(); + state->specularExp = PSPSpecularCoef(gstate.getMaterialSpecularCoef()); if (state->specularExp <= 0.0f) state->specularExp = 0.0f; else if (std::isnan(state->specularExp)) @@ -193,20 +194,11 @@ void ComputeState(State *state, bool hasColor0) { state->usesWorldNormal = gstate.getUVGenMode() == GE_TEXMAP_ENVIRONMENT_MAP || anyDiffuse || anySpecular; } -static inline float GenerateLightCoord(VertexData &vertex, const WorldCoords &worldnormal, int light) { - // TODO: Should specular lighting should affect this, too? Doesn't in GLES. - Vec3 L = GetLightVec(gstate.lpos, light); - // In other words, L.Length2() == 0.0f means Dot({0, 0, 1}, worldnormal). - float diffuse_factor = Dot(L.NormalizedOr001(cpu_info.bSSE4_1), worldnormal); - - return (diffuse_factor + 1.0f) / 2.0f; -} - -void GenerateLightST(VertexData &vertex, const WorldCoords &worldnormal) { +void GenerateLightST(VertexData &vertex, const WorldCoords &worldpos, const WorldCoords &worldnormal, const Vec3f &viewDir) { // Always calculate texture coords from lighting results if environment mapping is active // This should be done even if lighting is disabled altogether. - vertex.texturecoords.s() = GenerateLightCoord(vertex, worldnormal, gstate.getUVLS0()); - vertex.texturecoords.t() = GenerateLightCoord(vertex, worldnormal, gstate.getUVLS1()); + vertex.texturecoords.s() = PSPShadeMapCoord(gstate.getUVLS0(), worldpos, worldnormal, viewDir); + vertex.texturecoords.t() = PSPShadeMapCoord(gstate.getUVLS1(), worldpos, worldnormal, viewDir); } #if defined(_M_SSE) @@ -368,7 +360,7 @@ static void ProcessSIMD(VertexData &vertex, const WorldCoords &worldpos, const W } if (lstate.specular && diffuse_factor >= 0.0f) { - Vec3 H = L + Vec3(0.f, 0.f, 1.f); + Vec3 H = L + state.viewDir; float specular_factor = Dot33(H.NormalizedOr001(useSSE4), worldnormal); specular_factor = pspLightPow(specular_factor, state.specularExp); diff --git a/GPU/Software/Lighting.h b/GPU/Software/Lighting.h index c875aae08b..0c22f1e1cc 100644 --- a/GPU/Software/Lighting.h +++ b/GPU/Software/Lighting.h @@ -53,6 +53,7 @@ struct State { Vec4 baseAmbientColorFactor; float specularExp; + Vec3f viewDir; struct { bool colorForAmbient : 1; @@ -67,7 +68,7 @@ struct State { void ComputeState(State *state, bool hasColor0); -void GenerateLightST(VertexData &vertex, const WorldCoords &worldnormal); +void GenerateLightST(VertexData &vertex, const WorldCoords &worldpos, const WorldCoords &worldnormal, const Vec3f &viewDir); void Process(VertexData &vertex, const WorldCoords &worldpos, const WorldCoords &worldnormal, const State &state); } diff --git a/GPU/Software/Rasterizer.cpp b/GPU/Software/Rasterizer.cpp index b760364994..56abea7897 100644 --- a/GPU/Software/Rasterizer.cpp +++ b/GPU/Software/Rasterizer.cpp @@ -1159,7 +1159,7 @@ void DrawTriangleSlice( int32x4_t sec = vsetq_lane_s32(0, sec_color[i].ivec, 3); prim_color[i].ivec = vaddq_s32(prim_color[i].ivec, sec); #else - prim_color[i] = Vec4(sec_color[i], 0); + prim_color[i] += Vec4(sec_color[i], 0); #endif } } diff --git a/GPU/Software/SoftGpu.cpp b/GPU/Software/SoftGpu.cpp index 5cb910ae92..4b597d3409 100644 --- a/GPU/Software/SoftGpu.cpp +++ b/GPU/Software/SoftGpu.cpp @@ -111,7 +111,7 @@ const SoftwareCommandTableEntry softgpuCommandTable[] = { { GE_CMD_FOGENABLE, 0, SoftDirty::PIXEL_BASIC | SoftDirty::PIXEL_CACHED | SoftDirty::TRANSFORM_BASIC | SoftDirty::TRANSFORM_FOG | SoftDirty::TRANSFORM_MATRIX }, { GE_CMD_TEXMODE, 0, SoftDirty::SAMPLER_BASIC | SoftDirty::SAMPLER_TEXLIST | SoftDirty::RAST_TEX }, // Currently this doesn't affect any state, but maybe it should. - { GE_CMD_TEXSHADELS }, + { GE_CMD_TEXSHADELS, 0, SoftDirty::TRANSFORM_BASIC }, { GE_CMD_SHADEMODE, 0, SoftDirty::RAST_BASIC }, { GE_CMD_TEXFUNC, 0, SoftDirty::SAMPLER_BASIC }, { GE_CMD_COLORTEST, 0, SoftDirty::PIXEL_BASIC | SoftDirty::PIXEL_CACHED }, diff --git a/GPU/Software/TransformUnit.cpp b/GPU/Software/TransformUnit.cpp index 13e53c7841..a16ca43308 100644 --- a/GPU/Software/TransformUnit.cpp +++ b/GPU/Software/TransformUnit.cpp @@ -28,6 +28,7 @@ #include "GPU/Common/DrawEngineCommon.h" #include "GPU/Common/VertexDecoderCommon.h" #include "GPU/Common/SoftwareTransformCommon.h" +#include "GPU/Common/TransformCommon.h" #include "GPU/Common/VertexReader.h" #include "GPU/GPUStateSIMDUtil.h" #include "Common/Math/SIMDHeaders.h" @@ -259,6 +260,13 @@ void ComputeTransformState(TransformState *state, const VertexReader &vreader) { } else { state->lightingState.usesWorldNormal = state->uvGenMode == GE_TEXMAP_ENVIRONMENT_MAP; } + if (state->uvGenMode == GE_TEXMAP_ENVIRONMENT_MAP) { + // Shade mapping uses the light vector as lighting sees it, which depends on position for other lights. + if (!gstate.isDirectionalLight(gstate.getUVLS0()) || !gstate.isDirectionalLight(gstate.getUVLS1())) { + canSkipWorldPos = false; + } + } + state->lightingState.viewDir = PSPViewDirection(gstate.viewMatrix); float world[16]; float view[16]; @@ -446,7 +454,7 @@ ClipVertexData TransformUnit::ReadVertex(const VertexReader &vreader, const Tran Vec3 stq = Vec3ByMatrix43(source, gstate.tgenMatrix); vertex.v.texturecoords = Vec3Packedf(stq.x, stq.y, stq.z); } else if (state.uvGenMode == GE_TEXMAP_ENVIRONMENT_MAP) { - Lighting::GenerateLightST(vertex.v, worldnormal); + Lighting::GenerateLightST(vertex.v, worldpos, worldnormal, state.lightingState.viewDir); } PROFILE_THIS_SCOPE("light"); diff --git a/GPU/Vulkan/ShaderManagerVulkan.cpp b/GPU/Vulkan/ShaderManagerVulkan.cpp index 31b040c816..926b4e0067 100644 --- a/GPU/Vulkan/ShaderManagerVulkan.cpp +++ b/GPU/Vulkan/ShaderManagerVulkan.cpp @@ -369,7 +369,7 @@ enum class VulkanCacheDetectFlags { }; #define CACHE_HEADER_MAGIC 0xff51f420 -#define CACHE_VERSION 60 +#define CACHE_VERSION 61 struct VulkanCacheHeader { uint32_t magic; diff --git a/frametests b/frametests index 401ec1c03f..49a8baadc9 160000 --- a/frametests +++ b/frametests @@ -1 +1 @@ -Subproject commit 401ec1c03f717d4dca72b57ec6c979f15db757fb +Subproject commit 49a8baadc9f709e21ac6e76f0668521536e25526 diff --git a/pspautotests b/pspautotests index 7302fa573c..a3f220d0fd 160000 --- a/pspautotests +++ b/pspautotests @@ -1 +1 @@ -Subproject commit 7302fa573c3688b88325e5965d0aaef0b030ac59 +Subproject commit a3f220d0fdc57daa186140b550cc6de0c4945ab2 diff --git a/test.py b/test.py index a8bd5e5e10..ef639a0ebd 100755 --- a/test.py +++ b/test.py @@ -230,6 +230,8 @@ tests_good = [ "gpu/ge/intrsuspend", "gpu/ge/queue", "gpu/ge/queue2", + "gpu/lighting/shademap", + "gpu/lighting/specular", "gpu/primitives/indices", "gpu/primitives/invalidprim", "gpu/primitives/points", @@ -510,17 +512,9 @@ known_failures = { # The ISA returns the canonical NaN (0x7fc00000) from every operation, never the operand's # NaN, so a negative or signaling NaN input loses its sign and payload. Everything else passes. "cpu/fpu/roundmode", - # The software renderer's output differs from the reference by the same amount on both of - # these architectures, despite them using completely different SIMD paths. Unexplained. - "gpu/clipping/homogeneous", - "gpu/commands/cull", - "gpu/primitives/triangles", ], "loongarch64": [ "cpu/fpu/fpu", - "gpu/clipping/homogeneous", - "gpu/commands/cull", - "gpu/primitives/triangles", ], }