mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Merge pull request #21781 from hrydgard/more-crosssimd
Fix depth clear issue from previous commit, use fastcull on indexed draws
This commit is contained in:
5 files changed
+179
-23
No files matched your search
+101
-2
@@ -250,6 +250,25 @@ struct Vec4F32 {
|
||||
return Vec4F32{_mm_castsi128_ps(_mm_slli_epi32(_mm_loadu_si128((const __m128i *)src), 8))};
|
||||
}
|
||||
|
||||
float Dot3(Vec4F32 b) {
|
||||
// Zero out the W component before multiplying to ensure only X, Y, Z are summed
|
||||
alignas(16) static const uint32_t mask[4] = { 0xFFFFFFFF, 0xFFFFFFFF, 0xFFFFFFFF, 0x0 };
|
||||
__m128 masked = _mm_and_ps(v, _mm_load_ps((const float *)mask));
|
||||
__m128 mul = _mm_mul_ps(masked, b.v);
|
||||
__m128 shuf1 = _mm_shuffle_ps(mul, mul, _MM_SHUFFLE(2, 1, 0, 3));
|
||||
__m128 sum1 = _mm_add_ps(mul, shuf1);
|
||||
__m128 shuf2 = _mm_shuffle_ps(sum1, sum1, _MM_SHUFFLE(1, 0, 3, 2));
|
||||
return _mm_cvtss_f32(_mm_add_ps(sum1, shuf2));
|
||||
}
|
||||
|
||||
float Dot4(Vec4F32 b) {
|
||||
__m128 mul = _mm_mul_ps(v, b.v);
|
||||
__m128 shuf1 = _mm_shuffle_ps(mul, mul, _MM_SHUFFLE(2, 1, 0, 3));
|
||||
__m128 sum1 = _mm_add_ps(mul, shuf1);
|
||||
__m128 shuf2 = _mm_shuffle_ps(sum1, sum1, _MM_SHUFFLE(1, 0, 3, 2));
|
||||
return _mm_cvtss_f32(_mm_add_ps(sum1, shuf2));
|
||||
}
|
||||
|
||||
void Store(float *dst) { _mm_storeu_ps(dst, v); }
|
||||
void Store2(float *dst) { _mm_storel_epi64((__m128i *)dst, _mm_castps_si128(v)); }
|
||||
void StoreAligned(float *dst) { _mm_store_ps(dst, v); }
|
||||
@@ -791,6 +810,34 @@ struct Vec4F32 {
|
||||
return Vec4F32{ sum };
|
||||
}
|
||||
|
||||
float Dot3(Vec4F32 b) {
|
||||
// Zero out the W component before multiplying to ensure only X, Y, Z are summed
|
||||
float32x4_t masked = vsetq_lane_f32(0.0f, v, 3);
|
||||
float32x4_t mul = vmulq_f32(masked, b.v);
|
||||
#if PPSSPP_ARCH(ARM64_NEON)
|
||||
return vaddvq_f32(mul);
|
||||
#else
|
||||
float32x2_t sum_low = vget_low_f32(mul);
|
||||
float32x2_t sum_high = vget_high_f32(mul);
|
||||
float32x2_t sum = vadd_f32(sum_low, sum_high);
|
||||
sum = vpadd_f32(sum, sum);
|
||||
return vget_lane_f32(sum, 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
float Dot4(Vec4F32 b) {
|
||||
float32x4_t mul = vmulq_f32(v, b.v);
|
||||
#if PPSSPP_ARCH(ARM64_NEON)
|
||||
return vaddvq_f32(mul);
|
||||
#else
|
||||
float32x2_t sum_low = vget_low_f32(mul);
|
||||
float32x2_t sum_high = vget_high_f32(mul);
|
||||
float32x2_t sum = vadd_f32(sum_low, sum_high);
|
||||
sum = vpadd_f32(sum, sum);
|
||||
return vget_lane_f32(sum, 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<int i> float GetLane() const {
|
||||
return vgetq_lane_f32(v, i);
|
||||
}
|
||||
@@ -1122,7 +1169,19 @@ struct Vec4F32 {
|
||||
}
|
||||
|
||||
// NOTE: May be slow.
|
||||
float operator[](size_t index) const { return ((float *)&v)[index]; }
|
||||
float operator[](size_t index) const {
|
||||
// index is a compile-time constant parameter to the intrinsic.
|
||||
int ival;
|
||||
switch (index) {
|
||||
case 0: ival = __lsx_vpickve2gr_w((__m128i)v, 0);
|
||||
case 1: ival = __lsx_vpickve2gr_w((__m128i)v, 1);
|
||||
case 2: ival = __lsx_vpickve2gr_w((__m128i)v, 2);
|
||||
default: ival = __lsx_vpickve2gr_w((__m128i)v, 3);
|
||||
}
|
||||
float fval;
|
||||
memcpy(&fval, &ival, sizeof(float));
|
||||
return fval;
|
||||
}
|
||||
|
||||
Vec4F32 operator +(Vec4F32 other) const { return Vec4F32{ (__m128)__lsx_vfadd_s(v, other.v) }; }
|
||||
Vec4F32 operator -(Vec4F32 other) const { return Vec4F32{ (__m128)__lsx_vfsub_s(v, other.v) }; }
|
||||
@@ -1255,8 +1314,39 @@ struct Vec4F32 {
|
||||
return Vec4F32{ sum };
|
||||
}
|
||||
|
||||
float Dot3(Vec4F32 b) {
|
||||
// Zero out the W component before multiplying to ensure only X, Y, Z are summed
|
||||
__m128 masked = (__m128)__lsx_vinsgr2vr_w((__m128i)v, 0, 3);
|
||||
__m128 mul = (__m128)__lsx_vfmul_s(masked, b.v);
|
||||
// Sum all elements: horizontal add
|
||||
__m128 shuf1 = (__m128)__lsx_vshuf4i_w((__m128i)mul, 0b10110001); // Rotate elements
|
||||
__m128 sum1 = (__m128)__lsx_vfadd_s(mul, shuf1);
|
||||
__m128 shuf2 = (__m128)__lsx_vshuf4i_w((__m128i)sum1, 0b01001110); // Swap pairs
|
||||
__m128 sum2 = (__m128)__lsx_vfadd_s(sum1, shuf2);
|
||||
float fval;
|
||||
int ival = __lsx_vpickve2gr_w((__m128i)sum2, 0);
|
||||
memcpy(&fval, &ival, sizeof(float));
|
||||
return fval;
|
||||
}
|
||||
|
||||
float Dot4(Vec4F32 b) {
|
||||
__m128 mul = (__m128)__lsx_vfmul_s(v, b.v);
|
||||
// Sum all elements: horizontal add
|
||||
__m128 shuf1 = (__m128)__lsx_vshuf4i_w((__m128i)mul, 0b10110001); // Rotate elements
|
||||
__m128 sum1 = (__m128)__lsx_vfadd_s(mul, shuf1);
|
||||
__m128 shuf2 = (__m128)__lsx_vshuf4i_w((__m128i)sum1, 0b01001110); // Swap pairs
|
||||
__m128 sum2 = (__m128)__lsx_vfadd_s(sum1, shuf2);
|
||||
float fval;
|
||||
int ival = __lsx_vpickve2gr_w((__m128i)sum2, 0);
|
||||
memcpy(&fval, &ival, sizeof(float));
|
||||
return fval;
|
||||
}
|
||||
|
||||
template<int i> float GetLane() const {
|
||||
return __lsx_vpickve2gr_w((__m128i)v, i);
|
||||
int ival = __lsx_vpickve2gr_w((__m128i)v, i);
|
||||
float fval;
|
||||
memcpy(&fval, &ival, sizeof(float));
|
||||
return fval;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1758,6 +1848,15 @@ struct Vec4F32 {
|
||||
return Vec4F32{ { x, y, z, 1.0f } };
|
||||
}
|
||||
|
||||
float Dot3(Vec4F32 b) {
|
||||
// Only sum the first three elements (X, Y, Z), ignore W
|
||||
return v[0] * b.v[0] + v[1] * b.v[1] + v[2] * b.v[2];
|
||||
}
|
||||
|
||||
float Dot4(Vec4F32 b) {
|
||||
return v[0] * b.v[0] + v[1] * b.v[1] + v[2] * b.v[2] + v[3] * b.v[3];
|
||||
}
|
||||
|
||||
template<int i> float GetLane() const {
|
||||
return v[i];
|
||||
}
|
||||
|
||||
@@ -349,8 +349,8 @@ bool DrawEngineCommon::TestBoundingBox(const void *vdata, const void *inds, int
|
||||
// off to hardware, with whatever capabilities are available.
|
||||
//
|
||||
// NOTE: This doesn't handle through-mode or indexing (morph or skinning can be handled if they're implemented in software during decode).
|
||||
template<u32 posFmt>
|
||||
static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, ClipInfoFlags *clipInfoFlags) {
|
||||
template<u32 posFmt, u32 idxFmt>
|
||||
static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, const void *idata, int vertexCount, const VertexDecoder *dec, ClipInfoFlags *clipInfoFlags) {
|
||||
Mat4F32 cullMat(cullMatrix);
|
||||
alignas(16) static const float planesXYData[4] = { 1, -1, 1, -1 };
|
||||
Vec4F32 planesXY = Vec4F32::LoadAligned(planesXYData);
|
||||
@@ -362,14 +362,36 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int
|
||||
// In reality we should probably affect X and Y too, but meh.
|
||||
alignas(16) static const u32 vertexMaskData[4] = {0xFFFFFFFF, 0xFFFFFFFF, 0xFFFFFF00, 0xFFFFFFFF};
|
||||
const int stride = dec->VertexSize();
|
||||
const s8 *data = (const s8 *)vdata + dec->posoff;
|
||||
const s8 *srcdata = (const s8 *)vdata + dec->posoff;
|
||||
const s8 *data = srcdata;
|
||||
|
||||
const float vpZScale = gstate.getViewportZScale();
|
||||
|
||||
float minProjZ = FLT_MAX;
|
||||
float maxProjZ = -FLT_MAX;
|
||||
|
||||
for (int i = 0; i < vertexCount; i++, data += stride) {
|
||||
for (int i = 0; i < vertexCount; i++) {
|
||||
switch (idxFmt) {
|
||||
case GE_VTYPE_IDX_8BIT:
|
||||
{
|
||||
u8 idx = ((u8 *)idata)[i];
|
||||
data = (const s8 *)srcdata + idx * stride;
|
||||
break;
|
||||
}
|
||||
case GE_VTYPE_IDX_16BIT:
|
||||
{
|
||||
u16 idx = ((u16 *)idata)[i];
|
||||
data = (const s8 *)srcdata + idx * stride;
|
||||
break;
|
||||
}
|
||||
case GE_VTYPE_IDX_32BIT:
|
||||
{
|
||||
u32 idx = ((u32 *)idata)[i];
|
||||
data = (const s8 *)srcdata + idx * stride;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
Vec4F32 objPos;
|
||||
switch (posFmt) {
|
||||
case GE_VTYPE_POS_8BIT:
|
||||
@@ -398,6 +420,10 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int
|
||||
if (projZ > maxProjZ) { // else ruins the minss/maxss optimization.
|
||||
maxProjZ = projZ;
|
||||
}
|
||||
|
||||
if (idxFmt == GE_VTYPE_IDX_NONE) {
|
||||
data += stride;
|
||||
}
|
||||
}
|
||||
|
||||
if (!AllCompareBitsSet(insideMaskXY) || !AllCompareBitsSet(insideMaskZ)) {
|
||||
@@ -447,7 +473,7 @@ static bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int
|
||||
return true;
|
||||
}
|
||||
|
||||
bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *flags) {
|
||||
bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void *vdata, const void *idata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *flags) {
|
||||
// Although this may lead to drawing that shouldn't happen, the viewport is more complex on VR.
|
||||
// Let's always say objects are within bounds.
|
||||
if (gstate_c.Use(GPU_USE_VIRTUAL_REALITY)) {
|
||||
@@ -456,18 +482,50 @@ bool DrawEngineCommon::TestBoundingBoxFast(const float *cullMatrix, const void *
|
||||
return false;
|
||||
}
|
||||
|
||||
switch (vertType & GE_VTYPE_POS_MASK) {
|
||||
case GE_VTYPE_POS_8BIT:
|
||||
return ::TestBoundingBoxFast<GE_VTYPE_POS_8BIT>(cullMatrix, vdata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_16BIT:
|
||||
return ::TestBoundingBoxFast<GE_VTYPE_POS_16BIT>(cullMatrix, vdata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_FLOAT:
|
||||
return ::TestBoundingBoxFast<GE_VTYPE_POS_FLOAT>(cullMatrix, vdata, vertexCount, dec, flags);
|
||||
// Dispatching like this is a bit ugly, but we want to avoid every possible overhead *inside* TestBoundingBoxFast.
|
||||
// That said, I'm not 100% sure it's worth it..
|
||||
switch (vertType & GE_VTYPE_IDX_MASK) {
|
||||
case GE_VTYPE_IDX_NONE:
|
||||
switch (vertType & GE_VTYPE_POS_MASK) {
|
||||
case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_8BIT, GE_VTYPE_IDX_NONE>(cullMatrix, vdata, nullptr, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_16BIT, GE_VTYPE_IDX_NONE>(cullMatrix, vdata, nullptr, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast<GE_VTYPE_POS_FLOAT, GE_VTYPE_IDX_NONE>(cullMatrix, vdata, nullptr, vertexCount, dec, flags);
|
||||
default:
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case GE_VTYPE_IDX_8BIT:
|
||||
switch (vertType & GE_VTYPE_POS_MASK) {
|
||||
case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_8BIT, GE_VTYPE_IDX_8BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_16BIT, GE_VTYPE_IDX_8BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast<GE_VTYPE_POS_FLOAT, GE_VTYPE_IDX_8BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
default:
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case GE_VTYPE_IDX_16BIT:
|
||||
switch (vertType & GE_VTYPE_POS_MASK) {
|
||||
case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_8BIT, GE_VTYPE_IDX_16BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_16BIT, GE_VTYPE_IDX_16BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast<GE_VTYPE_POS_FLOAT, GE_VTYPE_IDX_16BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
default:
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case GE_VTYPE_IDX_32BIT:
|
||||
switch (vertType & GE_VTYPE_POS_MASK) {
|
||||
case GE_VTYPE_POS_8BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_8BIT, GE_VTYPE_IDX_32BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_16BIT: return ::TestBoundingBoxFast<GE_VTYPE_POS_16BIT, GE_VTYPE_IDX_32BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
case GE_VTYPE_POS_FLOAT: return ::TestBoundingBoxFast<GE_VTYPE_POS_FLOAT, GE_VTYPE_IDX_32BIT>(cullMatrix, vdata, idata, vertexCount, dec, flags);
|
||||
default:
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
// Shouldn't end up here with the checks outside this function.
|
||||
_dbg_assert_(false);
|
||||
return true;
|
||||
break;
|
||||
}
|
||||
_dbg_assert_(false);
|
||||
return true;
|
||||
}
|
||||
|
||||
// 2D bounding box test against scissor. No indexing yet.
|
||||
|
||||
@@ -116,7 +116,7 @@ public:
|
||||
|
||||
// This is a less accurate version of TestBoundingBox, but faster. Can have more false positives.
|
||||
// Doesn't support indexing.
|
||||
bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *clipInfoFlags);
|
||||
bool TestBoundingBoxFast(const float *cullMatrix, const void *vdata, const void *idata, int vertexCount, const VertexDecoder *dec, u32 vertType, ClipInfoFlags *clipInfoFlags);
|
||||
bool TestBoundingBoxThrough(const void *vdata, int vertexCount, const VertexDecoder *dec, u32 vertType, int *bytesRead);
|
||||
bool EstimateThroughPrimSafeSize(const void *verts, const void *inds, GEPrimitiveType prim, int vertexCount, const VertexDecoder *dec, u32 vertType, int *safeWidth, int *safeHeight);
|
||||
|
||||
|
||||
@@ -212,7 +212,7 @@ SoftwareTransformAction RunSoftwareTransform(SoftwareTransformParams ¶ms, in
|
||||
if (matchingComponents && stencilNotMasked) {
|
||||
DepthScaleFactors depthScale = GetDepthScaleFactors(gstate_c.UseFlags());
|
||||
// Need to rescale from a [0, 1] float. This is the final transformed value.
|
||||
float depth = depthScale.EncodeFromU16((float)(int)(transformed[1].z * 65535.0f));
|
||||
float depth = depthScale.EncodeFromU16(transformed[1].z);
|
||||
// Non-zero depth clears are unusual, but some drivers don't match drawn depth values to cleared values.
|
||||
// Games sometimes expect exact matches (see #12626, for example) for equal comparisons.
|
||||
if (!(params.everUsedEqualDepth && gstate.isClearModeDepthMask() && result->depth > 0.0f && result->depth < 1.0f)) {
|
||||
|
||||
+3
-4
@@ -1016,7 +1016,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
// For now, turn off culling on platforms where we don't have SIMD bounding box tests, like RISC-V.
|
||||
#if PPSSPP_ARCH(ARM_NEON) || PPSSPP_ARCH(SSE2)
|
||||
|
||||
#define PASSES_CULLING ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_MORPHCOUNT_MASK | GE_VTYPE_WEIGHT_MASK | GE_VTYPE_IDX_MASK)) || count > MAX_CULL_CHECK_COUNT)
|
||||
#define PASSES_CULLING ((vertexType & (GE_VTYPE_THROUGH_MASK | GE_VTYPE_MORPHCOUNT_MASK | GE_VTYPE_WEIGHT_MASK)) || count > MAX_CULL_CHECK_COUNT)
|
||||
|
||||
#else
|
||||
|
||||
@@ -1030,7 +1030,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
bool passCulling = PASSES_CULLING;
|
||||
if (!passCulling) {
|
||||
// Do software culling.
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, count, decoder, vertexType, &flags)) {
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, inds, count, decoder, vertexType, &flags)) {
|
||||
passCulling = true;
|
||||
} else {
|
||||
gpuStats.perFrame.numCulledDraws++;
|
||||
@@ -1120,8 +1120,7 @@ void GPUCommonHW::Execute_Prim(u32 op, u32 diff) {
|
||||
ClipInfoFlags flags{};
|
||||
if (!passCulling) {
|
||||
// Do software culling.
|
||||
_dbg_assert_((vertexType & GE_VTYPE_IDX_MASK) == GE_VTYPE_IDX_NONE);
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, count, decoder, vertexType, &flags)) {
|
||||
if (drawEngineCommon_->TestBoundingBoxFast(gstate_c.cullMatrix, verts, inds, count, decoder, vertexType, &flags)) {
|
||||
passCulling = true;
|
||||
} else {
|
||||
gpuStats.perFrame.numCulledDraws++;
|
||||
|
||||
Reference in new issue
Block a user