mirror of
https://github.com/hrydgard/ppsspp.git
synced 2026-10-01 14:58:14 +00:00
Fallback to software transform if draw intersects -W<Z<W
This commit is contained in:
1 parent
5eb114d3d9
commit
86f9618cb9
5 files changed
+35
-12
No files matched your search
@@ -349,8 +349,8 @@ static bool TestBoundingBoxFast(const float *worldViewProj, const void *vdata, i
|
||||
const int offset = dec->posoff;
|
||||
const s8 *data = (const s8 *)vdata + offset;
|
||||
|
||||
float vpZOffset = gstate.getViewportZCenter();
|
||||
float vpZScale = gstate.getViewportZScale();
|
||||
const float vpZOffset = gstate.getViewportZCenter();
|
||||
const float vpZScale = gstate.getViewportZScale();
|
||||
|
||||
for (int i = 0; i < vertexCount; i++, data += stride) {
|
||||
Vec4F32 objPos;
|
||||
@@ -366,14 +366,6 @@ static bool TestBoundingBoxFast(const float *worldViewProj, const void *vdata, i
|
||||
break;
|
||||
}
|
||||
Vec4F32 clipPos = objPos.AsVec3ByMatrix44(worldViewProjMat) & vertexMask;
|
||||
const float projZ = vpZScale * clipPos[2] / clipPos[3];
|
||||
if (projZ < minProjZ) {
|
||||
minProjZ = projZ;
|
||||
}
|
||||
if (projZ > maxProjZ) { // else ruins the minss/maxss optimization.
|
||||
maxProjZ = projZ;
|
||||
}
|
||||
|
||||
Vec4F32 posW = clipPos.ShuffleWWWW();
|
||||
if (cull) {
|
||||
Vec4F32 posXY = clipPos.ShuffleXXYY();
|
||||
@@ -386,9 +378,17 @@ static bool TestBoundingBoxFast(const float *worldViewProj, const void *vdata, i
|
||||
if (cull) {
|
||||
insideMaskZ |= planeDistZ.CompareGe(Vec4F32::Zero());
|
||||
}
|
||||
const float projZ = vpZScale * clipPos[2] / clipPos[3];
|
||||
if (projZ < minProjZ) {
|
||||
minProjZ = projZ;
|
||||
}
|
||||
if (projZ > maxProjZ) { // else ruins the minss/maxss optimization.
|
||||
maxProjZ = projZ;
|
||||
}
|
||||
}
|
||||
|
||||
if (!cull || (AllCompareBitsSet(insideMaskXY) && AllCompareBitsSet(insideMaskZ))) {
|
||||
depths->valid = true;
|
||||
depths->hitClipSpaceZW = AnyCompareBitsSet(anyOutsideMaskZ);
|
||||
// Before checking later, we apply a viewport to the projZ, so we can later compare against minZ/maxZ or the outer bounds.
|
||||
// But to save operations we don't do it here.
|
||||
@@ -449,7 +449,7 @@ bool DrawEngineCommon::TestBoundingBoxFast(const float *worldViewProj, const voi
|
||||
applyViewport.wx = -(maxViewport.x + minViewport.x) * viewportInvSize.x;
|
||||
applyViewport.wy = -(maxViewport.y + minViewport.y) * viewportInvSize.y;
|
||||
|
||||
// TODO: Optimize.
|
||||
// TODO: Optimize. It's possible to scale/offset a matrix in a quicker way.
|
||||
Matrix4ByMatrix4(mtx, worldViewProj, applyViewport.m);
|
||||
worldViewProj = mtx;
|
||||
}
|
||||
|
||||
@@ -73,10 +73,15 @@ struct alignas(16) Plane8 {
|
||||
};
|
||||
|
||||
struct BoundingDepths {
|
||||
bool valid = false;
|
||||
bool hitClipSpaceZW = false;
|
||||
float minProjZ = FLT_MAX;
|
||||
float maxProjZ = -FLT_MAX;
|
||||
void Merge(const BoundingDepths &other) {
|
||||
if (!valid) {
|
||||
*this = other;
|
||||
return;
|
||||
}
|
||||
hitClipSpaceZW |= other.hitClipSpaceZW;
|
||||
if (other.minProjZ < minProjZ) {
|
||||
minProjZ = other.minProjZ;
|
||||
@@ -380,4 +385,6 @@ protected:
|
||||
std::vector<DepthDraw> depthDraws_;
|
||||
|
||||
double rasterTimeStart_ = 0.0;
|
||||
|
||||
bool lastUseHwTransform_ = true;
|
||||
};
|
||||
@@ -98,6 +98,7 @@ struct GPUStatsPerFrame {
|
||||
int numDrawSyncs;
|
||||
int numListSyncs;
|
||||
int numFlushes;
|
||||
int numSoftTransformedDraws;
|
||||
int numSoftClippedTriangles;
|
||||
int numBBOXJumps;
|
||||
int numVertsSubmitted;
|
||||
|
||||
+2
-1
@@ -1805,7 +1805,7 @@ void GPUCommonHW::FormatGPUStatsCommon(StringWriter &w) {
|
||||
w.F(
|
||||
"DL processing time: %0.2f ms, %d drawsync, %d listsync\n"
|
||||
"Draw: %d (%d dec, %d culled), flushes %d, clears %d, bbox jumps %d\n"
|
||||
"Vertices: %d dec: %d drawn: %d clipped tris: %d\n"
|
||||
"%d soft. Vertices: %d dec: %d drawn: %d clipped tris: %d\n"
|
||||
"FBOs active: %d (evaluations: %d, created %d)\n"
|
||||
"Textures: %d, dec: %d, invalidated: %d, hashed: %d kB, clut %d\n"
|
||||
"readbacks %d (%d non-block), upload %d (cached %d), depal %d\n"
|
||||
@@ -1824,6 +1824,7 @@ void GPUCommonHW::FormatGPUStatsCommon(StringWriter &w) {
|
||||
gpuStats.perFrame.numFlushes,
|
||||
gpuStats.perFrame.numClears,
|
||||
gpuStats.perFrame.numBBOXJumps,
|
||||
gpuStats.perFrame.numSoftTransformedDraws,
|
||||
gpuStats.perFrame.numVertsSubmitted,
|
||||
gpuStats.perFrame.numVertsDecoded,
|
||||
gpuStats.perFrame.numUncachedVertsDrawn,
|
||||
|
||||
@@ -242,6 +242,18 @@ void DrawEngineVulkan::Flush() {
|
||||
}
|
||||
bool useHWTransform = CanUseHardwareTransform(prim) && provokingVertexOk;
|
||||
|
||||
if (useHWTransform && boundingDepths_.valid) {
|
||||
if (boundingDepths_.hitClipSpaceZW) {
|
||||
// Revert to software so we can clip more accurately.
|
||||
useHWTransform = false;
|
||||
}
|
||||
}
|
||||
|
||||
if (useHWTransform != lastUseHwTransform_) {
|
||||
gstate_c.Dirty(DIRTY_VERTEXSHADER_STATE);
|
||||
lastUseHwTransform_ = useHWTransform;
|
||||
}
|
||||
|
||||
// TODO: Here we can check depths_ to see if we need to fall back to software transform for clipping.
|
||||
|
||||
// The optimization to avoid indexing isn't really worth it on Vulkan since it means creating more pipelines.
|
||||
@@ -377,6 +389,8 @@ void DrawEngineVulkan::Flush() {
|
||||
DepthRasterSubmitRaw(prim, dec_, dec_->VertexType(), vertexCount);
|
||||
}
|
||||
} else {
|
||||
gpuStats.perFrame.numSoftTransformedDraws++;
|
||||
|
||||
PROFILE_THIS_SCOPE("soft");
|
||||
const VertexDecoder *swDec = dec_;
|
||||
if (swDec->nweights != 0) {
|
||||
|
||||
Reference in new issue
Block a user