From f26ca37dad7bb7f3d34b7cc75052dc0c4826bef9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Tue, 29 Sep 2026 13:21:07 -0600 Subject: [PATCH] Threads: Context switches cost what they do on hardware Timed on a PSP, each way of handing the CPU to another thread (the call and the switch together): hardware before now rotate to an equal thread 7 14 7 signal, better thread runs 10 17 10 it waits again, back to caller 10 19 12 wakeup, better thread runs 8 13 6 it sleeps again, back to caller 7 12 6 start a better thread, entry 30 28 30 thread ends, back to its waiter 21 13 20 notify, better thread's callback 14 13 14 A switch between two threads now costs 1150 cycles instead of 2700. Starting a better thread costs 2000 cycles more, ending a thread 3300, and setting up a callback 1800. Also splits a wait timeout's ~30us into the deadline being taken 12us into the call and the timeout going off 18us after it. That only changes the time left written back, which threads/semaphores/wait and threads/fpl/cancel pin between them. intr/vblank is re-recorded so it no longer depends on the phase of the frame. threads/callbacks/combos now passes. Co-Authored-By: Claude Opus 5.5 (1M context) --- Core/HLE/sceKernelThread.cpp | 27 ++++++++++++++++++++++----- Core/HLE/sceKernelThread.h | 5 ++++- test.py | 2 +- 3 files changed, 27 insertions(+), 7 deletions(-) diff --git a/Core/HLE/sceKernelThread.cpp b/Core/HLE/sceKernelThread.cpp index e68ba98959..921daca042 100644 --- a/Core/HLE/sceKernelThread.cpp +++ b/Core/HLE/sceKernelThread.cpp @@ -1539,8 +1539,9 @@ static bool __KernelChance(int percent) { // wait fails with SCE_KERNEL_ERROR_WAIT_TIMEOUT at once, without giving up the CPU or writing the // timeout back. For most calls that happens about 85% of the time for 0us, half the time for 1us, // 15% for 2us and never from 3us on. sceKernelAllocateVpl does more first: always up to 1us, then -// about 17% less per us. Longer ones end max(timeout, 205us) plus about 35us after the call, the -// last few of which we already spend around the event. +// about 17% less per us. Longer ones end max(timeout, 205us) plus about 35us after the call. The +// deadline is taken late in the call, so the time left written back when something else ends the +// wait counts from there. bool __KernelWaitTimesOutAtOnce(u32 timeoutPtr, int basePercent, int stepPercent) { if (!Memory::IsValid4AlignedAddress(timeoutPtr)) { return false; @@ -1550,7 +1551,7 @@ bool __KernelWaitTimesOutAtOnce(u32 timeoutPtr, int basePercent, int stepPercent } s64 __KernelWaitTimeoutUs(u32 micro) { - return (s64)std::max(micro, 205U) + WAIT_TIMEOUT_LATENCY_US; + return (s64)std::max(micro, 205U) + WAIT_TIMEOUT_DEADLINE_US + WAIT_TIMEOUT_LATENCY_US; } void hleThreadEndTimeout(u64 userdata, int cyclesLate) @@ -2199,6 +2200,9 @@ int __KernelStartThread(SceUID threadToStartID, int argSize, u32 argBlockPtr, bo KernelValidateThreadTarget(startThread->context.pc); __KernelChangeReadyState(cur, currentThread, true); if (__InterruptsEnabled()) { + // Handing over costs more: about 30us from the call to the new thread's entry on + // hardware, where starting a worse one takes 10-25us. + hleEatCycles(2000); g_startThreadHandoff = threadToStartID; hleReSchedule("thread started"); } @@ -2278,6 +2282,9 @@ int sceKernelGetThreadStackFreeSize(SceUID threadID) { return hleLogDebug(Log::sceKernel, sz & ~3); } +// What a thread ending costs, besides the switch away from it. +static const int THREAD_EXIT_CYCLES = 3300; + void __KernelReturnFromThread() { hleSkipDeadbeef(); @@ -2287,6 +2294,8 @@ void __KernelReturnFromThread() _dbg_assert_msg_(thread != NULL, "Returned from a NULL thread."); DEBUG_LOG(Log::sceKernel, "__KernelReturnFromThread: %d", exitStatus); + // About 20us from a thread ending to a thread waiting on it running, on hardware. + hleEatCycles(THREAD_EXIT_CYCLES); __KernelStopThread(currentThread, exitStatus, "thread returned"); hleReSchedule("thread returned"); @@ -2308,6 +2317,7 @@ int sceKernelExitThread(int exitStatus) { if (exitStatus < 0) { exitStatus = SCE_KERNEL_ERROR_ILLEGAL_ARGUMENT; } + hleEatCycles(THREAD_EXIT_CYCLES); __KernelStopThread(currentThread, exitStatus, "thread exited"); hleReSchedule("thread exited"); @@ -2343,6 +2353,7 @@ int sceKernelExitDeleteThread(int exitStatus) { INFO_LOG(Log::sceKernel,"sceKernelExitDeleteThread(%d)", exitStatus); uint32_t thread_attr = thread->nt.attr; uint32_t uid = thread->GetUID(); + hleEatCycles(THREAD_EXIT_CYCLES); __KernelDeleteThread(currentThread, exitStatus, "thread exited with delete"); hleReSchedule("thread exited with delete"); @@ -3265,13 +3276,15 @@ void __KernelSwitchContext(PSPThread *target, const char *reason) { } #endif - // Switching threads eats some cycles. This is a low approximation. + // Switching threads eats some cycles. Between two threads it's about 5us: rotating the ready + // queue to an equal thread, or waking a better one and having it block again, each take + // 7-10us on hardware including the calls themselves (pspautotests threads/scheduling/handoff). if (fromIdle && toIdle) { // Don't eat any cycles going between idle. } else if (fromIdle || toIdle) { currentMIPS->downcount -= 1200; } else { - currentMIPS->downcount -= 2700; + currentMIPS->downcount -= 1150; } __KernelUpdateBusySyscalls(target); @@ -3621,6 +3634,10 @@ static void __KernelRunCallbackOnThread(SceUID cbId, PSPThread *thread, bool res cb->nc.notifyCount = 0; cb->nc.notifyArg = 0; + // Setting up the call costs about 8us on top of the switch: 14us from a notify to a better + // thread's callback running on hardware (pspautotests threads/callbacks/combos). + currentMIPS->downcount -= 1800; + ActionAfterCallback *action = (ActionAfterCallback *) __KernelCreateAction(actionAfterCallback); if (action != NULL) action->setCallback(cbId); diff --git a/Core/HLE/sceKernelThread.h b/Core/HLE/sceKernelThread.h index 067daca441..904d4f2294 100644 --- a/Core/HLE/sceKernelThread.h +++ b/Core/HLE/sceKernelThread.h @@ -386,7 +386,10 @@ void __KernelWaitCurThread(WaitType type, SceUID waitId, u32 waitValue, u32 time bool __KernelWaitTimesOutAtOnce(u32 timeoutPtr, int basePercent = 85, int stepPercent = 35); s64 __KernelWaitTimeoutUs(u32 micro); // How long after its deadline a wait's timeout goes off. Not part of the time left written back. -const int WAIT_TIMEOUT_LATENCY_US = 30; +const int WAIT_TIMEOUT_LATENCY_US = 18; +// The deadline is taken this far into the call, after what we already charge before scheduling. +// threads/semaphores/wait and threads/fpl/cancel pin it between about 10 and 15us. +const int WAIT_TIMEOUT_DEADLINE_US = 12; void __KernelWaitCallbacksCurThread(WaitType type, SceUID waitID, u32 waitValue, u32 timeoutPtr); void __KernelReSchedule(const char *reason = "no reason"); void __KernelReSchedule(bool doCallbacks, const char *reason); diff --git a/test.py b/test.py index 6cc87065b4..78e41f170d 100755 --- a/test.py +++ b/test.py @@ -392,6 +392,7 @@ tests_good = [ "threads/mutex/unlock", "threads/mutex/unlock2", "threads/scheduling/dispatch", + "threads/callbacks/combos", "threads/scheduling/delayzero", "threads/scheduling/dispatchwake", "threads/scheduling/mutexhandoff", @@ -608,7 +609,6 @@ tests_next = [ # These two mbx tests only appeared to work because they papered over bugs - "threads/callbacks/combos", "threads/scheduling/scheduling", "threads/threads/create", "threads/tls/memory",