Threads: Context switches cost what they do on hardware

Timed on a PSP, each way of handing the CPU to another thread (the call
and the switch together):

                                     hardware   before   now
  rotate to an equal thread              7        14       7
  signal, better thread runs            10        17      10
  it waits again, back to caller        10        19      12
  wakeup, better thread runs             8        13       6
  it sleeps again, back to caller        7        12       6
  start a better thread, entry          30        28      30
  thread ends, back to its waiter       21        13      20
  notify, better thread's callback      14        13      14

A switch between two threads now costs 1150 cycles instead of 2700.
Starting a better thread costs 2000 cycles more, ending a thread 3300,
and setting up a callback 1800.

Also splits a wait timeout's ~30us into the deadline being taken 12us
into the call and the timeout going off 18us after it. That only changes
the time left written back, which threads/semaphores/wait and
threads/fpl/cancel pin between them. intr/vblank is re-recorded so it
no longer depends on the phase of the frame.

threads/callbacks/combos now passes.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
Henrik RydgårdandClaude Opus 5.5 committed 2026-09-30 09:06:30 -06:00
1 parent f5b4fd739d
commit f26ca37dad
3 files changed
+27 -7

No files matched your search

+22 -5
View File
@@ -1539,8 +1539,9 @@ static bool __KernelChance(int percent) {
// wait fails with SCE_KERNEL_ERROR_WAIT_TIMEOUT at once, without giving up the CPU or writing the
// timeout back. For most calls that happens about 85% of the time for 0us, half the time for 1us,
// 15% for 2us and never from 3us on. sceKernelAllocateVpl does more first: always up to 1us, then
// about 17% less per us. Longer ones end max(timeout, 205us) plus about 35us after the call, the
// last few of which we already spend around the event.
// about 17% less per us. Longer ones end max(timeout, 205us) plus about 35us after the call. The
// deadline is taken late in the call, so the time left written back when something else ends the
// wait counts from there.
bool __KernelWaitTimesOutAtOnce(u32 timeoutPtr, int basePercent, int stepPercent) {
if (!Memory::IsValid4AlignedAddress(timeoutPtr)) {
return false;
@@ -1550,7 +1551,7 @@ bool __KernelWaitTimesOutAtOnce(u32 timeoutPtr, int basePercent, int stepPercent
}
s64 __KernelWaitTimeoutUs(u32 micro) {
return (s64)std::max(micro, 205U) + WAIT_TIMEOUT_LATENCY_US;
return (s64)std::max(micro, 205U) + WAIT_TIMEOUT_DEADLINE_US + WAIT_TIMEOUT_LATENCY_US;
}
void hleThreadEndTimeout(u64 userdata, int cyclesLate)
@@ -2199,6 +2200,9 @@ int __KernelStartThread(SceUID threadToStartID, int argSize, u32 argBlockPtr, bo
KernelValidateThreadTarget(startThread->context.pc);
__KernelChangeReadyState(cur, currentThread, true);
if (__InterruptsEnabled()) {
// Handing over costs more: about 30us from the call to the new thread's entry on
// hardware, where starting a worse one takes 10-25us.
hleEatCycles(2000);
g_startThreadHandoff = threadToStartID;
hleReSchedule("thread started");
}
@@ -2278,6 +2282,9 @@ int sceKernelGetThreadStackFreeSize(SceUID threadID) {
return hleLogDebug(Log::sceKernel, sz & ~3);
}
// What a thread ending costs, besides the switch away from it.
static const int THREAD_EXIT_CYCLES = 3300;
void __KernelReturnFromThread()
{
hleSkipDeadbeef();
@@ -2287,6 +2294,8 @@ void __KernelReturnFromThread()
_dbg_assert_msg_(thread != NULL, "Returned from a NULL thread.");
DEBUG_LOG(Log::sceKernel, "__KernelReturnFromThread: %d", exitStatus);
// About 20us from a thread ending to a thread waiting on it running, on hardware.
hleEatCycles(THREAD_EXIT_CYCLES);
__KernelStopThread(currentThread, exitStatus, "thread returned");
hleReSchedule("thread returned");
@@ -2308,6 +2317,7 @@ int sceKernelExitThread(int exitStatus) {
if (exitStatus < 0) {
exitStatus = SCE_KERNEL_ERROR_ILLEGAL_ARGUMENT;
}
hleEatCycles(THREAD_EXIT_CYCLES);
__KernelStopThread(currentThread, exitStatus, "thread exited");
hleReSchedule("thread exited");
@@ -2343,6 +2353,7 @@ int sceKernelExitDeleteThread(int exitStatus) {
INFO_LOG(Log::sceKernel,"sceKernelExitDeleteThread(%d)", exitStatus);
uint32_t thread_attr = thread->nt.attr;
uint32_t uid = thread->GetUID();
hleEatCycles(THREAD_EXIT_CYCLES);
__KernelDeleteThread(currentThread, exitStatus, "thread exited with delete");
hleReSchedule("thread exited with delete");
@@ -3265,13 +3276,15 @@ void __KernelSwitchContext(PSPThread *target, const char *reason) {
}
#endif
// Switching threads eats some cycles. This is a low approximation.
// Switching threads eats some cycles. Between two threads it's about 5us: rotating the ready
// queue to an equal thread, or waking a better one and having it block again, each take
// 7-10us on hardware including the calls themselves (pspautotests threads/scheduling/handoff).
if (fromIdle && toIdle) {
// Don't eat any cycles going between idle.
} else if (fromIdle || toIdle) {
currentMIPS->downcount -= 1200;
} else {
currentMIPS->downcount -= 2700;
currentMIPS->downcount -= 1150;
}
__KernelUpdateBusySyscalls(target);
@@ -3621,6 +3634,10 @@ static void __KernelRunCallbackOnThread(SceUID cbId, PSPThread *thread, bool res
cb->nc.notifyCount = 0;
cb->nc.notifyArg = 0;
// Setting up the call costs about 8us on top of the switch: 14us from a notify to a better
// thread's callback running on hardware (pspautotests threads/callbacks/combos).
currentMIPS->downcount -= 1800;
ActionAfterCallback *action = (ActionAfterCallback *) __KernelCreateAction(actionAfterCallback);
if (action != NULL)
action->setCallback(cbId);
+4 -1
View File
@@ -386,7 +386,10 @@ void __KernelWaitCurThread(WaitType type, SceUID waitId, u32 waitValue, u32 time
bool __KernelWaitTimesOutAtOnce(u32 timeoutPtr, int basePercent = 85, int stepPercent = 35);
s64 __KernelWaitTimeoutUs(u32 micro);
// How long after its deadline a wait's timeout goes off. Not part of the time left written back.
const int WAIT_TIMEOUT_LATENCY_US = 30;
const int WAIT_TIMEOUT_LATENCY_US = 18;
// The deadline is taken this far into the call, after what we already charge before scheduling.
// threads/semaphores/wait and threads/fpl/cancel pin it between about 10 and 15us.
const int WAIT_TIMEOUT_DEADLINE_US = 12;
void __KernelWaitCallbacksCurThread(WaitType type, SceUID waitID, u32 waitValue, u32 timeoutPtr);
void __KernelReSchedule(const char *reason = "no reason");
void __KernelReSchedule(bool doCallbacks, const char *reason);
+1 -1
View File
@@ -392,6 +392,7 @@ tests_good = [
"threads/mutex/unlock",
"threads/mutex/unlock2",
"threads/scheduling/dispatch",
"threads/callbacks/combos",
"threads/scheduling/delayzero",
"threads/scheduling/dispatchwake",
"threads/scheduling/mutexhandoff",
@@ -608,7 +609,6 @@ tests_next = [
# These two mbx tests only appeared to work because they papered over bugs
"threads/callbacks/combos",
"threads/scheduling/scheduling",
"threads/threads/create",
"threads/tls/memory",