Threads: Let better threads preempt a long sceKernelCreateThread

On hardware the kernel fills a new thread's stack with interrupts on, so
a better thread that wakes during it runs before the call returns, the
time it takes doesn't count towards the call, and worse threads get
nothing (pspautotests threads/scheduling/preemptsyscall). PPSSPP ate the
whole cost at once and only rescheduled at the end.

__KernelBusyDelayResult() models such a syscall: the caller waits, an
idle thread stands in for it while nothing better wants the CPU, and its
remaining cycles only count down while that's the case. When done it
goes back ahead of threads of its own priority, having never given up
the CPU. sceKernelCreateThread uses it for the stack fill, unless a
thread event handler is about to run.

Booting 75 games against master shows no difference.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
Henrik RydgårdandClaude Opus 5.5 committed 2026-09-29 09:56:28 -06:00
1 parent 16135e4e2f
commit 8301c35d7d
2 files changed
+162 -10

No files matched your search

+156 -9
View File
@@ -484,6 +484,15 @@ bool __KernelCheckThreadCallbacks(PSPThread *thread, bool force);
static int g_inCbCount = 0;
static SceUID currentCallbackThreadID = 0;
static int readyCallbacksCount = 0;
// Syscalls keeping the CPU busy, see __KernelBusyDelayResult().
struct BusySyscall {
SceUID threadID;
s64 remainingCycles;
bool counting;
};
static std::vector<BusySyscall> busySyscalls;
static int eventBusySyscallDone = -1;
static void __KernelBusySyscallDone(u64 userdata, int cyclesLate);
static SceUID currentThread;
// When the running thread last changed, so each thread can be billed for the time it actually ran
// (nt.runForClocks). Not serialized - it's re-based on load, which only skews the very first slice.
@@ -802,6 +811,8 @@ void __KernelThreadingInit() {
eventScheduledWakeup = CoreTiming::RegisterEvent("ScheduledWakeup", &hleScheduledWakeup);
eventThreadEndTimeout = CoreTiming::RegisterEvent("ThreadEndTimeout", &hleThreadEndTimeout);
eventBusySyscallDone = CoreTiming::RegisterEvent("BusySyscallDone", &__KernelBusySyscallDone);
busySyscalls.clear();
actionAfterMipsCall = __KernelRegisterActionType(ActionAfterMipsCall::Create);
actionAfterCallback = __KernelRegisterActionType(ActionAfterCallback::Create);
actionAfterExitCallback = __KernelRegisterActionType(ActionAfterExitCallback::Create);
@@ -828,7 +839,7 @@ void __KernelThreadingDoState(PointerWrap &p)
g_exitCallbackPending = false;
}
auto s = p.Section("sceKernelThread", 1, 5);
auto s = p.Section("sceKernelThread", 1, 6);
if (!s)
return;
@@ -875,6 +886,22 @@ void __KernelThreadingDoState(PointerWrap &p)
Do(p, pausedDelays);
if (s >= 6) {
Do(p, eventBusySyscallDone);
u32 busyCount = (u32)busySyscalls.size();
Do(p, busyCount);
busySyscalls.resize(busyCount);
for (BusySyscall &busy : busySyscalls) {
Do(p, busy.threadID);
Do(p, busy.remainingCycles);
Do(p, busy.counting);
}
} else {
eventBusySyscallDone = -1;
busySyscalls.clear();
}
CoreTiming::RestoreRegisterEvent(eventBusySyscallDone, "BusySyscallDone", &__KernelBusySyscallDone);
__SetCurrentThread(kernelObjects.GetFast<PSPThread>(currentThread), currentThread, __KernelGetThreadName(currentThread));
lastSwitchCycles = CoreTiming::GetTicks(currentMIPS);
@@ -1601,11 +1628,121 @@ static void __ReportThreadQueueEmpty() {
}
// Returns NULL if the current thread is fine.
// A syscall that keeps the CPU busy for a long time, like sceKernelCreateThread filling a big
// stack. The kernel runs it with interrupts on, so a better thread that wakes meanwhile preempts
// it, and the time that thread takes doesn't count towards the syscall. Worse threads don't get a
// look in. (tests/threads/scheduling/preemptsyscall.) Modelled as a wait that only an idle thread
// may stand in for, whose countdown only runs while one does.
// hleDelayResult waits with id 1.
static const SceUID BUSY_SYSCALL_WAIT_ID = 2;
static bool __KernelIsIdleThread(SceUID threadID) {
return threadID == threadIdleID[0] || threadID == threadIdleID[1];
}
// Also drops entries for threads that stopped waiting (terminated, say.)
static PSPThread *__KernelBestBusyThread() {
PSPThread *best = nullptr;
for (size_t i = 0; i < busySyscalls.size(); ) {
u32 error;
PSPThread *t = kernelObjects.Get<PSPThread>(busySyscalls[i].threadID, error);
if (!t || !t->isWaitingFor(WAITTYPE_HLEDELAY, BUSY_SYSCALL_WAIT_ID)) {
if (busySyscalls[i].counting) {
CoreTiming::UnscheduleEvent(eventBusySyscallDone, busySyscalls[i].threadID);
}
busySyscalls.erase(busySyscalls.begin() + i);
continue;
}
if (!best || t->nt.currentPriority < best->nt.currentPriority) {
best = t;
}
++i;
}
return best;
}
// The best busy syscall makes progress only while an idle thread stands in for it.
static void __KernelUpdateBusySyscalls(PSPThread *target) {
if (busySyscalls.empty()) {
return;
}
PSPThread *best = __KernelBestBusyThread();
const bool idleStandsIn = target && __KernelIsIdleThread(target->GetUID());
for (BusySyscall &busy : busySyscalls) {
const bool shouldCount = idleStandsIn && best && busy.threadID == best->GetUID();
if (shouldCount && !busy.counting) {
CoreTiming::ScheduleEvent(busy.remainingCycles, eventBusySyscallDone, busy.threadID);
busy.counting = true;
} else if (!shouldCount && busy.counting) {
const s64 left = CoreTiming::UnscheduleEvent(eventBusySyscallDone, busy.threadID);
busy.remainingCycles = left > 0 ? left : 0;
busy.counting = false;
}
}
}
static void __KernelBusySyscallDone(u64 userdata, int cyclesLate) {
const SceUID threadID = (SceUID)userdata;
for (size_t i = 0; i < busySyscalls.size(); ++i) {
if (busySyscalls[i].threadID == threadID) {
busySyscalls.erase(busySyscalls.begin() + i);
break;
}
}
u32 error;
PSPThread *t = kernelObjects.Get<PSPThread>(threadID, error);
if (t && t->isWaitingFor(WAITTYPE_HLEDELAY, BUSY_SYSCALL_WAIT_ID)) {
__KernelResumeThreadFromWait(threadID, t->getWaitInfo().waitValue);
// It never gave up the CPU, so it goes back ahead of threads of the same priority.
if (t->nt.status == THREADSTATUS_READY) {
threadReadyQueue.remove(t->nt.currentPriority, threadID);
threadReadyQueue.push_front(t->nt.currentPriority, threadID);
}
__KernelReSchedule("busy syscall done");
}
}
u32 __KernelBusyDelayResult(u32 result, int cycles, const char *reason) {
if (cycles <= 0 || !__KernelIsDispatchEnabled() || __IsInInterrupt()) {
hleEatCycles(cycles);
return result;
}
busySyscalls.push_back(BusySyscall{ __KernelGetCurThread(), cycles, false });
__KernelWaitCurThread(WAITTYPE_HLEDELAY, BUSY_SYSCALL_WAIT_ID, result, 0, false, reason);
return result;
}
static PSPThread *__KernelNextThread() {
SceUID bestThread;
// If the current thread is running, it's a valid candidate.
PSPThread *cur = __GetCurrentThread();
// A busy syscall keeps the CPU unless something better wants it. An idle thread stands in.
PSPThread *busy = busySyscalls.empty() ? nullptr : __KernelBestBusyThread();
if (busy) {
const u32 busyPriority = busy->nt.currentPriority;
const SceUID bestReady = threadReadyQueue.peek_first();
PSPThread *bestReadyThread = bestReady != 0 ? kernelObjects.GetFast<PSPThread>(bestReady) : nullptr;
const bool readyOutranks = bestReadyThread && bestReadyThread->nt.currentPriority < busyPriority;
const bool curOutranks = cur && cur->isRunning() && !__KernelIsIdleThread(currentThread) && cur->nt.currentPriority < busyPriority;
if (!readyOutranks && !curOutranks) {
if (cur && cur->isRunning() && __KernelIsIdleThread(currentThread)) {
return nullptr;
}
for (SceUID idleID : threadIdleID) {
PSPThread *idle = kernelObjects.GetFast<PSPThread>(idleID);
if (idle && idle->nt.status == THREADSTATUS_READY) {
threadReadyQueue.remove(idle->nt.currentPriority, idleID);
if (cur && cur->isRunning()) {
__KernelChangeReadyState(cur, currentThread, true);
}
return idle;
}
}
}
}
if (cur && cur->isRunning()) {
bestThread = threadReadyQueue.pop_first_better(cur->nt.currentPriority);
if (bestThread != 0)
@@ -1850,7 +1987,7 @@ SceUID __KernelCreateThreadInternal(const char *threadName, SceUID moduleID, u32
}
// Note: Removed all the uses of hleReport* etc.
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel) {
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel, int *busyCyclesOut) {
if (!threadName) {
ERROR_LOG(Log::sceKernel, "__KernelCreateThread: NULL thread name");
return SCE_KERNEL_ERROR_ERROR;
@@ -1921,18 +2058,25 @@ int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32
// Measured by tests/threads/scheduling/costs: about 150us, plus filling the stack with 0xFF at
// around a cycle per byte - 1.3ms for a 256KB stack.
int createCycles = 32000;
int fillCycles = 0;
if ((attr & PSP_THREAD_ATTR_NO_FILLSTACK) == 0 && stacksize > 0) {
createCycles += stacksize - stacksize / 64;
fillCycles = stacksize - stacksize / 64;
}
hleEatCycles(createCycles);
hleEatCycles(32000);
// This won't schedule to the new thread, but it may to one woken from eating cycles.
// Technically, this should not eat all at once, and reschedule in the middle, but that's hard.
hleReSchedule("thread created");
// Before triggering, set v0, since we restore on return.
RETURN(id);
__KernelThreadTriggerEvent((attr & PSP_THREAD_ATTR_KERNEL) != 0, id, THREADEVENT_CREATE);
const bool handled = __KernelThreadTriggerEvent((attr & PSP_THREAD_ATTR_KERNEL) != 0, id, THREADEVENT_CREATE);
if (busyCyclesOut && !handled) {
*busyCyclesOut = fillCycles;
} else {
// A handler is about to run as a call on this thread, so no busy wait around it.
hleEatCycles(fillCycles);
if (busyCyclesOut)
*busyCyclesOut = 0;
}
return id;
}
@@ -1940,11 +2084,12 @@ int sceKernelCreateThread(const char *threadName, u32 entry, u32 prio, int stack
PSPThread *cur = __GetCurrentThread();
SceUID module = __KernelGetCurThreadModuleId();
bool allowKernel = KernelModuleIsKernelMode(module) || hleIsKernelMode() || (cur ? (cur->nt.attr & PSP_THREAD_ATTR_KERNEL) != 0 : false);
int retval = __KernelCreateThread(threadName, module, entry, prio, stacksize, attr, optionAddr, allowKernel);
int busyCycles = 0;
int retval = __KernelCreateThread(threadName, module, entry, prio, stacksize, attr, optionAddr, allowKernel, &busyCycles);
if (retval < 0) {
return hleLogError(Log::sceKernel, retval);
} else {
return hleLogInfo(Log::sceKernel, retval);
return __KernelBusyDelayResult(hleLogInfo(Log::sceKernel, retval), busyCycles, "thread stack filled");
}
}
@@ -3053,6 +3198,8 @@ void __KernelSwitchContext(PSPThread *target, const char *reason) {
currentMIPS->downcount -= 2700;
}
__KernelUpdateBusySyscalls(target);
if (target)
{
// No longer waiting.
+6 -1
View File
@@ -38,7 +38,12 @@ class BlockAllocator;
int sceKernelChangeThreadPriority(SceUID threadID, int priority);
SceUID __KernelCreateThreadInternal(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr);
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel);
// With busyCyclesOut, the cost of filling the stack is left for the caller to take (see
// __KernelBusyDelayResult) instead of being eaten here.
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel, int *busyCyclesOut = nullptr);
// For a syscall that keeps the CPU busy for a long time. The caller gets the result after the
// given cycles, but better threads that wake meanwhile run first, and worse ones don't run.
u32 __KernelBusyDelayResult(u32 result, int cycles, const char *reason);
int sceKernelCreateThread(const char *threadName, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr);
int sceKernelDelayThread(u32 usec);
int sceKernelDelayThreadCB(u32 usec);