Merge pull request #22383 from hrydgard/preemptible-syscalls

Fixes after hardware scheduling tests by Claude
This commit is contained in:
Henrik Rydgård authored and GitHub committed 2026-09-29 12:35:27 -06:00
commit 1b8553956f
16 files changed
+329 -72

No files matched your search

+3
View File
@@ -222,6 +222,8 @@ public:
virtual int ResetPlayPosition(int sample, int bytesWrittenFirstBuf, int bytesWrittenSecondBuf, bool *delay) = 0;
virtual int GetBufferInfoForResetting(AtracResetBufferInfo *bufferInfo, int sample, bool *delay) = 0; // NOTE: Not const! This can cause SkipFrames!
virtual int SetData(const Track &track, u32 buffer, u32 readSize, u32 bufferSize, u32 fileSize, int outputChannels, bool isAA3) = 0;
// How many frames the last SetData decoded and threw away, to get to the first sample.
int SkippedFramesOnSetData() const { return setDataSkippedFrames_; }
virtual int GetSecondBufferInfo(u32 *fileOffset, u32 *desiredSize) const = 0;
virtual int SetSecondBuffer(u32 secondBuffer, u32 secondBufferSize) = 0;
@@ -242,6 +244,7 @@ public:
protected:
u16 outputChannels_ = 2;
int setDataSkippedFrames_ = 0;
// TODO: Save the internal state of this, now technically possible.
AudioDecoder *decoder_ = nullptr;
+2 -1
View File
@@ -1019,8 +1019,9 @@ int Atrac2::SetData(const Track &track, u32 bufferAddr, u32 readSize, u32 buffer
info.fileDataEnd, info.decodePos, info.numSkipFrames, info.numChan
);
int skipCount = 0; // TODO: use for delay
int skipCount = 0;
retval = SkipFrames(&skipCount);
setDataSkippedFrames_ = skipCount;
// Seen in Mui Mui house. Things go very wrong after this..
if (retval == SCE_ERROR_ATRAC_API_FAIL) {
+33 -15
View File
@@ -32,8 +32,11 @@
#include "Core/HLE/sceKernel.h"
#include "Core/HLE/sceUtility.h"
#include "Core/HLE/sceVideocodec.h"
#include "Core/HLE/sceKernelMemory.h"
#include "Core/HLE/scePower.h"
#include "Core/HLE/sceAtrac.h"
#include "Core/HLE/sceAudiocodec.h"
#include "Core/HLE/AtracCtx.h"
#include "Core/HLE/AtracCtx2.h"
#include "Core/System.h"
@@ -81,7 +84,21 @@
// TODO: We should add checks that the utility module is loaded.
static const int atracDecodeDelay = 2300;
// The Media Engine does the decoding while the caller waits. Setting data decodes the frames before
// the first sample (thrown away), which on hardware makes it cost a decoder setup plus a frame
// decode: ~900us for mono Atrac3, ~3.5ms for stereo Atrac3+ (pspautotests threads/scheduling/callcosts).
static int AtracFrameUs(const AtracBase *atrac) {
return AudioCodecDecodeUs(atrac->CodecType(), atrac->Channels(), atrac->BytesPerFrame());
}
static int AtracSetDataDelay(const AtracBase *atrac) {
const int us = AudioCodecInitUs(atrac->CodecType(), atrac->Channels() == 1) + atrac->SkippedFramesOnSetData() * AtracFrameUs(atrac);
return MEScheduleJob(PowerScaleFromDefaultClock(us));
}
static int AtracDecodeDelay(const AtracBase *atrac) {
return MEScheduleJob(PowerScaleFromDefaultClock(AtracFrameUs(atrac)));
}
static bool atracInited = true;
static AtracBase *atracContexts[PSP_MAX_ATRAC_IDS];
@@ -94,7 +111,7 @@ static int g_atracBSS = 0;
static bool g_muteFlag[PSP_MAX_ATRAC_IDS]{}; // Not saved, just for debugging.
// On a PSP, the Media Engine does the decoding, and the samples land in the output buffer when
// sceAtracDecodeData returns, about atracDecodeDelay later. A game can still be playing out of that
// sceAtracDecodeData returns, a frame decode later. A game can still be playing out of that
// memory in the meantime: Fired Up decodes into a buffer that overlaps the first 16 samples of the
// one it has just handed to sceAudio, and relies on the mixer having read them first. So the
// samples are written just before the thread wakes rather than when the call is made.
@@ -423,6 +440,7 @@ static u32 sceAtracDecodeData(int atracID, u32 outAddr, u32 numSamplesAddr, u32
}
if (ret == 0 || ret == SCE_ERROR_ATRAC_API_FAIL) {
const int delay = AtracDecodeDelay(atrac);
const u32 written = std::min((u32)(atrac->GetOutputChannels() * 2 * numSamplesWritten), (u32)previous.size());
if (outPtr && written != 0) {
AtracPendingOutput pending{ ++g_pendingOutputId, outAddr };
@@ -430,10 +448,10 @@ static u32 sceAtracDecodeData(int atracID, u32 outAddr, u32 numSamplesAddr, u32
memcpy(outPtr, previous.data(), previous.size());
g_pendingOutput.push_back(std::move(pending));
// Just ahead of the thread waking up.
CoreTiming::ScheduleEvent(usToCycles(atracDecodeDelay) - 1, g_atracOutputEvent, g_pendingOutputId);
CoreTiming::ScheduleEvent(usToCycles(delay) - 1, g_atracOutputEvent, g_pendingOutputId);
}
// Decoded or at least attempted to decode data, delay thread
return hleDelayResult(hleNoLog(ret), "atrac decode data", atracDecodeDelay);
return hleDelayResult(hleNoLog(ret), "atrac decode data", delay);
}
return hleNoLog(ret);
@@ -753,7 +771,7 @@ static u32 sceAtracSetHalfwayBuffer(int atracID, u32 buffer, u32 readSize, u32 b
}
// not sure the real delay time
return hleDelayResult(hleLogDebug(Log::Atrac, ret), "atrac set data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, ret), "atrac set data", AtracSetDataDelay(atrac));
}
static u32 sceAtracSetSecondBuffer(int atracID, u32 secondBuffer, u32 secondBufferSize) {
@@ -788,7 +806,7 @@ static u32 sceAtracSetData(int atracID, u32 buffer, u32 bufferSize) {
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebug(Log::Atrac, ret), "atrac set data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, ret), "atrac set data", AtracSetDataDelay(atrac));
}
static int sceAtracSetDataAndGetID(u32 buffer, int bufferSize) {
@@ -819,7 +837,7 @@ static int sceAtracSetDataAndGetID(u32 buffer, int bufferSize) {
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", AtracSetDataDelay(atracContexts[atracID]));
}
static int sceAtracSetHalfwayBufferAndGetID(u32 buffer, u32 readSize, u32 bufferSize) {
@@ -845,7 +863,7 @@ static int sceAtracSetHalfwayBufferAndGetID(u32 buffer, u32 readSize, u32 buffer
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", AtracSetDataDelay(atracContexts[atracID]));
}
static u32 sceAtracStartEntry() {
@@ -959,7 +977,7 @@ static int sceAtracSetMOutHalfwayBuffer(int atracID, u32 buffer, u32 readSize, u
// Must not delay.
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebugOrError(Log::Atrac, ret), "atrac set data mono", 100);
return hleDelayResult(hleLogDebugOrError(Log::Atrac, ret), "atrac set data mono", AtracSetDataDelay(atrac));
}
// Note: This doesn't seem to be part of any available libatrac3plus library.
@@ -985,7 +1003,7 @@ static u32 sceAtracSetMOutData(int atracID, u32 buffer, u32 bufferSize) {
return hleLogError(Log::Atrac, ret);
}
// It's OK if this fails, at least with NO_MONO...
return hleDelayResult(hleLogDebugOrError(Log::Atrac, ret), "atrac set data mono", 100);
return hleDelayResult(hleLogDebugOrError(Log::Atrac, ret), "atrac set data mono", AtracSetDataDelay(atrac));
}
// Note: This doesn't seem to be part of any available libatrac3plus library.
@@ -1012,7 +1030,7 @@ static int sceAtracSetMOutDataAndGetID(u32 buffer, u32 bufferSize) {
UnregisterAndDeleteAtrac(atracID);
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebugOrError(Log::Atrac, atracID), "atrac set data", 100);
return hleDelayResult(hleLogDebugOrError(Log::Atrac, atracID), "atrac set data", AtracSetDataDelay(atracContexts[atracID]));
}
static int sceAtracSetMOutHalfwayBufferAndGetID(u32 buffer, u32 readSize, u32 bufferSize) {
@@ -1041,7 +1059,7 @@ static int sceAtracSetMOutHalfwayBufferAndGetID(u32 buffer, u32 readSize, u32 bu
UnregisterAndDeleteAtrac(atracID);
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", AtracSetDataDelay(atracContexts[atracID]));
}
static int sceAtracSetAA3DataAndGetID(u32 buffer, u32 bufferSize, u32 fileSize, u32 metadataSizeAddr) {
@@ -1063,7 +1081,7 @@ static int sceAtracSetAA3DataAndGetID(u32 buffer, u32 bufferSize, u32 fileSize,
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set aa3 data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set aa3 data", AtracSetDataDelay(atracContexts[atracID]));
}
static int sceAtracSetAA3HalfwayBufferAndGetID(u32 buffer, u32 readSize, u32 bufferSize, u32 fileSize) {
@@ -1089,7 +1107,7 @@ static int sceAtracSetAA3HalfwayBufferAndGetID(u32 buffer, u32 readSize, u32 buf
return hleLogError(Log::Atrac, ret);
}
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", 100);
return hleDelayResult(hleLogDebug(Log::Atrac, atracID), "atrac set data", AtracSetDataDelay(atracContexts[atracID]));
}
// TODO: Should see if these are stored contiguously in memory somewhere, or if there really are
@@ -1155,7 +1173,7 @@ static int sceAtracLowLevelDecode(int atracID, u32 sourceAddr, u32 sourceBytesCo
}
NotifyMemInfo(MemBlockFlags::WRITE, samplesAddr, bytesWritten, "AtracLowLevelDecode");
return hleDelayResult(hleLogDebug(Log::Atrac, retval), "low level atrac decode data", atracDecodeDelay);
return hleDelayResult(hleLogDebug(Log::Atrac, retval), "low level atrac decode data", AtracDecodeDelay(atrac));
}
// These three are the external interface used by sceSas' AT3 integration.
+11 -2
View File
@@ -239,15 +239,19 @@ static int MECall(int result, int us) {
return hleDelayResult(result, "audiocodec", MEScheduleJob(PowerScaleFromDefaultClock(us)));
}
static int InitUs(int codec, const SceAudiocodecCodec *ctx) {
int AudioCodecInitUs(int codec, bool monoAt3Plus) {
switch (codec) {
case PSP_CODEC_AT3PLUS: return ((ctx->fmt.at3.formatByte1 >> 2) & 7) == 1 ? 524 : 646;
case PSP_CODEC_AT3PLUS: return monoAt3Plus ? 524 : 646;
case PSP_CODEC_AT3: return 210;
case PSP_CODEC_MP3: return 517;
default: return 230;
}
}
static int InitUs(int codec, const SceAudiocodecCodec *ctx) {
return AudioCodecInitUs(codec, codec == PSP_CODEC_AT3PLUS && ((ctx->fmt.at3.formatByte1 >> 2) & 7) == 1);
}
// libmp4.prx puts the sample rate here. On hardware 22050 and 44100 are accepted, 0 and 12345
// aren't; that the rest of the standard AAC rates are accepted is an assumption. 0 if not valid.
static int AacSampleRateFromContext(const SceAudiocodecCodec *ctx) {
@@ -395,6 +399,11 @@ static int EstimateDecodeUs(int codec, int channels, int frameBytes, const SceAu
}
}
int AudioCodecDecodeUs(int codec, int channels, int frameBytes) {
_dbg_assert_(codec == PSP_CODEC_AT3PLUS || codec == PSP_CODEC_AT3);
return EstimateDecodeUs(codec, channels, frameBytes, nullptr);
}
static int sceAudiocodecInit(u32 ctxPtr, int codec) {
return __AudioCodecInitCommon(ctxPtr, codec, false);
}
+5
View File
@@ -128,3 +128,8 @@ class AudioDecoder;
extern std::map<u32, AudioDecoder *> g_audioDecoderContexts;
bool IsAtrac3StreamJointStereo(int codecType, int bytesPerFrame, int channels);
// ME time at the default clock, for sceAtrac too (it drives the same decoder): setting up a decoder,
// and decoding one frame. AudioCodecDecodeUs takes Atrac3 and Atrac3+ only.
int AudioCodecInitUs(int codec, bool monoAt3Plus);
int AudioCodecDecodeUs(int codec, int channels, int frameBytes);
+41 -18
View File
@@ -1053,20 +1053,39 @@ static u32 npdrmRead(FileNode *f, u8 *data, int size) {
return size;
}
// With dispatch suspended, a call fails in the driver when it tries to wait. The memory stick driver
// returns SCE_KERNEL_ERROR_CAN_NOT_WAIT, while usbhostfs (host0: under PSPLink, what homebrew
// developers run from) returns -1, which is what host0: does here too.
static u32 IoDispatchDisabledError(const std::string &filename) {
std::string outPath;
IFileSystem *system = nullptr;
IFileSystem *host = pspFileSystem.GetSystem("host0:");
if (host && pspFileSystem.MapFilePath(filename, &outPath, &system) == 0 && system == host) {
return (u32)-1;
}
return SCE_KERNEL_ERROR_CAN_NOT_WAIT;
}
int __IoOpenDelayUs(const char *filename) {
// UMD: Speed varies from 1-6ms.
// Card: Path depth matters, but typically between 10-13ms on a standard Pro Duo.
return pspFileSystem.FlagsFromFilename(filename) & FileSystemFlags::UMD ? 4000 : 10000;
}
int __IoReadDelayUs(int size) {
int us;
if (PSP_CoreParameter().compat.flags().ForceUMDReadSpeed || g_Config.iIOTimingMethod == IOTIMING_UMDSLOWREALISTIC) {
us = size / 4.2;
} else {
us = size / 100;
}
return std::max(us, 100);
}
static bool __IoRead(int &result, int id, u32 data_addr, int size, int &us) {
PROFILE_THIS_SCOPE("io_rw");
// Low estimate, may be improved later from the ReadFile result.
if (PSP_CoreParameter().compat.flags().ForceUMDReadSpeed || g_Config.iIOTimingMethod == IOTIMING_UMDSLOWREALISTIC) {
us = size / 4.2;
}
else {
us = size / 100;
}
if (us < 100) {
us = 100;
}
us = __IoReadDelayUs(size);
if (id == PSP_STDIN) {
DEBUG_LOG(Log::sceIo, "sceIoRead STDIN");
@@ -1147,8 +1166,11 @@ static u32 sceIoRead(int id, u32 data_addr, int size) {
}
if (id > 2) {
if (f->asyncBusy()) {
return hleLogWarning(Log::sceIo, SCE_KERNEL_ERROR_ASYNC_BUSY, "async busy");
}
if (!__KernelIsDispatchEnabled()) {
return hleLogError(Log::sceIo, SCE_KERNEL_ERROR_CAN_NOT_WAIT, "dispatch disabled");
return hleLogError(Log::sceIo, IoDispatchDisabledError(f->fullpath), "dispatch disabled");
}
if (__IsInInterrupt()) {
return hleLogError(Log::sceIo, SCE_KERNEL_ERROR_ILLEGAL_CONTEXT, "inside interrupt");
@@ -1290,7 +1312,7 @@ static u32 sceIoWrite(int id, u32 data_addr, int size) {
FileNode *f = __IoGetFd(id, error);
if (id > 2 && f != NULL) {
if (!__KernelIsDispatchEnabled()) {
return hleLogError(Log::sceIo, SCE_KERNEL_ERROR_CAN_NOT_WAIT, "dispatch disabled");
return hleLogError(Log::sceIo, IoDispatchDisabledError(f->fullpath), "dispatch disabled");
}
if (__IsInInterrupt()) {
return hleLogError(Log::sceIo, SCE_KERNEL_ERROR_ILLEGAL_CONTEXT, "inside interrupt");
@@ -1315,6 +1337,10 @@ static u32 sceIoWrite(int id, u32 data_addr, int size) {
if (__IsInInterrupt()) {
return hleLogError(Log::sceIo, SCE_KERNEL_ERROR_ILLEGAL_CONTEXT);
}
if (id <= 2) {
// A write to stdout or stderr doesn't give up the CPU (pspautotests threads/scheduling/dispatch).
return hleLogDebug(Log::sceIo, result);
}
return hleDelayResult(hleLogDebug(Log::sceIo, result), "io write", us);
} else {
return hleLogDebug(Log::sceIo, result);
@@ -1606,7 +1632,7 @@ static u32 sceIoOpen(const char *filename, int flags, int mode) {
if (!__KernelIsDispatchEnabled()) {
hleEatCycles(48000);
return hleLogError(Log::sceIo, SCE_KERNEL_ERROR_CAN_NOT_WAIT, "dispatch disabled");
return hleLogError(Log::sceIo, IoDispatchDisabledError(filename), "dispatch disabled");
}
int error;
@@ -1641,10 +1667,7 @@ static u32 sceIoOpen(const char *filename, int flags, int mode) {
// These are fast to open, no delay or even rescheduling happens.
return hleLogDebug(Log::sceIo, id);
}
// UMD: Speed varies from 1-6ms.
// Card: Path depth matters, but typically between 10-13ms on a standard Pro Duo.
int delay = pspFileSystem.FlagsFromFilename(filename) & FileSystemFlags::UMD ? 4000 : 10000;
return hleDelayResult(hleLogDebug(Log::sceIo, id), "file opened", delay);
return hleDelayResult(hleLogDebug(Log::sceIo, id), "file opened", __IoOpenDelayUs(filename));
}
}
+3
View File
@@ -38,6 +38,9 @@ u32 sceIoIoctl(u32 id, u32 cmd, u32 indataPtr, u32 inlen, u32 outdataPtr, u32 ou
int __IoIoctl(u32 id, u32 cmd, u32 indataPtr, u32 inlen, u32 outdataPtr, u32 outlen, int &usec);
u32 __IoGetFileHandleFromId(u32 id, u32 &outError);
// What sceIoOpen and sceIoRead charge, for loaders that read whole files without going through them.
int __IoOpenDelayUs(const char *filename);
int __IoReadDelayUs(int size);
void ConvertTmToPspDateTime(ScePspDateTime& date_out, const tm& date_in, int microSeconds);
KernelObject *__KernelFileNodeObject();
+13 -3
View File
@@ -2453,6 +2453,7 @@ u32 sceKernelLoadModule(const char *name, u32 flags, u32 optionAddr) {
const u32 error = hleLogError(Log::Loader, SCE_KERNEL_ERROR_FILEERR, "module file size is 0");
return hleDelayResult(error, "module loaded", 500);
}
const int fileSize = (int)fileData.size();
// A .sprx installed by a PKG game update comes wrapped in an NPDRM "\0PSPEDAT" container: a
// 0x90-byte header naming the content ID, then the payload at the offset in its u16 at 0x0C.
@@ -2544,8 +2545,11 @@ u32 sceKernelLoadModule(const char *name, u32 flags, u32 optionAddr) {
INFO_LOG(Log::sceModule,"%i=sceKernelLoadModule(name=%s,flag=%08x,(...))", module->GetUID(), name, flags);
}
// TODO: This is not the right timing and probably not the right wait type, just an approximation.
return hleDelayResult(hleNoLog(module->GetUID()), "module loaded", 500);
// Opening and reading the file, then about 1ms plus 30us per KB of loader work, the caller waiting
// throughout. The loader part is what a load took on hardware beyond an open and a read of the
// same file (pspautotests threads/scheduling/callcosts, from an SD card in a memory stick adapter).
const int loadUs = __IoOpenDelayUs(name) + __IoReadDelayUs(fileSize) + 1000 + fileSize / 34;
return hleDelayResult(hleNoLog(module->GetUID()), "module loaded", loadUs);
}
static u32 sceKernelLoadModuleNpDrm(const char *name, u32 flags, u32 optionAddr) {
@@ -2714,7 +2718,9 @@ static u32 sceKernelUnloadModule(u32 moduleId) {
module->Cleanup();
kernelObjects.Destroy<PSPModule>(moduleId);
return hleDelayResult(hleLogDebug(Log::sceModule, moduleId), "module unloaded", 500);
// About 400us of work that better threads can preempt, and worse ones don't get in on
// (tests/threads/scheduling/syscallkinds).
return __KernelBusyDelayResult(hleLogDebug(Log::sceModule, moduleId), (int)usToCycles(400), "module unloaded");
}
u32 __KernelStopUnloadSelfModuleWithOrWithoutStatus(u32 exitCode, u32 argSize, u32 argp, u32 statusAddr, u32 optionAddr, bool WithStatus) {
@@ -2982,6 +2988,10 @@ u32 sceKernelFindModuleByName(const char *name)
}
// The id in question here is a file handle.
// On hardware, from a game's own fd on ms0: or host0: this fails with
// SCE_KERNEL_ERROR_PROHIBIT_LOADMODULE_DEVICE: the file has to pass a kernel-only ioctl (0x00208001),
// which a user fd doesn't (it returns ILLEGAL_PERM). sceKernelLoadModule opens the file itself, so it
// passes. Whether a disc0:/umd0: fd passes is untested, so we let every device through.
static u32 sceKernelLoadModuleByID(u32 id, u32 flags, u32 lmoptionPtr) {
u32 error;
u32 handle = __IoGetFileHandleFromId(id, error);
+189 -14
View File
@@ -419,7 +419,7 @@ void PSPThread::Cleanup() {
}
void PSPThread::DoState(PointerWrap &p) {
auto s = p.Section("Thread", 1, 5);
auto s = p.Section("Thread", 1, 6);
if (!s)
return;
@@ -460,6 +460,11 @@ void PSPThread::DoState(PointerWrap &p) {
Do(p, waitingThreads);
Do(p, pausedWaits);
}
if (s >= 6) {
Do(p, hasWaited);
} else {
hasWaited = true;
}
}
@@ -484,6 +489,15 @@ bool __KernelCheckThreadCallbacks(PSPThread *thread, bool force);
static int g_inCbCount = 0;
static SceUID currentCallbackThreadID = 0;
static int readyCallbacksCount = 0;
// Syscalls keeping the CPU busy, see __KernelBusyDelayResult().
struct BusySyscall {
SceUID threadID;
s64 remainingCycles;
bool counting;
};
static std::vector<BusySyscall> busySyscalls;
static int eventBusySyscallDone = -1;
static void __KernelBusySyscallDone(u64 userdata, int cyclesLate);
static SceUID currentThread;
// When the running thread last changed, so each thread can be billed for the time it actually ran
// (nt.runForClocks). Not serialized - it's re-based on load, which only skews the very first slice.
@@ -802,6 +816,8 @@ void __KernelThreadingInit() {
eventScheduledWakeup = CoreTiming::RegisterEvent("ScheduledWakeup", &hleScheduledWakeup);
eventThreadEndTimeout = CoreTiming::RegisterEvent("ThreadEndTimeout", &hleThreadEndTimeout);
eventBusySyscallDone = CoreTiming::RegisterEvent("BusySyscallDone", &__KernelBusySyscallDone);
busySyscalls.clear();
actionAfterMipsCall = __KernelRegisterActionType(ActionAfterMipsCall::Create);
actionAfterCallback = __KernelRegisterActionType(ActionAfterCallback::Create);
actionAfterExitCallback = __KernelRegisterActionType(ActionAfterExitCallback::Create);
@@ -828,7 +844,7 @@ void __KernelThreadingDoState(PointerWrap &p)
g_exitCallbackPending = false;
}
auto s = p.Section("sceKernelThread", 1, 5);
auto s = p.Section("sceKernelThread", 1, 6);
if (!s)
return;
@@ -875,6 +891,22 @@ void __KernelThreadingDoState(PointerWrap &p)
Do(p, pausedDelays);
if (s >= 6) {
Do(p, eventBusySyscallDone);
u32 busyCount = (u32)busySyscalls.size();
Do(p, busyCount);
busySyscalls.resize(busyCount);
for (BusySyscall &busy : busySyscalls) {
Do(p, busy.threadID);
Do(p, busy.remainingCycles);
Do(p, busy.counting);
}
} else {
eventBusySyscallDone = -1;
busySyscalls.clear();
}
CoreTiming::RestoreRegisterEvent(eventBusySyscallDone, "BusySyscallDone", &__KernelBusySyscallDone);
__SetCurrentThread(kernelObjects.GetFast<PSPThread>(currentThread), currentThread, __KernelGetThreadName(currentThread));
lastSwitchCycles = CoreTiming::GetTicks(currentMIPS);
@@ -1416,6 +1448,7 @@ void __KernelWaitCurThread(WaitType type, SceUID waitID, u32 waitValue, u32 time
WARN_LOG_REPORT(Log::sceKernel, "Waiting thread for %d that was already waiting for %d", type, thread->nt.waitType);
thread->nt.waitID = waitID;
thread->nt.waitType = type;
thread->hasWaited = true;
__KernelChangeThreadState(thread, ThreadStatus(THREADSTATUS_WAIT | (thread->nt.status & THREADSTATUS_SUSPEND)));
thread->nt.numReleases++;
thread->waitInfo.waitValue = waitValue;
@@ -1601,11 +1634,121 @@ static void __ReportThreadQueueEmpty() {
}
// Returns NULL if the current thread is fine.
// A syscall that keeps the CPU busy for a long time, like sceKernelCreateThread filling a big
// stack. The kernel runs it with interrupts on, so a better thread that wakes meanwhile preempts
// it, and the time that thread takes doesn't count towards the syscall. Worse threads don't get a
// look in. (tests/threads/scheduling/preemptsyscall.) Modelled as a wait that only an idle thread
// may stand in for, whose countdown only runs while one does.
// hleDelayResult waits with id 1.
static const SceUID BUSY_SYSCALL_WAIT_ID = 2;
static bool __KernelIsIdleThread(SceUID threadID) {
return threadID == threadIdleID[0] || threadID == threadIdleID[1];
}
// Also drops entries for threads that stopped waiting (terminated, say.)
static PSPThread *__KernelBestBusyThread() {
PSPThread *best = nullptr;
for (size_t i = 0; i < busySyscalls.size(); ) {
u32 error;
PSPThread *t = kernelObjects.Get<PSPThread>(busySyscalls[i].threadID, error);
if (!t || !t->isWaitingFor(WAITTYPE_HLEDELAY, BUSY_SYSCALL_WAIT_ID)) {
if (busySyscalls[i].counting) {
CoreTiming::UnscheduleEvent(eventBusySyscallDone, busySyscalls[i].threadID);
}
busySyscalls.erase(busySyscalls.begin() + i);
continue;
}
if (!best || t->nt.currentPriority < best->nt.currentPriority) {
best = t;
}
++i;
}
return best;
}
// The best busy syscall makes progress only while an idle thread stands in for it.
static void __KernelUpdateBusySyscalls(PSPThread *target) {
if (busySyscalls.empty()) {
return;
}
PSPThread *best = __KernelBestBusyThread();
const bool idleStandsIn = target && __KernelIsIdleThread(target->GetUID());
for (BusySyscall &busy : busySyscalls) {
const bool shouldCount = idleStandsIn && best && busy.threadID == best->GetUID();
if (shouldCount && !busy.counting) {
CoreTiming::ScheduleEvent(busy.remainingCycles, eventBusySyscallDone, busy.threadID);
busy.counting = true;
} else if (!shouldCount && busy.counting) {
const s64 left = CoreTiming::UnscheduleEvent(eventBusySyscallDone, busy.threadID);
busy.remainingCycles = left > 0 ? left : 0;
busy.counting = false;
}
}
}
static void __KernelBusySyscallDone(u64 userdata, int cyclesLate) {
const SceUID threadID = (SceUID)userdata;
for (size_t i = 0; i < busySyscalls.size(); ++i) {
if (busySyscalls[i].threadID == threadID) {
busySyscalls.erase(busySyscalls.begin() + i);
break;
}
}
u32 error;
PSPThread *t = kernelObjects.Get<PSPThread>(threadID, error);
if (t && t->isWaitingFor(WAITTYPE_HLEDELAY, BUSY_SYSCALL_WAIT_ID)) {
__KernelResumeThreadFromWait(threadID, t->getWaitInfo().waitValue);
// It never gave up the CPU, so it goes back ahead of threads of the same priority.
if (t->nt.status == THREADSTATUS_READY) {
threadReadyQueue.remove(t->nt.currentPriority, threadID);
threadReadyQueue.push_front(t->nt.currentPriority, threadID);
}
__KernelReSchedule("busy syscall done");
}
}
u32 __KernelBusyDelayResult(u32 result, int cycles, const char *reason) {
if (cycles <= 0 || !__KernelIsDispatchEnabled() || __IsInInterrupt()) {
hleEatCycles(cycles);
return result;
}
busySyscalls.push_back(BusySyscall{ __KernelGetCurThread(), cycles, false });
__KernelWaitCurThread(WAITTYPE_HLEDELAY, BUSY_SYSCALL_WAIT_ID, result, 0, false, reason);
return result;
}
static PSPThread *__KernelNextThread() {
SceUID bestThread;
// If the current thread is running, it's a valid candidate.
PSPThread *cur = __GetCurrentThread();
// A busy syscall keeps the CPU unless something better wants it. An idle thread stands in.
PSPThread *busy = busySyscalls.empty() ? nullptr : __KernelBestBusyThread();
if (busy) {
const u32 busyPriority = busy->nt.currentPriority;
const SceUID bestReady = threadReadyQueue.peek_first();
PSPThread *bestReadyThread = bestReady != 0 ? kernelObjects.GetFast<PSPThread>(bestReady) : nullptr;
const bool readyOutranks = bestReadyThread && bestReadyThread->nt.currentPriority < busyPriority;
const bool curOutranks = cur && cur->isRunning() && !__KernelIsIdleThread(currentThread) && cur->nt.currentPriority < busyPriority;
if (!readyOutranks && !curOutranks) {
if (cur && cur->isRunning() && __KernelIsIdleThread(currentThread)) {
return nullptr;
}
for (SceUID idleID : threadIdleID) {
PSPThread *idle = kernelObjects.GetFast<PSPThread>(idleID);
if (idle && idle->nt.status == THREADSTATUS_READY) {
threadReadyQueue.remove(idle->nt.currentPriority, idleID);
if (cur && cur->isRunning()) {
__KernelChangeReadyState(cur, currentThread, true);
}
return idle;
}
}
}
}
if (cur && cur->isRunning()) {
bestThread = threadReadyQueue.pop_first_better(cur->nt.currentPriority);
if (bestThread != 0)
@@ -1747,6 +1890,7 @@ void __KernelResetThread(PSPThread *t, int lowestPriority) {
t->nt.exitStatus = SCE_KERNEL_ERROR_NOT_DORMANT;
t->isProcessingCallbacks = false;
t->hasWaited = false;
t->currentCallbackId = 0;
t->currentMipscallId = 0;
t->pendingMipsCalls.clear();
@@ -1850,7 +1994,7 @@ SceUID __KernelCreateThreadInternal(const char *threadName, SceUID moduleID, u32
}
// Note: Removed all the uses of hleReport* etc.
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel) {
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel, int *busyCyclesOut) {
if (!threadName) {
ERROR_LOG(Log::sceKernel, "__KernelCreateThread: NULL thread name");
return SCE_KERNEL_ERROR_ERROR;
@@ -1921,18 +2065,25 @@ int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32
// Measured by tests/threads/scheduling/costs: about 150us, plus filling the stack with 0xFF at
// around a cycle per byte - 1.3ms for a 256KB stack.
int createCycles = 32000;
int fillCycles = 0;
if ((attr & PSP_THREAD_ATTR_NO_FILLSTACK) == 0 && stacksize > 0) {
createCycles += stacksize - stacksize / 64;
fillCycles = stacksize - stacksize / 64;
}
hleEatCycles(createCycles);
hleEatCycles(32000);
// This won't schedule to the new thread, but it may to one woken from eating cycles.
// Technically, this should not eat all at once, and reschedule in the middle, but that's hard.
hleReSchedule("thread created");
// Before triggering, set v0, since we restore on return.
RETURN(id);
__KernelThreadTriggerEvent((attr & PSP_THREAD_ATTR_KERNEL) != 0, id, THREADEVENT_CREATE);
const bool handled = __KernelThreadTriggerEvent((attr & PSP_THREAD_ATTR_KERNEL) != 0, id, THREADEVENT_CREATE);
if (busyCyclesOut && !handled) {
*busyCyclesOut = fillCycles;
} else {
// A handler is about to run as a call on this thread, so no busy wait around it.
hleEatCycles(fillCycles);
if (busyCyclesOut)
*busyCyclesOut = 0;
}
return id;
}
@@ -1940,11 +2091,12 @@ int sceKernelCreateThread(const char *threadName, u32 entry, u32 prio, int stack
PSPThread *cur = __GetCurrentThread();
SceUID module = __KernelGetCurThreadModuleId();
bool allowKernel = KernelModuleIsKernelMode(module) || hleIsKernelMode() || (cur ? (cur->nt.attr & PSP_THREAD_ATTR_KERNEL) != 0 : false);
int retval = __KernelCreateThread(threadName, module, entry, prio, stacksize, attr, optionAddr, allowKernel);
int busyCycles = 0;
int retval = __KernelCreateThread(threadName, module, entry, prio, stacksize, attr, optionAddr, allowKernel, &busyCycles);
if (retval < 0) {
return hleLogError(Log::sceKernel, retval);
} else {
return hleLogInfo(Log::sceKernel, retval);
return __KernelBusyDelayResult(hleLogInfo(Log::sceKernel, retval), busyCycles, "thread stack filled");
}
}
@@ -2461,10 +2613,30 @@ static s64 __KernelDelayThreadUs(u64 usec) {
return usec + 10;
}
// A delay's deadline is now + usec, and the clock is read again when the alarm is set. If the
// deadline has passed by then, the call returns 0 at once without giving up the CPU. The two reads are about 0.6us apart, so a delay of 0 returns at
// once about 60% of the time, and 1 almost never. On a thread's first wait after it starts, when the
// code is presumably out of the cache, they're about 1.65us apart: a delay of 1 returns at once
// about two times in three, and 2 never (pspautotests threads/scheduling/delayzero).
// Our cycle counts are too regular to use the tick phase (a polling loop could lock into never
// yielding), so it's pseudo-random off the tick count instead.
static bool __KernelDelayReturnsAtOnce(u32 usec) {
const PSPThread *thread = __GetCurrentThread();
const int gapPercent = thread && !thread->hasWaited ? 165 : 60;
const int chancePercent = gapPercent - (int)std::min(usec, 2U) * 100;
if (chancePercent <= 0) {
return false;
}
u64 x = (u64)CoreTiming::GetTicks(currentMIPS) * 0x9E3779B97F4A7C15ULL;
x ^= x >> 29;
return (int)(x % 100) < chancePercent;
}
int sceKernelDelayThreadCB(u32 usec) {
hleEatCycles(2000);
// Note: Sometimes (0) won't delay, potentially based on how much the thread is doing.
// But a loop with just 0 often does delay, and games depend on this. So we err on that side.
if (__KernelDelayReturnsAtOnce(usec) && !__KernelCurHasReadyCallbacks()) {
return hleLogDebug(Log::sceKernel, 0, "deadline already passed");
}
SceUID curThread = __KernelGetCurThread();
s64 delayUs = __KernelDelayThreadUs(usec);
__KernelScheduleWakeup(curThread, delayUs);
@@ -2474,8 +2646,9 @@ int sceKernelDelayThreadCB(u32 usec) {
int sceKernelDelayThread(u32 usec) {
hleEatCycles(2000);
// Note: Sometimes (0) won't delay, potentially based on how much the thread is doing.
// But a loop with just 0 often does delay, and games depend on this. So we err on that side.
if (__KernelDelayReturnsAtOnce(usec)) {
return hleLogDebug(Log::sceKernel, 0, "deadline already passed");
}
SceUID curThread = __KernelGetCurThread();
s64 delayUs = __KernelDelayThreadUs(usec);
__KernelScheduleWakeup(curThread, delayUs);
@@ -3053,6 +3226,8 @@ void __KernelSwitchContext(PSPThread *target, const char *reason) {
currentMIPS->downcount -= 2700;
}
__KernelUpdateBusySyscalls(target);
if (target)
{
// No longer waiting.
+8 -1
View File
@@ -38,7 +38,12 @@ class BlockAllocator;
int sceKernelChangeThreadPriority(SceUID threadID, int priority);
SceUID __KernelCreateThreadInternal(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr);
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel);
// With busyCyclesOut, the cost of filling the stack is left for the caller to take (see
// __KernelBusyDelayResult) instead of being eaten here.
int __KernelCreateThread(const char *threadName, SceUID moduleID, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr, bool allowKernel, int *busyCyclesOut = nullptr);
// For a syscall that keeps the CPU busy for a long time. The caller gets the result after the
// given cycles, but better threads that wake meanwhile run first, and worse ones don't run.
u32 __KernelBusyDelayResult(u32 result, int cycles, const char *reason);
int sceKernelCreateThread(const char *threadName, u32 entry, u32 prio, int stacksize, u32 attr, u32 optionAddr);
int sceKernelDelayThread(u32 usec);
int sceKernelDelayThreadCB(u32 usec);
@@ -300,6 +305,8 @@ public:
KernelThreadDebugInterface debug;
bool isProcessingCallbacks = false;
// False until the thread first waits after being started (see __KernelDelayReturnsAtOnce).
bool hasWaited = true;
u32 currentMipscallId = -1;
SceUID currentCallbackId = -1;
+3 -5
View File
@@ -334,11 +334,9 @@ static int sceKernelVolatileMemTryLock(int type, u32 paddr, u32 psize) {
switch (error) {
case 0:
// HACK: This fixes Crash Tag Team Racing.
// Should only wait 1200 cycles though according to Unknown's testing,
// and with that it's still broken. So it's not this, unfortunately.
// Leaving it in for the 0.9.8 release anyway.
hleEatCycles(500000);
// Under 100us on hardware (tests/threads/scheduling/syscallkinds). This used to eat 500000
// cycles as a hack for Crash Tag Team Racing, which no longer needs it.
hleEatCycles(1200);
DEBUG_LOG(Log::HLE, "sceKernelVolatileMemTryLock(%i, %08x, %08x) - success", type, paddr, psize);
break;
+6 -3
View File
@@ -64,15 +64,18 @@ static u32 sceVaudioChReserve(int sampleCount, int freq, int format) {
// reserved before handing over - it does not undo that when the reserve fails. So a caller
// that got 0x80268002 because Output2 held the channel is told 0x80000021 next time round,
// until a release clears it.
// From here on the call waits about 250us, whether it succeeds or not; on hardware worse
// threads get to run meanwhile (pspautotests audio/sceaudio/reserve shows the reschedule).
vaudioReserved = true;
const int reserveUs = 250;
if (freq != 0 && !SRCFrequencyAllowed(freq)) {
ERROR_LOG(Log::sceAudio, "sceVaudioChReserve(%i, %i, %i) - invalid frequency", sampleCount, freq, format);
return SCE_ERROR_AUDIO_INVALID_FREQUENCY;
return hleDelayResult(SCE_ERROR_AUDIO_INVALID_FREQUENCY, "vaudio reserve", reserveUs);
}
// We still have to check the channel also, which gives a different error.
if (g_audioSRC.reserved) {
ERROR_LOG(Log::sceAudio, "sceVaudioChReserve(%i, %i, %i) - channel already reserved", sampleCount, freq, format);
return SCE_ERROR_AUDIO_CHANNEL_ALREADY_RESERVED;
return hleDelayResult(SCE_ERROR_AUDIO_CHANNEL_ALREADY_RESERVED, "vaudio reserve", reserveUs);
}
DEBUG_LOG(Log::sceAudio, "sceVaudioChReserve(%i, %i, %i)", sampleCount, freq, format);
g_audioSRC.clear();
@@ -80,7 +83,7 @@ static u32 sceVaudioChReserve(int sampleCount, int freq, int format) {
g_audioSRC.sampleCount = sampleCount;
g_audioSRC.format = format == 2 ? PSP_AUDIO_FORMAT_STEREO : PSP_AUDIO_FORMAT_MONO;
__AudioSetSRCFrequency(freq);
return 0;
return hleDelayResult(0, "vaudio reserve", reserveUs);
}
static u32 sceVaudioChRelease() {
+3 -3
View File
@@ -431,9 +431,9 @@ void ArmJit::Comp_mxc1(MIPSOpcode op)
gpr.MapDirtyIn(MIPS_REG_FPCOND, rt);
}
// Update MIPS state
// TODO: Technically, should mask by 0x0181FFFF. Maybe just put all of FCR31 in the reg?
STR(gpr.R(rt), CTXREG, offsetof(MIPSState, fcr31));
// Update MIPS state. Only these bits can be written (pspautotests cpu/fpu/fcr).
ANDI2R(SCRATCHREG1, gpr.R(rt), 0x0181FFFF, SCRATCHREG2);
STR(SCRATCHREG1, CTXREG, offsetof(MIPSState, fcr31));
if (!wasImm) {
#if PPSSPP_ARCH(ARMV7)
UBFX(gpr.R(MIPS_REG_FPCOND), gpr.R(rt), 23, 1);
+3 -3
View File
@@ -396,9 +396,9 @@ void Arm64Jit::Comp_mxc1(MIPSOpcode op)
gpr.MapDirtyIn(MIPS_REG_FPCOND, rt);
}
// Update MIPS state
// TODO: Technically, should mask by 0x0181FFFF. Maybe just put all of FCR31 in the reg?
STR(INDEX_UNSIGNED, gpr.R(rt), CTXREG, offsetof(MIPSState, fcr31));
// Update MIPS state. Only these bits can be written (pspautotests cpu/fpu/fcr).
ANDI2R(SCRATCH1, gpr.R(rt), 0x0181FFFF, SCRATCH2);
STR(INDEX_UNSIGNED, SCRATCH1, CTXREG, offsetof(MIPSState, fcr31));
if (!wasImm) {
UBFX(gpr.R(MIPS_REG_FPCOND), gpr.R(rt), 23, 1);
// TODO: We do have the fcr31 value in a register here, could use that in UpdateRoundingMode to avoid reloading it.