Fix four WebSocket debugger bugs found while stress testing

memory.readString could kill the connection: it copied raw emulated memory
straight into a JSON string, so any address not holding valid UTF-8 produced an
invalid WebSocket text frame.

hle.data.remove wiped the name of a function sharing the address. Labels are
shared between data and function symbols, so removing the data label left the
function showing up in hle.func.list with an empty name.

hle.data.add silently did nothing outside a loaded module. GetModuleIndex()
returns -1 for e.g. a heap or stack address, and symbols under that index never
reach the active maps - so the add reported success while the symbol was
invisible to list, and rename/remove then failed with "No data symbol found".
Falls back to module index 0 ("no module, absolute address"), which is the right
answer for a label the user put somewhere after a memory.search.

hle.thread.list reported the thread's stack base address in a field called
initialStackSize. Renamed to initialStack, matching the SceKernelThreadInfo
field it comes from.

Co-Authored-By: Claude Opus 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_01GZq8ZtJmFY7bkX5FVkr3P9
This commit is contained in:
Henrik RydgårdandClaude Opus 5 committed 2026-08-17 00:25:39 +02:00
1 parent 0e15445b54
commit 6a05ef290d
5 files changed
+146 -7

No files matched your search

+75
View File
@@ -242,6 +242,81 @@ std::string SanitizeUTF8(std::string_view utf8string) {
return s;
}
// Length of the well-formed UTF-8 sequence starting at s, or 0 if it isn't one. Strict per
// RFC 3629: rejects overlong encodings, surrogates (U+D800..U+DFFF) and anything above U+10FFFF,
// all of which some decoders accept but which aren't legal UTF-8 and get rejected downstream.
static int ValidUTF8SequenceLength(const unsigned char *s, size_t remaining) {
const unsigned char c = s[0];
int length;
unsigned char min2, max2; // Allowed range of the *second* byte, which is the constrained one.
if (c < 0x80)
return 1;
else if (c >= 0xC2 && c <= 0xDF)
length = 2, min2 = 0x80, max2 = 0xBF;
else if (c == 0xE0)
length = 3, min2 = 0xA0, max2 = 0xBF; // Would be overlong below A0.
else if (c >= 0xE1 && c <= 0xEC)
length = 3, min2 = 0x80, max2 = 0xBF;
else if (c == 0xED)
length = 3, min2 = 0x80, max2 = 0x9F; // Above 9F is a surrogate.
else if (c >= 0xEE && c <= 0xEF)
length = 3, min2 = 0x80, max2 = 0xBF;
else if (c == 0xF0)
length = 4, min2 = 0x90, max2 = 0xBF; // Would be overlong below 90.
else if (c >= 0xF1 && c <= 0xF3)
length = 4, min2 = 0x80, max2 = 0xBF;
else if (c == 0xF4)
length = 4, min2 = 0x80, max2 = 0x8F; // Above 8F is past U+10FFFF.
else
return 0; // Continuation byte with nothing to continue, or C0/C1/F5..FF.
if (remaining < (size_t)length)
return 0;
if (s[1] < min2 || s[1] > max2)
return 0;
for (int i = 2; i < length; ++i) {
if ((s[i] & 0xC0) != 0x80)
return 0;
}
return length;
}
std::string ReplaceInvalidUTF8(std::string_view utf8string) {
static const char REPLACEMENT[] = "\xEF\xBF\xBD"; // U+FFFD
const unsigned char *bytes = (const unsigned char *)utf8string.data();
const size_t size = utf8string.size();
// Overwhelmingly the common case - avoid the copy entirely when there's nothing to fix.
size_t pos = 0;
while (pos < size) {
int length = ValidUTF8SequenceLength(bytes + pos, size - pos);
if (length == 0)
break;
pos += length;
}
if (pos == size)
return std::string(utf8string);
std::string s;
s.reserve(size);
s.append(utf8string.substr(0, pos));
while (pos < size) {
int length = ValidUTF8SequenceLength(bytes + pos, size - pos);
if (length == 0) {
// Not the start of anything legal - swallow exactly one byte so we resynchronize on
// the next one rather than skipping over a valid sequence that follows.
s.append(REPLACEMENT, sizeof(REPLACEMENT) - 1);
pos++;
} else {
s.append(utf8string.substr(pos, length));
pos += length;
}
}
return s;
}
static size_t ConvertUTF8ToUCS2Internal(char16_t *dest, size_t destSize, std::string_view source) {
const char16_t *const orig = dest;
const char16_t *const destEnd = dest + destSize;
+6
View File
@@ -96,6 +96,12 @@ bool UTF8StringHasNonASCII(std::string_view utf8string);
// Removes overlong encodings and similar.
std::string SanitizeUTF8(std::string_view utf8string);
// Returns a copy with every byte sequence that isn't well-formed UTF-8 replaced by U+FFFD.
// Where SanitizeUTF8() stops at the first bad byte, this one keeps going, so it's the right choice
// for text of unknown provenance (raw emulated memory, foreign files) that has to end up somewhere
// that *requires* valid UTF-8 - a JSON string, or a WebSocket text frame.
std::string ReplaceInvalidUTF8(std::string_view utf8string);
std::string CodepointToUTF8(uint32_t codePoint);
+21 -5
View File
@@ -92,7 +92,7 @@ static bool DataTypeFromString(const std::string &s, DataType *out) {
// - statuses: array of string status names, e.g. 'running'. Typically only one set.
// - pc: unsigned integer address of next instruction on thread.
// - entry: unsigned integer address thread execution started at.
// - initialStackSize: unsigned integer, size of initial stack.
// - initialStack: unsigned integer, address of the base of the thread's stack.
// - currentStackSize: unsigned integer, size of stack (e.g. if resized.)
// - priority: numeric priority level, lower values are better priority.
// - waitType: numeric wait type, if the thread is waiting, or 0 if not waiting.
@@ -127,7 +127,9 @@ void WebSocketHLEThreadList(DebuggerRequest &req) {
json.pop();
json.writeUint("pc", th.curPC);
json.writeUint("entry", th.entrypoint);
json.writeUint("initialStackSize", th.initialStack);
// Named after the SceKernelThreadInfo field it comes from - it's the stack base
// address, not a size, despite what this used to be called.
json.writeUint("initialStack", th.initialStack);
json.writeUint("currentStackSize", th.stackSize);
json.writeInt("priority", th.priority);
json.writeInt("waitType", (int)th.waitType);
@@ -800,8 +802,17 @@ void WebSocketHLEDataAdd(DebuggerRequest &req) {
// Route the actual symbol manipulation to the CPU thread instead of poking at it directly
// from this WebSocket handler thread - see Core_RunOnCPUThread() in Core.h.
Core_RunOnCPUThread([&] {
g_symbolMap->AddData(addr, size, type);
g_symbolMap->AddLabel(name.c_str(), addr);
// GetModuleIndex() returns -1 for an address that isn't inside any loaded module, and
// symbols added under that index never make it into the "active" maps - so they'd silently
// vanish (invisible to hle.data.list, and not findable by remove/rename). Module index 0
// means "no module, absolute address", which is exactly what we want for a label the user
// put on the heap, the stack, or scratchpad after e.g. a memory.search.
int moduleIndex = g_symbolMap->GetModuleIndex(addr);
if (moduleIndex < 0)
moduleIndex = 0;
g_symbolMap->AddData(addr, size, type, moduleIndex);
g_symbolMap->AddLabel(name.c_str(), addr, moduleIndex);
g_symbolMap->SortSymbols();
// Clear cache so the disassembly view picks up the new annotation.
@@ -843,7 +854,12 @@ void WebSocketHLEDataRemove(DebuggerRequest &req) {
}
u32 dataSize = g_symbolMap->GetDataSize(dataBegin);
g_symbolMap->RemoveData(dataBegin, true);
// Labels are shared between data and function symbols, so dropping the label along with the
// data would also wipe the name of a function starting at the same address (leaving it
// showing up in hle.func.list with an empty name). Only take the label with us if no
// function is using it too.
const bool functionOwnsLabel = g_symbolMap->GetFunctionStart(dataBegin) == dataBegin;
g_symbolMap->RemoveData(dataBegin, !functionOwnsLabel);
g_symbolMap->SortSymbols();
g_disassemblyManager.clear();
+8 -2
View File
@@ -19,6 +19,7 @@
#include <cstring>
#include <mutex>
#include "Common/Data/Encoding/Base64.h"
#include "Common/Data/Encoding/Utf8.h"
#include "Common/StringUtils.h"
#include "Core/Core.h"
#include "Core/Debugger/WebSocket/MemorySubscriber.h"
@@ -249,7 +250,9 @@ void WebSocketMemoryRead(DebuggerRequest &req) {
// - type: optional, 'utf-8' (default) or 'base64'.
//
// Response (same event name) for 'utf8':
// - value: string value read.
// - value: string value read. Since this reads arbitrary emulated memory, which is under no
// obligation to hold text at all, any byte sequence that isn't valid UTF-8 is replaced with
// U+FFFD. Use 'base64' if you need the bytes exactly as they are.
//
// Response (same event name) for 'base64':
// - base64: base64 encode of binary data, not including NUL.
@@ -287,7 +290,10 @@ void WebSocketMemoryReadString(DebuggerRequest &req) {
JsonWriter &json = req.Respond();
if (type == "utf-8") {
json.writeString("value", raw);
// Must not go out raw: WebSocket text frames are required to be valid UTF-8, so a stray
// byte from some non-text address would make a conforming client (browsers included) drop
// the connection - taking down the whole debugger session over one bad read.
json.writeString("value", ReplaceInvalidUTF8(raw));
} else if (type == "base64") {
json.writeString("base64", Base64Encode((const uint8_t *)raw.data(), raw.size()));
}
+36
View File
@@ -466,6 +466,42 @@ bool TestUtf8() {
EXPECT_TRUE(output == "abc");
}
// ReplaceInvalidUTF8 must always return well-formed UTF-8, keeping the good parts. This one
// guards a WebSocket text frame (memory.readString reads arbitrary emulated memory), where a
// single bad byte getting through disconnects conforming clients.
{
const std::string replacement = "\xEF\xBF\xBD"; // U+FFFD
// Valid input is returned untouched, including 1/2/3/4-byte sequences.
const std::string allValid = "abc \xC3\xA9 \xE2\x82\xAC \xF0\x9F\x8E\xAE";
EXPECT_TRUE(ReplaceInvalidUTF8(allValid) == allValid);
EXPECT_TRUE(ReplaceInvalidUTF8("") == "");
// Unlike SanitizeUTF8, it keeps going past the bad byte instead of truncating there.
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("ab\xFF" "cd")) == "ab" + replacement + "cd");
// One replacement per bad byte, and resynchronization on the next valid sequence.
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\x80\x80")) == replacement + replacement);
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\xC3")) == replacement);
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\xC3?")) == replacement + "?");
// Sequences that lenient decoders accept but that aren't legal UTF-8: overlong encodings,
// surrogates, and anything past U+10FFFF.
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\xC0\xAF")) == replacement + replacement);
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\xE0\x80\xAF")) == replacement + replacement + replacement);
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\xED\xA0\x80")) == replacement + replacement + replacement);
EXPECT_TRUE(ReplaceInvalidUTF8(std::string("\xF4\x90\x80\x80")) == replacement + replacement + replacement + replacement);
// Whatever the input, the output must itself survive a re-run unchanged - i.e. be valid.
for (int b = 0; b < 256; ++b) {
std::string input = "a";
input += (char)b;
input += "b";
const std::string once = ReplaceInvalidUTF8(input);
EXPECT_TRUE(ReplaceInvalidUTF8(once) == once);
}
}
return true;
}