From 8baabbaf49d73fb6327ba1aeb03e85600c6ab041 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 18:28:58 +0200 Subject: [PATCH 01/68] Let an explicit function map override the JAL-target scan On an ELF with no symbols and no DWARF, parse() carves functions from JAL targets. Those carvings end at the next JAL target or, for the last one in a region, at the end of the code section, so on a single-PROGBITS executable they can run straight through interleaved rodata. loadGhidraFunctionMap() appended its rows to the same vector and then purged auto-named entries only where no map row shared the start address. Since both the carvings ("sub_") and the names Ghidra exports by default ("FUN_") count as auto-generated, a carving that shared a start with a map row survived the purge and then won the "larger end" tie-break, so the imprecise bounds replaced the ones the map had just supplied. Collect the map rows into a local vector, drop every auto-named carving once the map has parsed, and append the rows afterwards. Entries named from symbols or DWARF are unaffected. On a 3 MB Metrowerks-built PS2 executable with an 11,491-row map, 5,613 functions (48.8%) had been emitted with inflated bounds; the worst grew from 368 bytes to 0x51 KB and produced 22 MB of C++ decoding string data as instructions. Output for that function is now 19 KB and total output drops from 235 MB to 180 MB. --- ps2xRecomp/src/lib/elf_parser.cpp | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/ps2xRecomp/src/lib/elf_parser.cpp b/ps2xRecomp/src/lib/elf_parser.cpp index 91e43ee32..865dd6aad 100644 --- a/ps2xRecomp/src/lib/elf_parser.cpp +++ b/ps2xRecomp/src/lib/elf_parser.cpp @@ -986,6 +986,7 @@ namespace ps2recomp int skippedNonExecutable = 0; int skippedInvalidRange = 0; std::unordered_set mapStarts; + std::vector mapFunctions; while (std::getline(file, line)) { if (line.empty()) @@ -1029,7 +1030,7 @@ namespace ps2recomp func.isStub = false; func.isSkipped = false; - m_extraFunctions.push_back(std::move(func)); + mapFunctions.push_back(std::move(func)); mapStarts.insert(start); count++; } @@ -1062,14 +1063,27 @@ namespace ps2recomp } } + // An explicit function map is authoritative over the internal JAL-target + // scan. Those carvings end at the next JAL target or, for the last one in + // a region, at the end of the code section - which on single-PROGBITS + // executables runs straight through interleaved rodata. Because both the + // carvings ("sub_") and typical map names ("FUN_") count as + // auto-generated, a carving sharing a start with a map row used to + // survive this purge and then win the "larger end" tie-break below, + // replacing precise bounds with runaway ones. Drop every auto-named + // carving instead, then append the map rows. m_extraFunctions.erase( std::remove_if(m_extraFunctions.begin(), m_extraFunctions.end(), [&](const Function &func) { - return IsAutoGeneratedName(func.name) && !mapStarts.contains(func.start); + return IsAutoGeneratedName(func.name); }), m_extraFunctions.end()); + m_extraFunctions.insert(m_extraFunctions.end(), + std::make_move_iterator(mapFunctions.begin()), + std::make_move_iterator(mapFunctions.end())); + std::sort(m_extraFunctions.begin(), m_extraFunctions.end(), [](const Function &a, const Function &b) { From 5c906ddf4be51e4082d36fbf8779643bc5273968 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 18:28:59 +0200 Subject: [PATCH 02/68] Warn instead of aborting when no function covers the ELF entry point Not every toolchain points e_entry at an instruction. Metrowerks CodeWarrior for PS2 emits a crt0 data table there -- scratchpad addresses and size words -- with the first real instruction some way past it, so no function covers the entry address and neither entryName nor getFunctionName() resolves. The emitter threw in that case, which aborted the run after every per-function source had already been written but before register_functions.cpp, ps2_recompiled_functions.h and ps2_recompiled_stubs.h were generated, leaving an output directory that looks complete and is not. Skip the entry registration with a warning instead. Synthesizing a name would emit a table reference to a definition that was never generated and fail at link time, and the start address for such a binary has to come from configuration regardless. --- ps2xRecomp/src/lib/function_table_emitter.cpp | 20 +++++++++++++++++-- 1 file changed, 18 insertions(+), 2 deletions(-) diff --git a/ps2xRecomp/src/lib/function_table_emitter.cpp b/ps2xRecomp/src/lib/function_table_emitter.cpp index 27dc9b74c..84d3850ea 100644 --- a/ps2xRecomp/src/lib/function_table_emitter.cpp +++ b/ps2xRecomp/src/lib/function_table_emitter.cpp @@ -95,9 +95,25 @@ namespace ps2recomp } if (entryTarget.empty()) { - throw std::runtime_error("No entry function name available for registration."); + // Not every toolchain points e_entry at code. Metrowerks CodeWarrior + // for PS2 emits a crt0 data table there, so no function covers the + // address and no name resolves. Registering a synthesized name would + // reference a definition that was never emitted, so skip the entry + // instead - the real start address comes from configuration - and + // keep emitting the rest of the table rather than aborting after the + // per-function sources were already written. + if (cg.m_reporter) + { + std::ostringstream oss; + oss << "No function covers the ELF entry point; skipping its table registration. " + << "Set the entry explicitly if the runtime should start here."; + cg.m_reporter->warning("function-table", oss.str()); + } + } + else + { + addEntry(cg.m_bootstrapInfo.entry, entryTarget); } - addEntry(cg.m_bootstrapInfo.entry, entryTarget); } for (const auto &[address, name] : normalFunctions) From 20848417af22664f6b391d31c2f744d001032588 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 18:29:01 +0200 Subject: [PATCH 03/68] Compare all 64 bits in the relational branches BEQ and BNE already compare the full GPR, but BLEZ, BGTZ, BLTZ and BGEZ (and their likely/and-link variants) were emitted against the low word only. The R5900 compares the whole 64-bit register, so any value whose upper half is significant takes the wrong branch. Compilers reach these opcodes through the dsll32/dsra32 sign-extension idiom, which leaves a canonical value and hides the bug; code that keeps a genuine 64-bit quantity in the register does not. --- ps2xRecomp/src/lib/control_flow_emitter.cpp | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/ps2xRecomp/src/lib/control_flow_emitter.cpp b/ps2xRecomp/src/lib/control_flow_emitter.cpp index 4bf8dc550..ce3020675 100644 --- a/ps2xRecomp/src/lib/control_flow_emitter.cpp +++ b/ps2xRecomp/src/lib/control_flow_emitter.cpp @@ -404,12 +404,16 @@ namespace ps2recomp case OPCODE_BNE: case OPCODE_BNEL: return fmt::format("GPR_U64(ctx, {}) != GPR_U64(ctx, {})", rsReg, rtReg); + // The R5900 compares the full 64-bit GPR for the relational branches, as it + // already does for BEQ/BNE above. Comparing only the low word takes the wrong + // branch whenever the upper half is significant, which happens with the + // dsll32/dsra32 sign-extension idiom compilers emit ahead of these opcodes. case OPCODE_BLEZ: case OPCODE_BLEZL: - return fmt::format("GPR_S32(ctx, {}) <= 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) <= 0", rsReg); case OPCODE_BGTZ: case OPCODE_BGTZL: - return fmt::format("GPR_S32(ctx, {}) > 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) > 0", rsReg); case OPCODE_REGIMM: switch (m_branchInst.rt) { @@ -417,12 +421,12 @@ namespace ps2recomp case REGIMM_BLTZL: case REGIMM_BLTZAL: case REGIMM_BLTZALL: - return fmt::format("GPR_S32(ctx, {}) < 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) < 0", rsReg); case REGIMM_BGEZ: case REGIMM_BGEZL: case REGIMM_BGEZAL: case REGIMM_BGEZALL: - return fmt::format("GPR_S32(ctx, {}) >= 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) >= 0", rsReg); default: return "false"; } From 0627c0d7427c8736ccd8391cd53f336ed7ab3556 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 18:29:02 +0200 Subject: [PATCH 04/68] Resolve sibling component paths relative to the current directory ps2xRecomp and ps2xAnalyzer reached ps2xRuntime through CMAKE_SOURCE_DIR, which is the top-level source directory of whatever build is running. That holds only when this repository is itself the top level; adding it to another project with add_subdirectory() made both components look for ps2xRuntime/cmake/ReleaseMode.cmake under the consuming project and fail at configure time. Use CMAKE_CURRENT_SOURCE_DIR-relative paths, as ps2xRecomp already does for its ps2xRuntime include directory and ps2xRuntime does for its own cmake include. Note that each component declares its own project(), so PROJECT_SOURCE_DIR is not an alternative here. --- ps2xAnalyzer/CMakeLists.txt | 6 +++--- ps2xRecomp/CMakeLists.txt | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/ps2xAnalyzer/CMakeLists.txt b/ps2xAnalyzer/CMakeLists.txt index 5933bca14..97d4cee26 100644 --- a/ps2xAnalyzer/CMakeLists.txt +++ b/ps2xAnalyzer/CMakeLists.txt @@ -27,8 +27,8 @@ add_library(ps2_analyzer_lib STATIC ${PS2ANALYZER_LIB_SOURCES}) target_include_directories(ps2_analyzer_lib PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include - ${CMAKE_SOURCE_DIR}/ps2xRecomp/include - ${CMAKE_SOURCE_DIR}/ps2xRuntime/include + ${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRecomp/include + ${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRuntime/include ) target_link_libraries(ps2_analyzer_lib PUBLIC @@ -50,7 +50,7 @@ install(TARGETS ps2_analyzer ps2_analyzer_lib ARCHIVE DESTINATION lib ) -include("${CMAKE_SOURCE_DIR}/ps2xRuntime/cmake/ReleaseMode.cmake") +include("${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRuntime/cmake/ReleaseMode.cmake") if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") EnableFastReleaseMode(ps2_analyzer_lib) diff --git a/ps2xRecomp/CMakeLists.txt b/ps2xRecomp/CMakeLists.txt index 857102a22..5ed0d0799 100644 --- a/ps2xRecomp/CMakeLists.txt +++ b/ps2xRecomp/CMakeLists.txt @@ -130,7 +130,7 @@ install(DIRECTORY include/ DESTINATION include ) -include("${CMAKE_SOURCE_DIR}/ps2xRuntime/cmake/ReleaseMode.cmake") +include("${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRuntime/cmake/ReleaseMode.cmake") if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") EnableFastReleaseMode(ps2_recomp_lib) From 75d5085a79fbebf45ff4e4070a1c15d60edc236e Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 18:29:04 +0200 Subject: [PATCH 05/68] Don't promote a whole function because of an indirect call When a computed jump cannot be resolved to a jump table, every instruction in the function becomes an entry point, because the jump could land on any of them. That fallback was also applied to JALR, which is not a jump but a call: it transfers control to another function and returns to the instruction after the delay slot. That return address is already queued as a resume target a few lines above, so nothing else in the function needs to be reachable from outside. Indirect calls are ordinary code -- function pointers, virtual dispatch, callbacks -- so the fallback fired constantly. On a 3 MB PS2 executable, 2,210 of the 2,422 unresolved sites were JALR, and 1,078 of the 1,282 affected functions contained no unresolved jump at all. Restrict the fallback to JR. Promoted entries drop from 189,876 to 1,688, registered table entries from 156,783 to 75,386, the generated registration file from 13 MB to 6 MB, and total output from 180 MB to 163 MB. Every indirect call site in real code keeps its return-address resume entry (the only sites that lose one are bogus functions carved out of rodata, where the address is outside the function anyway). --- ps2xRecomp/src/lib/control_flow_analyzer.cpp | 9 ++++++++- ps2xTest/src/code_generator_tests.cpp | 19 +++++++++++++++---- 2 files changed, 23 insertions(+), 5 deletions(-) diff --git a/ps2xRecomp/src/lib/control_flow_analyzer.cpp b/ps2xRecomp/src/lib/control_flow_analyzer.cpp index a099f5cf9..2b27ffc9f 100644 --- a/ps2xRecomp/src/lib/control_flow_analyzer.cpp +++ b/ps2xRecomp/src/lib/control_flow_analyzer.cpp @@ -385,7 +385,14 @@ namespace ps2recomp } } } - if (!foundTable) + // Only an unresolved computed *jump* can land on an arbitrary + // instruction of this function and therefore force every address to + // become an entry point. JALR is a call: it transfers control to + // another function and comes back to the instruction after the delay + // slot, which is already queued as a resume target above. Treating a + // call like a jump here promotes the whole function for what is + // usually just a function pointer or virtual dispatch. + if (!foundTable && jrInst->function != SPECIAL_JALR) { needsIndirectFallback = true; } diff --git a/ps2xTest/src/code_generator_tests.cpp b/ps2xTest/src/code_generator_tests.cpp index 2dd9c531f..d9128c644 100644 --- a/ps2xTest/src/code_generator_tests.cpp +++ b/ps2xTest/src/code_generator_tests.cpp @@ -529,7 +529,7 @@ void register_code_generator_tests() "unresolved JR should not pretend it has a resolved local jump table"); }); - tc.Run("unresolved JALR marks internal labels as indirect fallback resume entries", [](TestCase &t) { + tc.Run("unresolved JALR resumes after the call without promoting the function", [](TestCase &t) { Function func; func.name = "unresolved_jalr_fallback"; func.start = 0x3200; @@ -548,10 +548,21 @@ void register_code_generator_tests() CodeGenerator gen({}, {}); CodeGenerator::AnalysisResult analysis = gen.collectInternalBranchTargets(func, instructions); - t.IsTrue(analysis.indirectFallbackEntryPoints.contains(0x320Cu), - "unresolved JALR should register internal labels as resumable entries for the owning function"); + // JALR is a call: it returns past the delay slot, so 0x320C is the only + // address in this function that has to be reachable from outside. + t.IsTrue(analysis.resumeEntryPoints.contains(0x320Cu), + "unresolved JALR should mark its return pc as resumable"); t.IsTrue(analysis.entryPoints.contains(0x320Cu), - "unresolved JALR fallback targets should still emit labels in the owner"); + "unresolved JALR resume pc should still emit a label in the owner"); + + // Both sets feed the same owner resume-target list, so the return pc is + // registered either way; what must not happen is the whole-function + // promotion reserved for jumps that could land anywhere. + t.IsFalse(analysis.indirectFallbackEntryPoints.contains(0x3210u), + "an indirect call must not promote unrelated instructions to entry points"); + t.IsFalse(analysis.indirectFallbackEntryPoints.contains(0x3200u), + "an indirect call must not promote the function start to a fallback entry"); + t.IsFalse(analysis.jumpTableTargets.contains(0x3204u), "unresolved JALR should not pretend it has a resolved local jump table"); }); From becb2be5bd0dc3deebeec8454df21ea7d7062b45 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 22:57:02 +0200 Subject: [PATCH 06/68] Let a thread resume at the instruction after a syscall A syscall can hand control back to the scheduler before the instruction after it runs. SetSyscall lets the guest install its own handler for a syscall number; dispatchSyscallOverride then suspends the calling thread and queues that handler as a GuestInvocation. When the invocation finishes, EeScheduler resumes the parent thread at the address the generated code stored just before calling handleSyscall -- the instruction right after the syscall. The analyzer never marked that address as an entry point. It queues resume entries for JAL and JALR only, so no generated function could be re-entered there, EeScheduler's hasFunction() check failed, and the thread was made dormant instead of resumed. The thread simply stops; because the scheduler then drains normally and run() returns, it looks like a clean shutdown rather than a fault, which makes it awkward to recognise. This is reachable during early boot on a real title. Dragon Quest VIII hits it in crt0: the Metrowerks startup code installs a handler for syscall 0x83 and immediately issues it, and execution ends there, roughly ten functions into the binary. Note the offset is +4, not the +8 used for JAL and JALR -- syscall has no delay slot. Co-Authored-By: Claude Opus 5 --- ps2xRecomp/src/lib/control_flow_analyzer.cpp | 17 ++++++++ ps2xTest/src/code_generator_tests.cpp | 45 ++++++++++++++++++++ 2 files changed, 62 insertions(+) diff --git a/ps2xRecomp/src/lib/control_flow_analyzer.cpp b/ps2xRecomp/src/lib/control_flow_analyzer.cpp index 2b27ffc9f..bf105e079 100644 --- a/ps2xRecomp/src/lib/control_flow_analyzer.cpp +++ b/ps2xRecomp/src/lib/control_flow_analyzer.cpp @@ -135,6 +135,23 @@ namespace ps2recomp for (const auto &inst : instructions) { + // A syscall can hand control back to the scheduler before the + // instruction after it runs: SetSyscall lets the guest install its + // own handler, and dispatchSyscallOverride then suspends the + // calling thread and queues that handler as a GuestInvocation. When + // the invocation completes, the scheduler resumes the parent thread + // at the address the generated code stored before calling + // handleSyscall -- i.e. right here. Without an entry point there, + // EeScheduler's hasFunction() check fails and the thread is made + // dormant instead of resumed. + // + // Note the offset is +4, not the +8 used for JAL/JALR: syscall has + // no delay slot. + if (inst.opcode == OPCODE_SPECIAL && inst.function == SPECIAL_SYSCALL) + { + queueResumeEntryTarget(inst.address + 4u); + } + bool isStaticJump = (inst.opcode == OPCODE_J || inst.opcode == OPCODE_JAL); if (inst.isBranch && inst.opcode != OPCODE_J && inst.opcode != OPCODE_JAL) { diff --git a/ps2xTest/src/code_generator_tests.cpp b/ps2xTest/src/code_generator_tests.cpp index d9128c644..c8d496828 100644 --- a/ps2xTest/src/code_generator_tests.cpp +++ b/ps2xTest/src/code_generator_tests.cpp @@ -144,6 +144,17 @@ static Instruction makeJr(uint32_t address, uint8_t rs) return inst; } +static Instruction makeSyscall(uint32_t address) +{ + Instruction inst{}; + inst.address = address; + inst.opcode = OPCODE_SPECIAL; + inst.function = SPECIAL_SYSCALL; + inst.hasDelaySlot = false; + inst.raw = (OPCODE_SPECIAL << 26) | SPECIAL_SYSCALL; + return inst; +} + static void printGeneratedCode(const std::string& name, const std::string& code) { #ifdef PRINT_GENERATED_CODE @@ -567,6 +578,40 @@ void register_code_generator_tests() "unresolved JALR should not pretend it has a resolved local jump table"); }); + tc.Run("syscall marks the following instruction as a resume entry", [](TestCase &t) { + // Shape of a real SDK syscall wrapper: + // addiu $v1, $zero, ; syscall ; jr $ra ; + // SetSyscall lets a guest install its own handler for a syscall number, + // and the runtime then suspends the calling thread to run that handler + // as a separate invocation. The parent thread's saved pc is the address + // after the syscall, so the scheduler needs an entry point there to + // resume it -- otherwise the thread is made dormant and silently dies. + Function func; + func.name = "syscall_wrapper"; + func.start = 0x4000; + func.end = 0x4010; + func.isRecompiled = true; + func.isStub = false; + + std::vector instructions{ + makeAddiu(0x4000, 3, 0, 0x83), + makeSyscall(0x4004), + makeJr(0x4008, 31), + makeNop(0x400C), + }; + + CodeGenerator gen({}, {}); + CodeGenerator::AnalysisResult analysis = gen.collectInternalBranchTargets(func, instructions); + + // +4, not +8: syscall has no delay slot. + t.IsTrue(analysis.resumeEntryPoints.contains(0x4008u), + "syscall should mark the next instruction as resumable"); + t.IsTrue(analysis.entryPoints.contains(0x4008u), + "syscall resume pc should emit a label in the owner"); + t.IsFalse(analysis.resumeEntryPoints.contains(0x400Cu), + "syscall must not claim a delay slot it does not have"); + }); + tc.Run("resume entry targets emit a top-level pc switch in the owner wrapper", [](TestCase &t) { Function func; func.name = "resume_owner"; From 6bf1feff8bd38892f1a464322f455fe016814c5c Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 22:57:21 +0200 Subject: [PATCH 07/68] Return true after queueing a guest syscall override dispatchSyscallOverride returns bool, and its caller uses that to decide whether the built-in handler still needs to run. The success path -- the one that actually queues the guest's handler as an invocation -- fell off the end of the function without returning. That is undefined behaviour, and the practical failure mode is bad: whatever happened to be in the return register decided whether the built-in syscall ran in addition to the guest's override, so a game that overrides a syscall could get the effect applied twice, or not at all, depending on the build. Co-Authored-By: Claude Opus 5 --- ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp index 3530e65bb..babcc0e80 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp @@ -440,6 +440,11 @@ namespace ps2_syscalls parent.r[2] = completed.r[2]; }; scheduler.invokeCurrent(std::move(invocation)); + // The invocation is queued and the caller must not fall through to the + // built-in handler. Falling off the end of a non-void function here was + // undefined behaviour: whatever happened to be in the return register + // decided whether the built-in ran as well as the guest's override. + return true; } static bool tryResolveGuestSyscallMirrorAddr(uint32_t syscallIndex, uint32_t &guestAddr) From 226bae250b285930da75accc7a7ee0045942d807 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 22:57:21 +0200 Subject: [PATCH 08/68] Set the x86 SIMD baseline that the code already requires ps2_runtime.h includes unconditionally, and the recompiler emits SSE4.1-only intrinsics (_mm_blendv_ps and friends) for the COP2 and FPU select idioms. SSE4.1 is therefore a hard requirement of the codebase, not a tuning option. Nothing sets it for GCC or Clang. MSVC does not need it -- its intrinsics are not gated behind a target feature -- and the only place any x86 feature is named is /arch:AVX2 inside EnableFastReleaseMode, which is MSVC-only and applies to Release and RelWithDebInfo only. GCC and Clang default to the plain x86-64 baseline, which is SSE2, so on a stock Linux or macOS toolchain the affected translation units fail with "always_inline function ... requires target feature 'sse4.1'". Adds EnableX86SimdBaseline and applies it to ps2_runtime in every configuration rather than in EnableFastReleaseMode, since a Debug build needs it just as much. PUBLIC, so recompiled game code linking against ps2_runtime inherits it -- that code is where most of the SSE4.1 intrinsics actually are. Guarded on the target processor so ARM builds are unaffected. Co-Authored-By: Claude Opus 5 --- ps2xRuntime/CMakeLists.txt | 5 +++++ ps2xRuntime/cmake/ReleaseMode.cmake | 20 ++++++++++++++++++++ 2 files changed, 25 insertions(+) diff --git a/ps2xRuntime/CMakeLists.txt b/ps2xRuntime/CMakeLists.txt index e4dc1956d..aad5fb80b 100644 --- a/ps2xRuntime/CMakeLists.txt +++ b/ps2xRuntime/CMakeLists.txt @@ -552,6 +552,11 @@ if(PS2X_IS_VITA) endif() endif() +# Every configuration, not just Release: SSE4.1 is required to compile, not an +# optimisation. PUBLIC so recompiled game code linking against ps2_runtime +# inherits it -- that code is where most of the SSE4.1 intrinsics actually are. +EnableX86SimdBaseline(ps2_runtime) + if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") EnableFastReleaseMode(ps2_runtime) EnableFastReleaseMode(ps2EntryRunner) diff --git a/ps2xRuntime/cmake/ReleaseMode.cmake b/ps2xRuntime/cmake/ReleaseMode.cmake index 92ef0d54d..0b97e9016 100644 --- a/ps2xRuntime/cmake/ReleaseMode.cmake +++ b/ps2xRuntime/cmake/ReleaseMode.cmake @@ -2,6 +2,26 @@ include(CheckIPOSupported) check_ipo_supported(RESULT IPO_SUPPORTED OUTPUT IPO_ERROR) +# ps2_runtime.h unconditionally includes and the recompiler emits +# SSE4.1-only intrinsics (_mm_blendv_ps and friends) for the COP2/FPU select +# idioms, so SSE4.1 is a hard requirement of the codebase rather than a tuning +# knob. MSVC enables it implicitly (its intrinsics are not gated by a target +# feature), but GCC and Clang default to the plain x86-64 baseline, which is +# SSE2 -- so on any stock Linux or macOS toolchain those translation units fail +# to compile with "always_inline function ... requires target feature 'sse4.1'". +# +# This must not live in EnableFastReleaseMode: that is only applied for Release +# and RelWithDebInfo, whereas the requirement applies to every configuration. +function(EnableX86SimdBaseline TargetName) + if(MSVC) + return() + endif() + if(NOT CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64|i[3-6]86|x86)$") + return() + endif() + target_compile_options(${TargetName} PUBLIC -msse4.1) +endfunction() + function(EnableFastReleaseMode TargetName) message("> Enabling optimization for: ${TargetName}") if(MSVC) From 50dc8681cedc62852604b29a04134c4c3126aec1 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Mon, 17 Aug 2026 22:57:21 +0200 Subject: [PATCH 09/68] Make the DECI2 text log limit configurable logDeci2Text stops after 256 lines and then silently drops everything else. That is a sensible default against a game that spams kputs, but it makes the runtime unusable as the candidate side of a boot-trace diff: real captures run to many thousands of lines, and a silent truncation partway through looks exactly like the guest stopping. Adds PS2X_DECI2_LOG_LIMIT, with 0 meaning unlimited. Unset or unparseable keeps the existing 256. Co-Authored-By: Claude Opus 5 --- ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp | 36 +++++++++++++++++-- 1 file changed, 34 insertions(+), 2 deletions(-) diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp index 039da2fb0..5792b0552 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp @@ -2,6 +2,8 @@ #include "Common.h" #include "ps2_runtime.h" +#include + namespace { struct Deci2Session @@ -62,11 +64,41 @@ namespace return text; } + // Default cap on DECI2 text lines, so a game that spams kputs cannot drown + // the console. Override with PS2X_DECI2_LOG_LIMIT; 0 means unlimited, which + // is what you want when diffing a full boot trace against a reference + // capture from real hardware or an emulator -- those run to many thousands + // of lines and a silent truncation at 256 looks exactly like the guest + // stopping. + static uint32_t deci2TextLogLimit() + { + static const uint32_t limit = []() -> uint32_t + { + constexpr uint32_t kDefaultMaxDeci2TextLogs = 256u; + const char *env = std::getenv("PS2X_DECI2_LOG_LIMIT"); + if (!env || *env == '\0') + { + return kDefaultMaxDeci2TextLogs; + } + + char *parseEnd = nullptr; + const unsigned long parsed = std::strtoul(env, &parseEnd, 10); + if (parseEnd == env || *parseEnd != '\0' || parsed > 0xFFFFFFFFul) + { + return kDefaultMaxDeci2TextLogs; + } + + return static_cast(parsed); + }(); + + return limit; + } + static void logDeci2Text(const char *prefix, const std::string &text) { - constexpr uint32_t kMaxDeci2TextLogs = 256u; + const uint32_t maxDeci2TextLogs = deci2TextLogLimit(); const uint32_t logIndex = g_deci2LogCount.fetch_add(1u, std::memory_order_relaxed); - if (logIndex >= kMaxDeci2TextLogs) + if (maxDeci2TextLogs != 0u && logIndex >= maxDeci2TextLogs) { return; } From 2ac2ce632082ff8b7683f0380361ddcbc410bdbc Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Tue, 18 Aug 2026 15:33:24 +0200 Subject: [PATCH 10/68] Start the main thread with COP0 Status.IE set Guest code reads COP0 Status to decide whether interrupts are enabled, and two different bits are involved: IE (bit 0) the architectural MIPS interrupt enable, set once by the kernel during boot and normally left set. EIE (bit 16) the EE-specific enable that `ei` and `di` toggle. We never execute the boot ROM, so nothing was setting either one, and R5900Context started with Status at zero. That is not cosmetic. libkernel's StartThread opens with `mfc0 Status; xori 1; andi 1` and bails out with -1 when IE is clear -- its "you must call iStartThread from an interrupt handler" guard. With Status at zero that guard fired every time, so every StartThread failed. Dragon Quest VIII hits this during boot: it creates its CD streaming thread, gets -1, prints "Can't start thread for streaming." and then deadlocks with every thread blocked and none runnable. Nothing in the runtime logs anything, because from its point of view the guest simply asked a question and got an answer. EIE matters for the matching reason: DIntr reports whether it was set so the caller knows whether to pair it with an EIntr. Starting at zero makes DIntr always answer "already disabled", so the re-enable never happens. Two changes, both needed: - R5900Context's constructor now sets Status to EIE | IE rather than 0. BEV is deliberately left clear -- that selects the boot exception vectors, which is the pre-handoff state, not this one. - PS2Runtime's constructor no longer memsets m_cpuContext. R5900Context already zeroes itself before applying its reset values, so the memset only threw those values away. Threads created later were unaffected because EeScheduler::startThread assigns `R5900Context{}`, which is why this presented as "the main thread cannot start threads" rather than something more obviously global. Co-Authored-By: Claude Opus 5 --- ps2xRuntime/include/ps2_runtime.h | 31 +++++++++++++++++++++++++---- ps2xRuntime/src/lib/ps2_runtime.cpp | 10 +++++++++- 2 files changed, 36 insertions(+), 5 deletions(-) diff --git a/ps2xRuntime/include/ps2_runtime.h b/ps2xRuntime/include/ps2_runtime.h index a899408aa..961e78ea8 100644 --- a/ps2xRuntime/include/ps2_runtime.h +++ b/ps2xRuntime/include/ps2_runtime.h @@ -150,10 +150,33 @@ struct alignas(16) R5900Context // Reset COP0 registers cop0_random = 47; // Start at maximum value - // cop0_status = 0x400000; // BEV set, ERL clear, kernel mode - // 0x00400000 = BEV (Boot Exception Vectors). - // 0x00000000 = Normal mode (after BIOS handoff). - cop0_status = 0x00000000; + // Status as the EE kernel leaves it when it hands control to the game, + // which is the state recompiled code starts in -- we never execute the + // boot ROM that would otherwise set this up. + // + // Both interrupt-enable bits matter, and they are not the same bit: + // + // IE (bit 0) the architectural MIPS interrupt enable. The kernel + // sets it once during boot and it normally stays set. + // EIE (bit 16) the EE-specific enable that the `ei` and `di` + // instructions toggle. + // + // Interrupts are only really on when both are set, and guest code reads + // them separately. libkernel's StartThread, for instance, opens with + // `mfc0 Status; xori 1; andi 1` and refuses to run when IE is clear -- + // that is its "you must call iStartThread from an interrupt handler" + // guard. Leaving Status at zero made that guard fire forever, so every + // StartThread returned -1 and any game that creates a thread stalled + // with no diagnostic. + // + // EIE matters for the matching reason: DIntr reports whether it was set + // so the caller knows whether to pair it with an EIntr. Starting at zero + // makes DIntr always answer "already disabled" and the re-enable never + // happens. + // + // BEV (0x00400000) is deliberately not set: that selects the boot + // exception vectors, which is the pre-handoff state, not this one. + cop0_status = 0x00010001; // EIE | IE cop0_prid = 0x00002e20; // CPU ID for R5900 in_delay_slot = false; diff --git a/ps2xRuntime/src/lib/ps2_runtime.cpp b/ps2xRuntime/src/lib/ps2_runtime.cpp index ddacb0c7d..e5d23d9b5 100644 --- a/ps2xRuntime/src/lib/ps2_runtime.cpp +++ b/ps2xRuntime/src/lib/ps2_runtime.cpp @@ -491,7 +491,15 @@ PS2Runtime::PS2Runtime() } #endif - std::memset(&m_cpuContext, 0, sizeof(m_cpuContext)); + // Assign a default-constructed context rather than memset-ing this one. + // R5900Context's constructor already zeroes itself and then applies the + // architectural reset values on top -- COP0 Status, PRId, Random. A raw + // memset here silently threw those away, leaving Status at 0 for the main + // thread while every thread created later, which goes through + // `target->context = R5900Context{}` in EeScheduler::startThread, got the + // correct values. Guest code reads Status.IE to decide whether interrupts + // are enabled, so the main thread believed they were permanently off. + m_cpuContext = R5900Context{}; // R0 is always zero in MIPS m_cpuContext.r[0] = _mm_set1_epi32(0); From 2648de05c5a44ecb363f11e8fa673d38e4f4e410 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Tue, 18 Aug 2026 22:26:32 +0200 Subject: [PATCH 11/68] Drop the unreachable return in dispatchSyscallOverride ran-j pointed out on ran-j/PS2Recomp#211 that invokeCurrent is [[noreturn]] -- it throws EeDispatcherTransfer to unwind back to the scheduler -- so control never reaches the end of the function and the return I added was dead code. He closed the PR; this drops the change so the branch matches upstream again. --- ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp | 5 ----- 1 file changed, 5 deletions(-) diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp index babcc0e80..3530e65bb 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp @@ -440,11 +440,6 @@ namespace ps2_syscalls parent.r[2] = completed.r[2]; }; scheduler.invokeCurrent(std::move(invocation)); - // The invocation is queued and the caller must not fall through to the - // built-in handler. Falling off the end of a non-void function here was - // undefined behaviour: whatever happened to be in the return register - // decided whether the built-in ran as well as the guest's override. - return true; } static bool tryResolveGuestSyscallMirrorAddr(uint32_t syscallIndex, uint32_t &guestAddr) From 21671a2818808ea52bb9df008f6972f13aa49fad Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Wed, 19 Aug 2026 19:37:41 +0200 Subject: [PATCH 12/68] Decode a trailing branch's delay slot past the mapped function end decodeFunction stops at function.end, so when the last instruction inside the range is a branch its delay slot falls outside and the emitter substitutes a NOP (makeSyntheticDelaySlot). A real instruction disappears with no diagnostic. For a jr $ra epilogue that instruction is the addiu $sp, $sp, N, so every call to the function leaks stack. In DQ8 that was __umoddi3, called from _strtol_r while parsing a text asset: after a few calls _strtol_r read its own saved $ra from the wrong offset, got 0, and walked the main thread off the end of crt0. The symptom looked like a scheduler hang. 286 functions in DQ8's map end on a branch, 165 of them where the next mapped function starts more than one word later. --- ps2xRecomp/src/lib/ps2_recompiler.cpp | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/ps2xRecomp/src/lib/ps2_recompiler.cpp b/ps2xRecomp/src/lib/ps2_recompiler.cpp index ab779a4f5..54fdfcf2c 100644 --- a/ps2xRecomp/src/lib/ps2_recompiler.cpp +++ b/ps2xRecomp/src/lib/ps2_recompiler.cpp @@ -1920,6 +1920,9 @@ namespace ps2recomp uint32_t start = function.start; uint32_t end = function.end; + // A branch at the mapped end still owns the next word as its delay + // slot; without this the emitter substitutes a NOP and loses it. + bool delaySlotExtended = false; for (uint32_t address = start; address < end; address += 4) { @@ -1977,6 +1980,12 @@ namespace ps2recomp } instructions.push_back(inst); + + if (!delaySlotExtended && inst.hasDelaySlot && address + 4u == end) + { + delaySlotExtended = true; + end += 4u; + } } catch (const std::exception &e) { From 91ac659402c55aa2fe73db1129e0c6a67c87dcc1 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Wed, 19 Aug 2026 19:37:55 +0200 Subject: [PATCH 13/68] Run interrupt handlers on an invocation stack, not a recorded $sp dispatchIrq seeded the handler invocation with handler.sp, which is whatever $sp happened to be when the guest called AddIntcHandler. By the time the handler fires that thread has long since moved on and is using those addresses for something else, so the handler builds its frame on top of live data. In DQ8 the VBLANK_END handler landed inside the main thread's stack and zeroed a saved $ra, and the thread returned to 0. The dispatcher already does the right thing for an invocation whose $sp is 0 -- it hands out an invocation stack -- so passing a recorded $sp was bypassing it. Same fix for the GS VSync callback and the Alarm event, which shared the pattern. That alone would have moved the corruption rather than removed it: reserveAsyncCallbackStack carves invocation stacks downwards from the top of RAM, and the EE kernel puts a game's initial stack there too (DQ8's main thread runs at 0x01F40000..0x02000000). reserveGuestStackFromAsyncPool now pulls the pool below any guest stack that reaches into it, called wherever a thread's stack is recorded. --- ps2xRuntime/include/runtime/ee_scheduler.h | 1 + ps2xRuntime/src/lib/Kernel/EeScheduler.cpp | 28 +++++++++++++++++++--- 2 files changed, 26 insertions(+), 3 deletions(-) diff --git a/ps2xRuntime/include/runtime/ee_scheduler.h b/ps2xRuntime/include/runtime/ee_scheduler.h index fae8b8540..ad937be85 100644 --- a/ps2xRuntime/include/runtime/ee_scheduler.h +++ b/ps2xRuntime/include/runtime/ee_scheduler.h @@ -367,6 +367,7 @@ class EeScheduler void assertExecutor() const; [[nodiscard]] int allocateThreadId(); GuestThread &acquireInvocationThread(); + void reserveGuestStackFromAsyncPool(uint32_t guestStackBase); void enqueueReady(GuestThread &thread, bool front = false); void removeReady(GuestThread &thread); [[nodiscard]] GuestThread *selectReady(); diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index 3a6ec7d94..1a6b3d3e1 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -414,9 +414,26 @@ void EeScheduler::setupCurrentThread(uint32_t stack, uint32_t stackSize, uint32_ target->stack = stack; target->stackSize = stackSize; target->gp = gp; + reserveGuestStackFromAsyncPool(stack); publishSnapshot(); } +// Invocation stacks are carved from the top of RAM, where the EE kernel also +// puts a game's initial stack. Keep the pool below any guest stack. +void EeScheduler::reserveGuestStackFromAsyncPool(uint32_t guestStackBase) +{ + if (guestStackBase == 0u) + { + return; + } + std::lock_guard lock(m_runtime.m_asyncCallbackStackMutex); + if (guestStackBase > m_runtime.m_asyncCallbackStackFloor && + guestStackBase < m_runtime.m_asyncCallbackStackTop) + { + m_runtime.m_asyncCallbackStackTop = guestStackBase; + } +} + int EeScheduler::createThread(const EeThreadCreateParams ¶ms) { assertExecutor(); @@ -443,6 +460,7 @@ int EeScheduler::createThread(const EeThreadCreateParams ¶ms) thread.currentPriority = params.priority; thread.status = EeThreadStatus::Dormant; m_threads.emplace(id, std::move(thread)); + reserveGuestStackFromAsyncPool(params.stack); publishSnapshot(); return id; } @@ -1278,7 +1296,9 @@ void EeScheduler::dispatchIrq(bool dmac, uint32_t cause) SET_GPR_U32(&invocation.context, 4, cause); SET_GPR_U32(&invocation.context, 5, handler.argument); SET_GPR_U32(&invocation.context, 28, handler.gp); - SET_GPR_U32(&invocation.context, 29, handler.sp); + // Not handler.sp: that thread has moved on. $sp = 0 makes the + // dispatcher hand out an invocation stack, as the EE does. + SET_GPR_U32(&invocation.context, 29, 0u); SET_GPR_U32(&invocation.context, 31, 0u); queueInvocation(std::move(invocation)); } @@ -1888,7 +1908,8 @@ void EeScheduler::processEvent(const EeEvent &event) invocation.context.pc = m_gsVSyncCallback; SET_GPR_U32(&invocation.context, 4, static_cast(m_vsyncTick)); SET_GPR_U32(&invocation.context, 28, m_gsVSyncCallbackGp); - SET_GPR_U32(&invocation.context, 29, m_gsVSyncCallbackSp); + // See dispatchIrq. + SET_GPR_U32(&invocation.context, 29, 0u); SET_GPR_U32(&invocation.context, 31, 0u); queueInvocation(std::move(invocation)); } @@ -1918,7 +1939,8 @@ void EeScheduler::processEvent(const EeEvent &event) SET_GPR_U32(&invocation.context, 5, static_cast(alarm.ticks)); SET_GPR_U32(&invocation.context, 6, alarm.argument); SET_GPR_U32(&invocation.context, 28, alarm.gp); - SET_GPR_U32(&invocation.context, 29, alarm.sp); + // See dispatchIrq. + SET_GPR_U32(&invocation.context, 29, 0u); SET_GPR_U32(&invocation.context, 31, 0u); queueInvocation(std::move(invocation)); break; From 1d4c4dc18d980943620cd79fa0a71539db59eeb4 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Wed, 19 Aug 2026 19:38:09 +0200 Subject: [PATCH 14/68] Let a guest own its C heap, and dispatch overlaid code Two related seams for games that do more than the runtime assumes. Guest-owned heap. A game shipping its own allocator (newlib and MSL both do) grows it through sbrk/EndOfHeap and hands out addresses the runtime must not also be allocating from. setGuestHeapCeiling makes EndOfHeap report the guest's ceiling rather than the runtime arena's limit, and SetupHeap no longer calls configureGuestHeap when one is set -- the guest is describing its own heap there, and honouring it dragged the runtime arena back on top of the guest's first chunk. guestHeapHardLimit exposes the top of the allocatable range, which guestHeapLimit cannot report before the heap is configured, so a caller can split the range without hardcoding the constant. Found in DQ8: the runtime's 128-byte GIF packet from sceGsResetGraph landed on the guest heap's top chunk header, dlmalloc read a zero size, and its own guard (set_head(top, PREV_INUSE), "will force null return from malloc") did exactly that. Overlay dispatch. The generated function table is one dense array over one address range, so it holds one function per guest address. Titles that stream code into a fixed arena break that. FunctionRegionResolver lets a host that translates each image separately say which table covers an address right now; nothing about overlay formats or loading enters the runtime. --- ps2xRuntime/include/ps2_runtime.h | 30 +++++++ .../src/lib/Kernel/Syscalls/System.cpp | 18 +++- ps2xRuntime/src/lib/ps2_runtime.cpp | 84 ++++++++++++++++--- 3 files changed, 119 insertions(+), 13 deletions(-) diff --git a/ps2xRuntime/include/ps2_runtime.h b/ps2xRuntime/include/ps2_runtime.h index 961e78ea8..15b106c98 100644 --- a/ps2xRuntime/include/ps2_runtime.h +++ b/ps2xRuntime/include/ps2_runtime.h @@ -345,6 +345,25 @@ class PS2Runtime SkipCallDebug = 3, }; + // Overlaid guest code: one dense table per streamed-in image, so the same + // arena address can mean different code depending on which is resident. + // The host owns the images and answers "which table covers this address?". + struct FunctionRegion + { + uint32_t base = 0u; // first guest address covered + uint32_t end = 0u; // one past the last covered address + uint32_t slotCount = 0u; // entries in `slots` + RecompiledFunction *slots = nullptr; // dense, indexed (addr - base) >> 2 + }; + + // Return the region covering `address`, or nullptr. Called on the guest + // thread from dispatch, so it must be cheap and must not block. + using FunctionRegionResolver = FunctionRegion *(*)(uint32_t address, void *userData); + + // Global rather than per-instance to match the generated table it extends. + // Passing nullptr removes the resolver. + static void setFunctionRegionResolver(FunctionRegionResolver resolver, void *userData); + bool replaceFunction(uint32_t address, RecompiledFunction func); // TODO remove this later need to update all tests bool registerFunction(uint32_t address, RecompiledFunction func); @@ -393,8 +412,18 @@ class PS2Runtime uint32_t guestRealloc(uint32_t guestAddr, uint32_t newSize, uint32_t alignment = 16u); void guestFree(uint32_t guestAddr); uint32_t guestHeapBase() const; + + // Ceiling for a guest that grows its own heap through sbrk/EndOfHeap. + // Zero (default) makes EndOfHeap report guestHeapLimit(), as before. + void setGuestHeapCeiling(uint32_t ceiling); + uint32_t guestHeapCeiling() const; + uint32_t guestHeapEnd() const; uint32_t guestHeapLimit() const; + + // Highest address any heap may reach. Unlike guestHeapLimit() this is + // valid before the heap is configured, so a caller can split the range. + uint32_t guestHeapHardLimit() const; uint32_t reserveAsyncCallbackStack(uint32_t size, uint32_t alignment = 16u); void drainCompletedDmacHandlers(uint8_t *rdram); @@ -515,6 +544,7 @@ class PS2Runtime uint32_t m_guestHeapBase = 0x00100000u; uint32_t m_guestHeapEnd = 0x00100000u; uint32_t m_guestHeapLimit = PS2_RAM_SIZE; + uint32_t m_guestHeapCeiling = 0u; uint32_t m_guestHeapSuggestedBase = 0x00100000u; bool m_guestHeapConfigured = false; uint32_t m_asyncCallbackStackFloor = 0x01F00000u; diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp index 3530e65bb..094a4039e 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp @@ -578,6 +578,14 @@ namespace ps2_syscalls if (runtime) { + // A guest running its own allocator is describing *its* heap; + // moving the runtime arena there would collide with it. + if (runtime->guestHeapCeiling() != 0u) + { + setReturnU32(ctx, heapBase); + return; + } + runtime->configureGuestHeap(heapBase, heapLimit); PS2_IF_AGRESSIVE_LOGS({ @@ -604,9 +612,13 @@ namespace ps2_syscalls static constexpr uint32_t kDefaultGuestHeapEnd = 0x01F00000u; - const uint32_t ret = runtime - ? runtime->guestHeapLimit() - : kDefaultGuestHeapEnd; + // A guest running its own allocator grows right up to this value. + uint32_t ret = kDefaultGuestHeapEnd; + if (runtime) + { + const uint32_t ceiling = runtime->guestHeapCeiling(); + ret = (ceiling != 0u) ? ceiling : runtime->guestHeapLimit(); + } setReturnU32(ctx, ret); } diff --git a/ps2xRuntime/src/lib/ps2_runtime.cpp b/ps2xRuntime/src/lib/ps2_runtime.cpp index e5d23d9b5..e107e4b18 100644 --- a/ps2xRuntime/src/lib/ps2_runtime.cpp +++ b/ps2xRuntime/src/lib/ps2_runtime.cpp @@ -1037,6 +1037,9 @@ void PS2Runtime::configureIoPathsFromElf(const std::string &elfPath) namespace { + std::atomic g_functionRegionResolver{nullptr}; + std::atomic g_functionRegionResolverUserData{nullptr}; + bool generatedFunctionTableSlot(uint32_t address, uint32_t &slot) { if ((address & 3u) != 0u || g_ps2RecompiledFunctionTableSlotCount == 0u) @@ -1053,21 +1056,67 @@ namespace slot = offset >> 2; return slot < g_ps2RecompiledFunctionTableSlotCount; } + + // Generated table first, then the host's region resolver for overlaid + // code. Returns a pointer into the owning table so replaceFunction can + // patch overlay entries too. + PS2Runtime::RecompiledFunction *functionTableSlotPointer(uint32_t address) + { + uint32_t slot = 0u; + if (generatedFunctionTableSlot(address, slot)) + { + return &g_ps2RecompiledFunctionTable[slot]; + } + + if ((address & 3u) != 0u) + { + return nullptr; + } + + const PS2Runtime::FunctionRegionResolver resolver = + g_functionRegionResolver.load(std::memory_order_acquire); + if (!resolver) + { + return nullptr; + } + + PS2Runtime::FunctionRegion *region = + resolver(address, g_functionRegionResolverUserData.load(std::memory_order_acquire)); + if (!region || !region->slots || address < region->base || address >= region->end) + { + return nullptr; + } + + const uint32_t regionSlot = (address - region->base) >> 2; + if (regionSlot >= region->slotCount) + { + return nullptr; + } + + return ®ion->slots[regionSlot]; + } +} + +void PS2Runtime::setFunctionRegionResolver(FunctionRegionResolver resolver, void *userData) +{ + g_functionRegionResolverUserData.store(userData, std::memory_order_release); + g_functionRegionResolver.store(resolver, std::memory_order_release); } bool PS2Runtime::replaceFunction(uint32_t address, RecompiledFunction func) { - uint32_t slot = 0u; - if (!generatedFunctionTableSlot(address, slot)) + RecompiledFunction *slot = functionTableSlotPointer(address); + if (!slot) { std::cerr << "[function-table] cannot replace guest PC 0x" << std::hex << address << ": outside generated dense table [0x" << g_ps2RecompiledFunctionTableBase << ", 0x" << g_ps2RecompiledFunctionTableEnd << ")" + << " and not claimed by a function-region resolver" << std::dec << std::endl; return false; } - g_ps2RecompiledFunctionTable[slot] = func; + *slot = func; return true; } @@ -1078,8 +1127,8 @@ bool PS2Runtime::registerFunction(uint32_t address, RecompiledFunction func) bool PS2Runtime::hasFunction(uint32_t address) const { - uint32_t slot = 0u; - return generatedFunctionTableSlot(address, slot) && g_ps2RecompiledFunctionTable[slot] != nullptr; + const RecompiledFunction *slot = functionTableSlotPointer(address); + return slot != nullptr && *slot != nullptr; } const char *describeGuestBranchKind(PS2Runtime::GuestBranchKind kind) @@ -1105,13 +1154,11 @@ PS2Runtime::RecompiledFunction PS2Runtime::lookupFunction(uint32_t address) { pushDispatchPc(address); - uint32_t slot = 0u; - if (generatedFunctionTableSlot(address, slot)) + if (const RecompiledFunction *slot = functionTableSlotPointer(address)) { - RecompiledFunction fn = g_ps2RecompiledFunctionTable[slot]; - if (fn != nullptr) + if (*slot != nullptr) { - return fn; + return *slot; } } @@ -1935,12 +1982,29 @@ uint32_t PS2Runtime::guestHeapEnd() const return m_guestHeapConfigured ? m_guestHeapEnd : m_guestHeapSuggestedBase; } +void PS2Runtime::setGuestHeapCeiling(uint32_t ceiling) +{ + std::lock_guard lock(m_guestHeapMutex); + m_guestHeapCeiling = ceiling; +} + +uint32_t PS2Runtime::guestHeapCeiling() const +{ + std::lock_guard lock(m_guestHeapMutex); + return m_guestHeapCeiling; +} + uint32_t PS2Runtime::guestHeapLimit() const { std::lock_guard lock(m_guestHeapMutex); return m_guestHeapConfigured ? m_guestHeapLimit : m_guestHeapSuggestedBase; } +uint32_t PS2Runtime::guestHeapHardLimit() const +{ + return std::min(kGuestHeapHardLimit, PS2_RAM_SIZE); +} + uint32_t PS2Runtime::reserveAsyncCallbackStack(uint32_t size, uint32_t alignment) { if (size == 0u) From 76af5aa99d6e75263c1f170225ff2234874c3048 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Wed, 19 Aug 2026 21:12:52 +0200 Subject: [PATCH 15/68] Raise INTC VIF0/VIF1 when a VIFcode carries the i bit The VIF interpreters decoded the bit and set VIFn_STAT.INT, but nothing turned that into an interrupt, so a game that asks to be told when its display list completes waits forever. The acknowledge side already worked: a VIFn_FBRST write with STC clears the STAT bits. PS2Memory latches the request (bit 0 VIF0, bit 1 VIF1) and EeScheduler::processPendingEvents drains it into dispatchIrq, mirroring how EE timer interrupts already travel from advanceEeTimers to the scheduler. DQ8 sends one such VIFcode -- 0x80000000, a VIF NOP with the i bit -- then blocks on a semaphore its INTC cause 5 handler signals. Without this the game stops with the title artwork loaded and never draws. With it, it renders. Not emulated: hardware stalls VIF1 at the interrupt and resumes on the STC write, while the interpreter runs on as it already did when it only set STAT.INT. DQ8 puts the i bit on the last VIFcode of the packet so there is nothing left to stall; a game that interrupts mid-packet would need it. --- ps2xRuntime/include/runtime/ps2_memory.h | 5 +++++ ps2xRuntime/src/lib/Kernel/EeScheduler.cpp | 11 +++++++++++ ps2xRuntime/src/lib/ps2_memory.cpp | 5 +++++ ps2xRuntime/src/lib/ps2_vif1_interpreter.cpp | 6 ++++++ 4 files changed, 27 insertions(+) diff --git a/ps2xRuntime/include/runtime/ps2_memory.h b/ps2xRuntime/include/runtime/ps2_memory.h index cea5b98a8..20d08ffc2 100644 --- a/ps2xRuntime/include/runtime/ps2_memory.h +++ b/ps2xRuntime/include/runtime/ps2_memory.h @@ -313,6 +313,10 @@ class PS2Memory [[nodiscard]] uint64_t cyclesUntilNextEeTimerInterrupt() const noexcept; void resetEeTimers() noexcept; + // A VIFcode carrying the i bit raises INTC VIF0/VIF1. Bit 0 is VIF0, + // bit 1 is VIF1; the scheduler drains this and dispatches the handlers. + uint32_t takePendingVifInterrupts() noexcept; + using GifPacketCallback = std::function; void setGifPacketCallback(GifPacketCallback cb) { m_gifPacketCallback = std::move(cb); } void setGifArbiter(GifArbiter *arbiter) { m_gifArbiter = arbiter; } @@ -374,6 +378,7 @@ class PS2Memory std::atomic m_gifCopyCount{0}; std::atomic m_gsWriteCount{0}; std::atomic m_vifWriteCount{0}; + std::atomic m_pendingVifInterrupts{0}; std::atomic m_vu0CodeGeneration{0}; std::atomic m_vu1CodeGeneration{0}; // I/O registers diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index 1a6b3d3e1..790f80820 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -1765,6 +1765,17 @@ void EeScheduler::processPendingEvents() dispatchIrq(false, 9u + timer); } } + // INTC VIF0 (4) and VIF1 (5), raised by a VIFcode carrying the i bit. + const uint32_t vifInterrupts = m_runtime.memory().takePendingVifInterrupts(); + if ((vifInterrupts & 0x1u) != 0u) + { + dispatchIrq(false, 4u); + } + if ((vifInterrupts & 0x2u) != 0u) + { + dispatchIrq(false, 5u); + } + std::deque pending; { std::lock_guard lock(m_eventMutex); diff --git a/ps2xRuntime/src/lib/ps2_memory.cpp b/ps2xRuntime/src/lib/ps2_memory.cpp index 7cb2ba463..454583fc0 100644 --- a/ps2xRuntime/src/lib/ps2_memory.cpp +++ b/ps2xRuntime/src/lib/ps2_memory.cpp @@ -404,6 +404,11 @@ void PS2Memory::resetEeTimers() noexcept m_eeTimers = {}; } +uint32_t PS2Memory::takePendingVifInterrupts() noexcept +{ + return m_pendingVifInterrupts.exchange(0u, std::memory_order_acq_rel); +} + uint32_t PS2Memory::advanceEeTimers(uint64_t eeCycles) noexcept { if (eeCycles == 0u) diff --git a/ps2xRuntime/src/lib/ps2_vif1_interpreter.cpp b/ps2xRuntime/src/lib/ps2_vif1_interpreter.cpp index 05fc764a8..fd95e43d1 100644 --- a/ps2xRuntime/src/lib/ps2_vif1_interpreter.cpp +++ b/ps2xRuntime/src/lib/ps2_vif1_interpreter.cpp @@ -77,7 +77,10 @@ void PS2Memory::processVIF0Data(const uint8_t *data, uint32_t sizeBytes) vif0_regs.code = cmd; vif0_regs.num = num; if (irq) + { vif0_regs.stat |= (1u << 11); + m_pendingVifInterrupts.fetch_or(0x1u, std::memory_order_relaxed); + } if (opcode == VIF_NOP) { @@ -312,7 +315,10 @@ void PS2Memory::processVIF1Data(const uint8_t *data, uint32_t sizeBytes) vif1_regs.code = cmd; vif1_regs.num = num; if (irq) + { vif1_regs.stat |= (1u << 11); // INT + m_pendingVifInterrupts.fetch_or(0x2u, std::memory_order_relaxed); + } if (opcode == VIF_NOP) { From 7ab0b7c71786d7d6fedc5cab282ffcd4d0fa047a Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Wed, 19 Aug 2026 21:12:52 +0200 Subject: [PATCH 16/68] Weave interlaced frames instead of line-doubling one field SMODE2.FFMD selects what the frame buffer holds, not how the CRT scans it, and PresentFromLocalMemory had it inverted. FFMD=0 buffers a whole frame whose two fields are alternate rows, so weaving to a progressive host frame is just reading every row; FFMD=1 buffers half the display height and each field reads all of it, which is where line-doubling belongs. Applying the FFMD=1 rule to an FFMD=0 game threw away half the vertical detail. Also snaps the two PCRTC read circuits together when they cover the same buffer one row apart. That configuration is a deliberate vertical flicker filter and merging it faithfully costs half a row of blur; PCSX2 snaps by default. This is the one deliberate departure from hardware here, and it is isolated -- deleting it costs about 1.6 dB. Measured on six DQ8 title-screen GS dumps: identical adjacent rows 224/447 -> 0/447, PSNR 28.97-29.12 dB -> 29.53-29.67 dB. The win holds under both resample directions. The FFMD=1 branch has no test coverage -- every dump is FFMD=0 -- and is reasoned from the spec rather than exercised. --- .../include/runtime/gs/gs_cpu_backend.h | 3 +- ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp | 49 ++++++++----------- 2 files changed, 22 insertions(+), 30 deletions(-) diff --git a/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h b/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h index 71e80324f..c428fed9a 100644 --- a/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h +++ b/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h @@ -56,7 +56,8 @@ class GSCpuBackend final : public GSRasterBackend bool useLocalMemoryLayout, bool frameBaseIsPages, uint32_t sourceOriginX, - uint32_t sourceOriginY) const; + uint32_t sourceOriginY, + bool doubleSourceRows = false) const; using WriteVramFunc = std::function; using ReadVramFunc = std::function; diff --git a/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp b/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp index 9c39ae2e7..329f3da01 100644 --- a/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp @@ -420,22 +420,6 @@ namespace return {(smode2 & 0x1ull) != 0ull, ((smode2 >> 1) & 0x1ull) != 0ull}; } - void applyFieldPresentation(std::vector &pixels, uint32_t width, uint32_t height, bool oddField) - { - if (pixels.empty() || width == 0u || height < 2u) - return; - const std::vector source = pixels; - for (uint32_t y = 0; y < height; ++y) - { - uint32_t sourceY = ((y >> 1u) << 1u) + (oddField ? 1u : 0u); - if (sourceY >= height) - sourceY = height - 1u; - std::memcpy(pixels.data() + y * kHostFrameWidth * 4u, - source.data() + sourceY * kHostFrameWidth * 4u, - width * 4u); - } - } - void normalizePresentationAlpha(std::vector &pixels, uint32_t width, uint32_t height) { for (uint32_t y = 0; y < height; ++y) @@ -1687,7 +1671,8 @@ bool GSCpuBackend::CopyFrameToHostRgba(const GSFrameReg &frame, bool useLocalMemoryLayout, bool frameBaseIsPages, uint32_t sourceOriginX, - uint32_t sourceOriginY) const + uint32_t sourceOriginY, + bool doubleSourceRows) const { if (!m_vram || m_vramSize == 0u) return false; @@ -1705,7 +1690,7 @@ bool GSCpuBackend::CopyFrameToHostRgba(const GSFrameReg &frame, for (uint32_t x = 0; x < width; ++x) { const uint32_t sx = sourceOriginX + x; - const uint32_t sy = sourceOriginY + y; + const uint32_t sy = sourceOriginY + (doubleSourceRows ? (y >> 1u) : y); if (frame.psm == GS_PSM_CT32 || frame.psm == GS_PSM_CT24) { uint32_t color = 0u; @@ -1776,12 +1761,14 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque PresentationFrame result{}; const GSPmodeState pmode = decodePmode(request.pmode); const GSSmode2State smode2 = decodeSMode2(request.smode2); - const bool fieldMode = smode2.interlaced && !smode2.frameMode; - const bool oddField = (request.vsyncTick & 1ull) != 0ull; + // SMODE2 FFMD=1 (FRAME) reads a half-height buffer whole once per field, so the + // woven host frame line-doubles it. FFMD=0 (FIELD) buffers a full frame whose + // two fields are alternate buffer rows: weaving is just reading every row. + const bool halfHeightSource = smode2.interlaced && smode2.frameMode; const GSFrameReg displayFrame1 = decodeDisplayFrame(request.dispfb1); const GSFrameReg displayFrame2 = decodeDisplayFrame(request.dispfb2); - const GSDisplayReadOrigin origin1 = decodeDisplayReadOrigin(request.dispfb1); - const GSDisplayReadOrigin origin2 = decodeDisplayReadOrigin(request.dispfb2); + GSDisplayReadOrigin origin1 = decodeDisplayReadOrigin(request.dispfb1); + GSDisplayReadOrigin origin2 = decodeDisplayReadOrigin(request.dispfb2); uint32_t width1 = 0u, height1 = 0u, width2 = 0u, height2 = 0u; decodeDisplaySize(request.display1, width1, height1); decodeDisplaySize(request.display2, width2, height2); @@ -1790,6 +1777,14 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque if (!valid1 && !valid2) return result; + // Both circuits on one buffer a single row apart is a deliberate flicker filter; + // snapping them together drops the half-row blur, as PCSX2 does by default. + const uint32_t originGap = origin1.y > origin2.y ? origin1.y - origin2.y : origin2.y - origin1.y; + const bool sameReadSource = displayFrame1.fbp == displayFrame2.fbp && displayFrame1.fbw == displayFrame2.fbw && + displayFrame1.psm == displayFrame2.psm && origin1.x == origin2.x; + if (valid1 && valid2 && sameReadSource && originGap == 1u) + origin1.y = origin2.y = std::min(origin1.y, origin2.y); + auto copySource = [&](const GSFrameReg &displayFrame, const GSDisplayReadOrigin &origin, uint32_t width, @@ -1805,12 +1800,12 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque usedPreferred = false; if (allowPreferred && request.hasPreferredSource && request.preferredDestFbp == displayFrame.fbp && (request.preferredSource.fbw != 0u || request.preferredSource.fbp != displayFrame.fbp) && - CopyFrameToHostRgba(request.preferredSource, width, height, pixels, preserveAlpha, true, false, 0u, 0u)) + CopyFrameToHostRgba(request.preferredSource, width, height, pixels, preserveAlpha, true, false, 0u, 0u, halfHeightSource)) { selected = request.preferredSource; usedPreferred = true; } - if (pixels.empty() && !CopyFrameToHostRgba(displayFrame, width, height, pixels, preserveAlpha, true, true, origin.x, origin.y)) + if (pixels.empty() && !CopyFrameToHostRgba(displayFrame, width, height, pixels, preserveAlpha, true, true, origin.x, origin.y, halfHeightSource)) return false; if (!usedPreferred && displayFrame.fbp == 0u && countNonBlackPixels(pixels, width, height) == 0u) @@ -1820,7 +1815,7 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque if (candidate.fbp == selected.fbp && candidate.fbw == selected.fbw && candidate.psm == selected.psm) continue; std::vector candidatePixels; - if (!CopyFrameToHostRgba(candidate, width, height, candidatePixels, preserveAlpha, true, true, 0u, 0u)) + if (!CopyFrameToHostRgba(candidate, width, height, candidatePixels, preserveAlpha, true, true, 0u, 0u, halfHeightSource)) continue; if (countNonBlackPixels(candidatePixels, width, height) == 0u) continue; @@ -1870,8 +1865,6 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque dst[3] = pmode.amod ? dst[3] : src[3]; } normalizePresentationAlpha(result.pixels, result.width, result.height); - if (fieldMode) - applyFieldPresentation(result.pixels, result.width, result.height, oddField); result.displayFbp = displayFrame1.fbp; result.sourceFbp = selected1.fbp; return result; @@ -1885,8 +1878,6 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque GSFrameReg selected = displayFrame; if (!copySource(displayFrame, origin, result.width, result.height, true, false, selected, result.pixels, result.usedPreferred)) return {}; - if (fieldMode) - applyFieldPresentation(result.pixels, result.width, result.height, oddField); normalizePresentationAlpha(result.pixels, result.width, result.height); result.displayFbp = displayFrame.fbp; result.sourceFbp = selected.fbp; From 60a0d2df114a634b2083b989c36c2d35ee963138 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Wed, 19 Aug 2026 23:59:07 +0200 Subject: [PATCH 17/68] Latch pad input and snap presentation to integer scale Two defects that both trace back to the frame rate, and one hazard beside them. Input. readState() sampled IsKeyDown() on the EE thread once per scePadRead, which is a level read of raylib state refreshed once per presented frame. At 7 fps that is a 140 ms window and a press-and-release inside it was never observed at all. Edge detection alone does not fix this: PollInputEvents copies current->previous and then calls glfwPollEvents, and GLFW's callback sets the key to 1 on press and 0 on release, so a tap whose press and release land in the same poll leaves both states at 0. Measured, 5 ms taps fired on 0 of 51 frames for IsKeyDown and IsKeyPressed alike. What survives is keyPressedQueue, which the callback appends to regardless. ps2PadPollHost() therefore runs on the render thread and publishes two atomics -- a held mask from IsKeyDown and a latch drained from GetKeyPressed -- and readState() ORs them, consuming the latch exactly once. Taps of 5/20/60 ms went from 0/15, 3/15, 6/15 seen to 15/15 in all three cases, one guest frame of button-down each. Declared in a new ps2_pad_host.h rather than on PSPadBackend: ps2_pad.h is reachable from ps2_runtime.h, so putting it there rebuilds the whole recompiled corpus for an input change. Presentation. The window is resizable and SetTextureFilter was never called, so a non-integer nearest-neighbour scale duplicated some pixel columns and dropped others -- at 700x500 only 510 of 512 source columns survived, and 1920x1080 gave runs of 2 and 3 pixels. One-pixel font stems came out uneven. Upscales now snap to floor(fitScale) with point sampling, bilinear is used only below 1.0 where integer scaling has nowhere to go, and the destination origin is floored because odd centring offsets were an independent half-pixel source. Verified over 15 window sizes: 8 of 12 upscales mangled before, 0 after, all 512 columns present with uniform run widths. The window now opens at 2x when that fits in 90% of the monitor. Also SetExitKey(KEY_NULL): Escape is bound to Circle, and raylib's default exit key is Escape, so pressing Circle closed the game. --- ps2xRuntime/include/runtime/ps2_pad_host.h | 10 + ps2xRuntime/src/lib/ps2_pad.cpp | 266 +++++++++++++++------ ps2xRuntime/src/lib/ps2_runtime.cpp | 47 +++- 3 files changed, 244 insertions(+), 79 deletions(-) create mode 100644 ps2xRuntime/include/runtime/ps2_pad_host.h diff --git a/ps2xRuntime/include/runtime/ps2_pad_host.h b/ps2xRuntime/include/runtime/ps2_pad_host.h new file mode 100644 index 000000000..b7652de89 --- /dev/null +++ b/ps2xRuntime/include/runtime/ps2_pad_host.h @@ -0,0 +1,10 @@ +#ifndef PS2_PAD_HOST_H +#define PS2_PAD_HOST_H + +// Render-thread half of the pad backend. Samples raylib input once per presented +// frame and latches press edges so a tap between two guest polls still registers. +// Declared here rather than on PSPadBackend so the recompiled corpus, which +// includes ps2_pad.h through ps2_runtime.h, does not rebuild for input changes. +void ps2PadPollHost(); + +#endif diff --git a/ps2xRuntime/src/lib/ps2_pad.cpp b/ps2xRuntime/src/lib/ps2_pad.cpp index 8590f1424..e5a26e5ca 100644 --- a/ps2xRuntime/src/lib/ps2_pad.cpp +++ b/ps2xRuntime/src/lib/ps2_pad.cpp @@ -1,11 +1,20 @@ #include "runtime/ps2_pad.h" +#include "runtime/ps2_pad_host.h" #include "ps2_host_backend.h" +#include +#include #include namespace { constexpr uint8_t kPadAnalogMarker = 0x73; constexpr uint8_t kPadStickCenter = 0x80; + constexpr uint32_t kPadStickNeutral = 0x80808080u; + constexpr int kGamepad = 0; + + // Drop a latch the guest never came back for, so a tap during a long + // non-polling stretch (loading, cutscene) does not fire much later. + constexpr uint64_t kLatchTimeoutMs = 1000u; constexpr uint16_t PAD_LEFT = 0x0080u; constexpr uint16_t PAD_DOWN = 0x0040u; @@ -23,6 +32,172 @@ namespace constexpr uint16_t PAD_L1 = 0x0400u; constexpr uint16_t PAD_R2 = 0x0200u; constexpr uint16_t PAD_L2 = 0x0100u; + + // Published by the render thread in ps2PadPollHost(), consumed on the EE thread. + std::atomic g_hostPolled{false}; + std::atomic g_held{0u}; // active-high PAD_* mask + std::atomic g_latched{0u}; // press edges not consumed yet + std::atomic g_sticks{kPadStickNeutral}; // packed rx,ry,lx,ly + std::atomic g_latchStampMs{0u}; + + struct KeyBinding + { + int key; + uint16_t mask; + }; + + constexpr KeyBinding kKeyBindings[] = { + {KEY_UP, PAD_UP}, {KEY_W, PAD_UP}, + {KEY_DOWN, PAD_DOWN}, {KEY_S, PAD_DOWN}, + {KEY_LEFT, PAD_LEFT}, {KEY_A, PAD_LEFT}, + {KEY_RIGHT, PAD_RIGHT}, {KEY_D, PAD_RIGHT}, + {KEY_X, PAD_CROSS}, {KEY_SPACE, PAD_CROSS}, + {KEY_C, PAD_CIRCLE}, {KEY_ESCAPE, PAD_CIRCLE}, + {KEY_Z, PAD_SQUARE}, {KEY_KP_0, PAD_SQUARE}, + {KEY_V, PAD_TRIANGLE}, {KEY_KP_1, PAD_TRIANGLE}, + {KEY_Q, PAD_L1}, + {KEY_E, PAD_R1}, + {KEY_LEFT_SHIFT, PAD_L2}, + {KEY_RIGHT_SHIFT, PAD_R2}, + {KEY_ENTER, PAD_START}, + {KEY_TAB, PAD_SELECT}, + }; + + struct PadBinding + { + int button; + uint16_t mask; + }; + + constexpr PadBinding kPadBindings[] = { + {GAMEPAD_BUTTON_LEFT_FACE_UP, PAD_UP}, + {GAMEPAD_BUTTON_LEFT_FACE_DOWN, PAD_DOWN}, + {GAMEPAD_BUTTON_LEFT_FACE_LEFT, PAD_LEFT}, + {GAMEPAD_BUTTON_LEFT_FACE_RIGHT, PAD_RIGHT}, + {GAMEPAD_BUTTON_RIGHT_FACE_DOWN, PAD_CROSS}, + {GAMEPAD_BUTTON_RIGHT_FACE_RIGHT, PAD_CIRCLE}, + {GAMEPAD_BUTTON_RIGHT_FACE_LEFT, PAD_SQUARE}, + {GAMEPAD_BUTTON_RIGHT_FACE_UP, PAD_TRIANGLE}, + {GAMEPAD_BUTTON_LEFT_TRIGGER_1, PAD_L1}, + {GAMEPAD_BUTTON_RIGHT_TRIGGER_1, PAD_R1}, + {GAMEPAD_BUTTON_LEFT_TRIGGER_2, PAD_L2}, + {GAMEPAD_BUTTON_RIGHT_TRIGGER_2, PAD_R2}, + {GAMEPAD_BUTTON_MIDDLE_RIGHT, PAD_START}, + {GAMEPAD_BUTTON_MIDDLE_LEFT, PAD_SELECT}, + {GAMEPAD_BUTTON_LEFT_THUMB, PAD_L3}, + {GAMEPAD_BUTTON_RIGHT_THUMB, PAD_R3}, + }; + + uint8_t axisToByte(float axis) + { + const float mapped = 128.0f + axis * 127.0f; + return static_cast(mapped < 0.0f ? 0.0f : (mapped > 255.0f ? 255.0f : mapped)); + } + + uint64_t nowMs() + { + using namespace std::chrono; + return static_cast( + duration_cast(steady_clock::now().time_since_epoch()).count()); + } + + uint32_t keyMask(int key) + { + uint32_t mask = 0u; + for (const KeyBinding &binding : kKeyBindings) + { + if (binding.key == key) + { + mask |= binding.mask; + } + } + return mask; + } + + // drainQueue must only be true on the render thread: GetKeyPressed() mutates + // raylib's queue, while every other call here is a plain read. + void sampleHost(bool drainQueue, uint32_t &held, uint32_t &pressed, uint32_t &sticks) + { + held = 0u; + pressed = 0u; + sticks = kPadStickNeutral; + + if (IsGamepadAvailable(kGamepad)) + { + for (const PadBinding &binding : kPadBindings) + { + if (IsGamepadButtonDown(kGamepad, binding.button)) + { + held |= binding.mask; + } + if (IsGamepadButtonPressed(kGamepad, binding.button)) + { + pressed |= binding.mask; + } + } + + const uint8_t rx = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_RIGHT_X)); + const uint8_t ry = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_RIGHT_Y)); + const uint8_t lx = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_LEFT_X)); + const uint8_t ly = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_LEFT_Y)); + sticks = (static_cast(rx) << 24) | (static_cast(ry) << 16) | + (static_cast(lx) << 8) | static_cast(ly); + return; + } + + for (const KeyBinding &binding : kKeyBindings) + { + if (IsKeyDown(binding.key)) + { + held |= binding.mask; + } + } + + if (drainQueue) + { + // raylib queues every GLFW press, including one released again inside + // the same poll, which IsKeyPressed() would already have missed. + for (int key = GetKeyPressed(); key != 0; key = GetKeyPressed()) + { + pressed |= keyMask(key); + } + } + } +} + +void ps2PadPollHost() +{ + if (!IsWindowReady()) + { + return; + } + + uint32_t held = 0u; + uint32_t pressed = 0u; + uint32_t sticks = kPadStickNeutral; + sampleHost(true, held, pressed, sticks); + + g_held.store(held); + g_sticks.store(sticks); + + const uint64_t now = nowMs(); + if (pressed != 0u) + { + if (g_latched.fetch_or(pressed) == 0u) + { + g_latchStampMs.store(now); + } + } + else + { + const uint32_t stale = g_latched.load(); + if (stale != 0u && (now - g_latchStampMs.load()) > kLatchTimeoutMs) + { + g_latched.fetch_and(~stale); + } + } + + g_hostPolled.store(true); } bool PSPadBackend::readState(int /*port*/, int /*slot*/, uint8_t *data, size_t size) @@ -37,88 +212,27 @@ bool PSPadBackend::readState(int /*port*/, int /*slot*/, uint8_t *data, size_t s data[3] = 0xFF; data[4] = data[5] = data[6] = data[7] = kPadStickCenter; - uint16_t btns = 0xFFFFu; - constexpr int kGamepad = 0; - const bool useGamepad = IsGamepadAvailable(kGamepad); - auto clearBit = [&btns](uint16_t mask) - { btns &= ~mask; }; - - if (useGamepad) + uint32_t active = 0u; + uint32_t sticks = kPadStickNeutral; + if (g_hostPolled.load()) { - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_FACE_UP)) - clearBit(PAD_UP); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_FACE_DOWN)) - clearBit(PAD_DOWN); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_FACE_LEFT)) - clearBit(PAD_LEFT); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_FACE_RIGHT)) - clearBit(PAD_RIGHT); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_FACE_DOWN)) - clearBit(PAD_CROSS); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_FACE_RIGHT)) - clearBit(PAD_CIRCLE); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_FACE_LEFT)) - clearBit(PAD_SQUARE); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_FACE_UP)) - clearBit(PAD_TRIANGLE); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_TRIGGER_1)) - clearBit(PAD_L1); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_TRIGGER_1)) - clearBit(PAD_R1); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_TRIGGER_2)) - clearBit(PAD_L2); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_TRIGGER_2)) - clearBit(PAD_R2); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_MIDDLE_RIGHT)) - clearBit(PAD_START); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_MIDDLE_LEFT)) - clearBit(PAD_SELECT); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_LEFT_THUMB)) - clearBit(PAD_L3); - if (IsGamepadButtonDown(kGamepad, GAMEPAD_BUTTON_RIGHT_THUMB)) - clearBit(PAD_R3); - - float lx = GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_LEFT_X); - float ly = GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_LEFT_Y); - float rx = GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_RIGHT_X); - float ry = GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_RIGHT_Y); - data[6] = static_cast(128 + lx * 127); - data[7] = static_cast(128 + ly * 127); - data[4] = static_cast(128 + rx * 127); - data[5] = static_cast(128 + ry * 127); + // exchange() so each latched press is handed to the guest exactly once. + active = g_held.load() | g_latched.exchange(0u); + sticks = g_sticks.load(); } else { - if (IsKeyDown(KEY_UP) || IsKeyDown(KEY_W)) - clearBit(PAD_UP); - if (IsKeyDown(KEY_DOWN) || IsKeyDown(KEY_S)) - clearBit(PAD_DOWN); - if (IsKeyDown(KEY_LEFT) || IsKeyDown(KEY_A)) - clearBit(PAD_LEFT); - if (IsKeyDown(KEY_RIGHT) || IsKeyDown(KEY_D)) - clearBit(PAD_RIGHT); - if (IsKeyDown(KEY_X) || IsKeyDown(KEY_SPACE)) - clearBit(PAD_CROSS); - if (IsKeyDown(KEY_C) || IsKeyDown(KEY_ESCAPE)) - clearBit(PAD_CIRCLE); - if (IsKeyDown(KEY_Z) || IsKeyDown(KEY_KP_0)) - clearBit(PAD_SQUARE); - if (IsKeyDown(KEY_V) || IsKeyDown(KEY_KP_1)) - clearBit(PAD_TRIANGLE); - if (IsKeyDown(KEY_Q)) - clearBit(PAD_L1); - if (IsKeyDown(KEY_E)) - clearBit(PAD_R1); - if (IsKeyDown(KEY_LEFT_SHIFT)) - clearBit(PAD_L2); - if (IsKeyDown(KEY_RIGHT_SHIFT)) - clearBit(PAD_R2); - if (IsKeyDown(KEY_ENTER)) - clearBit(PAD_START); - if (IsKeyDown(KEY_TAB)) - clearBit(PAD_SELECT); + // No host present loop (embedders that never call ps2PadPollHost). + uint32_t pressed = 0u; + sampleHost(false, active, pressed, sticks); } + data[4] = static_cast(sticks >> 24); + data[5] = static_cast(sticks >> 16); + data[6] = static_cast(sticks >> 8); + data[7] = static_cast(sticks); + + const uint16_t btns = static_cast(0xFFFFu & ~active); data[2] = static_cast(btns & 0xFF); data[3] = static_cast(btns >> 8); return true; diff --git a/ps2xRuntime/src/lib/ps2_runtime.cpp b/ps2xRuntime/src/lib/ps2_runtime.cpp index e107e4b18..345a30b51 100644 --- a/ps2xRuntime/src/lib/ps2_runtime.cpp +++ b/ps2xRuntime/src/lib/ps2_runtime.cpp @@ -5,6 +5,7 @@ #include "game_overrides.h" #include "ps2_runtime_macros.h" #include "runtime/gs/gs_frontend.h" +#include "runtime/ps2_pad_host.h" #include "runtime/ee_scheduler.h" #include "ThreadNaming.h" #include "Kernel/Stubs/Audio.h" @@ -19,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -673,6 +675,28 @@ bool PS2Runtime::syncCoreSubsystems() return true; } +// Grow the default window to an exact 2x of the PS2 output when the monitor has +// room, so it opens both larger and integer-scaled. Resizable either way. +static void applyDefaultWindowScale() +{ + constexpr int kScale = 2; + const int monitor = GetCurrentMonitor(); + const int monitorWidth = GetMonitorWidth(monitor); + const int monitorHeight = GetMonitorHeight(monitor); + const int width = HOST_WINDOW_WIDTH * kScale; + const int height = HOST_WINDOW_HEIGHT * kScale; + if (monitorWidth <= 0 || monitorHeight <= 0 || + width * 10 > monitorWidth * 9 || height * 10 > monitorHeight * 9) + { + return; + } + + const Vector2 origin = GetMonitorPosition(monitor); + SetWindowSize(width, height); + SetWindowPosition(static_cast(origin.x) + (monitorWidth - width) / 2, + static_cast(origin.y) + (monitorHeight - height) / 2); +} + bool PS2Runtime::initialize(const char *title) { try @@ -702,6 +726,9 @@ bool PS2Runtime::initialize(const char *title) #else SetConfigFlags(FLAG_WINDOW_RESIZABLE); InitWindow(HOST_WINDOW_WIDTH, HOST_WINDOW_HEIGHT, title); + // Escape is bound to Circle; without this raylib also closes on it. + SetExitKey(KEY_NULL); + applyDefaultWindowScale(); InitAudioDevice(); m_audioBackend.setAudioReady(IsAudioDeviceReady()); #endif @@ -2391,6 +2418,8 @@ void PS2Runtime::run() Image blank = GenImageColor(FB_WIDTH, FB_HEIGHT, BLANK); Texture2D frameTex = LoadTextureFromImage(blank); UnloadImage(blank); + int frameTexFilter = TEXTURE_FILTER_POINT; + SetTextureFilter(frameTex, frameTexFilter); std::atomic gameThreadFinished{false}; @@ -2459,13 +2488,24 @@ void PS2Runtime::run() const float srcHeight = static_cast(std::max(1u, presentHeight)); const float screenWidth = static_cast(GetScreenWidth()); const float screenHeight = static_cast(GetScreenHeight()); - const float scale = std::min(screenWidth / srcWidth, screenHeight / srcHeight); + const float fitScale = std::min(screenWidth / srcWidth, screenHeight / srcHeight); + // Snap upscales to an integer multiple: a fractional nearest-neighbour scale + // duplicates some pixel columns and not others, which shreds 1px font stems. + const bool integerScale = fitScale >= 1.0f; + const float scale = integerScale ? std::floor(fitScale) : fitScale; + const int wantedFilter = integerScale ? TEXTURE_FILTER_POINT : TEXTURE_FILTER_BILINEAR; + if (wantedFilter != frameTexFilter) + { + SetTextureFilter(frameTex, wantedFilter); + frameTexFilter = wantedFilter; + } const float dstWidth = srcWidth * scale; const float dstHeight = srcHeight * scale; const Rectangle srcRect{0.0f, 0.0f, srcWidth, srcHeight}; + // Whole-pixel origin, else centring can land the quad on a half pixel. const Rectangle dstRect{ - (screenWidth - dstWidth) * 0.5f, - (screenHeight - dstHeight) * 0.5f, + std::floor((screenWidth - dstWidth) * 0.5f), + std::floor((screenHeight - dstHeight) * 0.5f), dstWidth, dstHeight}; DrawTexturePro(frameTex, srcRect, dstRect, Vector2{0.0f, 0.0f}, 0.0f, WHITE); @@ -2474,6 +2514,7 @@ void PS2Runtime::run() m_debugUiDrawCallback(*this, m_debugUiUserData); } EndDrawing(); + ps2PadPollHost(); // EndDrawing() polled raylib input; latch edges for the EE thread if (WindowShouldClose()) { From de62c4c66867aef540d3e39506b8aefd70e6b54c Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Thu, 20 Aug 2026 00:41:43 +0200 Subject: [PATCH 18/68] Bound invocation nesting so a callback storm cannot exhaust the stack pool Pending invocations are pushed onto the running thread between any two guest function dispatches, including partway through one already in flight, so they nest rather than run in sequence. DQ8's movie streaming queues SIF RPC completion callbacks faster than they retire and reached depth 13 on one thread, exhausting the invocation stack pool and throwing. Past four deep the rest simply stay on m_pendingInvocations and are picked up as the depth drops. Real nesting is at most a couple deep -- an interrupt during a callback -- so this bounds the pathological case without changing legitimate behaviour, and an unbounded depth cannot be supported anyway: the pool is the fixed gap between the runtime arena and the guest's own stack. Measured on DQ8: 16 stacks consumed then failure, all but two of them thread 12 at depths 0 through 13, all GuestInvocationKind::RpcCallback. With the bound the game runs past the movie open into its sound settings screen. --- ps2xRuntime/src/lib/Kernel/EeScheduler.cpp | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index 790f80820..3ecc929d1 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -252,7 +252,12 @@ void EeScheduler::run() continue; } - if (!m_pendingInvocations.empty()) + // Invocations nest onto the running thread, so a callback storm stacks + // them faster than they retire -- DQ8's movie streaming reached depth 13 + // and exhausted the stack pool. Past this the rest simply stay queued. + constexpr size_t kMaxInvocationDepth = 4u; + if (!m_pendingInvocations.empty() && + running->invocations.size() < kMaxInvocationDepth) { GuestInvocation invocation = std::move(m_pendingInvocations.front()); m_pendingInvocations.pop_front(); From 96c2ecf9aec1d082854d0fbc2b80b1dfad5d41fe Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Thu, 20 Aug 2026 17:12:17 +0200 Subject: [PATCH 19/68] Let a PSS movie finish: decoder availability, stream end, upload gate Three defects that between them left DQ8's logo movie as a frozen frame that never terminated. All are reachable by any game that plays a PSS movie. 1. A runtime built without FFmpeg deadlocks. The stub decoder's feed() returns false and the failure path sets decoderFailed = false, which is right for the FFmpeg build where false means "no sequence header yet, resync". With no decoder that is not recoverable, so sceMpegGetPicture parks on a frame that can never arrive. decoderFailed was declared, cleared in two places and read in five as the give-up signal, but never once set: the path was dead code. Both decoder variants now carry kAvailable and feedElementaryStream raises the flag. Both demux entry points include it in the completeExternalWait predicate, without which a thread already parked -- the usual case, since the first GetPicture precedes the first demux -- is never woken. 2. Nothing could report end of stream to a game that streams with plain sceCdRead. The two available signals are a program-end code in the PSS and currentCdStreamEofSeen, and the latter is only ever raised from sceCdStPause /StRead/StStart/StStop. DQ8 contains no sceCdSt* function at all and its .MVI files carry no program-end code -- they end in 0xFF sector padding -- so the movie decoded all 270 pictures and then hung forever. sawSequenceEnd already tracked the video sequence-end code 00 00 01 B7 but was read only by a debug log; it now drives the wait predicate and the guest's own end flag. The flag itself is inner[0x00], reached through mpeg[0x40]. The game's sceMpegIsEnd equivalent is three instructions -- return **(mpeg+0x40) -- and the stub wrote every neighbouring field but that one. 3. mpeg[0x08] gates the caller's VRAM upload rather than counting pictures. DQ8 uploads its movie buffers only while it reads zero, so writing a running counter there stopped the movie updating after the very first frame. It now reports whether this call produced a picture. Also stop sceMpegReset carrying streamEnded forward unless the sceCdSt* producer is actually driving. appendPssBytes refuses new data while that flag is set until cdStreamGeneration advances, and only notifyMpegCdStreamStart advances it -- so for a plain-sceCdRead game the first ended stream wedged every later movie. --- ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp | 49 +++++++++++++++++++++-- 1 file changed, 45 insertions(+), 4 deletions(-) diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp index 44eaef4d9..4e2abd912 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp @@ -61,6 +61,8 @@ namespace ps2_stubs class MpegFfmpegDecoder { public: + static constexpr bool kAvailable = true; + MpegFfmpegDecoder() = default; ~MpegFfmpegDecoder() @@ -448,6 +450,8 @@ namespace ps2_stubs class MpegFfmpegDecoder { public: + static constexpr bool kAvailable = false; + bool feed(const uint8_t *, size_t, std::deque &, int64_t = -1, int64_t = -1) { static bool s_warnedNoFfmpeg = false; @@ -1038,6 +1042,16 @@ namespace ps2_stubs } playback.sawInput = true; + + // Without a video decoder no frame can ever arrive, so parking + // sceMpegGetPicture waiters would hang the movie forever. Mark the + // stream failed instead and let them return empty-handed. + if constexpr (!MpegFfmpegDecoder::kAvailable) + { + playback.decoderFailed = true; + return; + } + updateMpegPictureTiming(playback, data, size); if (playback.waitingForVideoSequenceHeader) { @@ -2090,6 +2104,7 @@ namespace ps2_stubs uint32_t traceIdx = 0u; bool eofChanged = false; bool backpressured = false; + bool decoderFailed = false; { std::lock_guard lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); @@ -2101,6 +2116,7 @@ namespace ps2_stubs recordCdStreamBytesDemuxedUnlocked(consumed, completedMpegIds, eofChanged); } decodedCount = playback.decodedFrames.size(); + decoderFailed = playback.decoderFailed; traceIdx = g_mpeg_stub_state.demuxPssTraceCount++; } @@ -2116,7 +2132,7 @@ namespace ps2_stubs return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); - if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted) + if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted || decoderFailed) { runtime->eeScheduler().completeExternalWait(kMpegPictureWaitType, mpegAddr, KE_OK); } @@ -2173,6 +2189,7 @@ namespace ps2_stubs uint32_t traceIdx = 0u; bool eofChanged = false; bool backpressured = false; + bool decoderFailed = false; { std::lock_guard lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); @@ -2192,6 +2209,7 @@ namespace ps2_stubs recordCdStreamBytesDemuxedUnlocked(consumed, completedMpegIds, eofChanged); } decodedCount = playback.decodedFrames.size(); + decoderFailed = playback.decoderFailed; traceIdx = g_mpeg_stub_state.demuxRingTraceCount++; } @@ -2209,7 +2227,7 @@ namespace ps2_stubs return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); - if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted) + if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted || decoderFailed) { runtime->eeScheduler().completeExternalWait(kMpegPictureWaitType, mpegAddr, KE_OK); } @@ -2288,13 +2306,18 @@ namespace ps2_stubs uint32_t height = kStubMovieHeight; uint32_t frameCount = 0u; bool haveFrame = false; + bool movieEnded = false; MpegDecodedFrame frame; { std::unique_lock lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); + // sawSequenceEnd is the only end-of-video signal a game streaming + // with plain sceCdRead can raise: the CD-stream EOF path belongs to + // sceCdSt*, and a PSS need not carry a program-end code. if (playback.decodedFrames.empty() && !g_mpeg_stub_state.currentCdStreamEofSeen && !playback.streamEnded && + !playback.sawSequenceEnd && !playback.decoderFailed) { if (g_mpeg_stub_state.getPictureWaitTraceCount < 32u) @@ -2387,17 +2410,30 @@ namespace ps2_stubs height = playback.height; frameCount = playback.picturesServed; } + + // Not on the call that serves the final frame -- the caller would + // tear the movie down before uploading it. + movieEnded = !haveFrame && playback.decodedFrames.empty() && + (playback.sawSequenceEnd || playback.streamEnded || + playback.decoderFailed || g_mpeg_stub_state.currentCdStreamEofSeen); } mpegGuestWrite32(rdram, mpegAddr + 0x00u, width); mpegGuestWrite32(rdram, mpegAddr + 0x04u, height); - mpegGuestWrite32(rdram, mpegAddr + 0x08u, frameCount); + // +0x08 gates the caller's VRAM upload: DQ8 uploads its movie buffers + // only while this reads zero, so a running picture counter here stops + // the movie updating after the very first frame. + mpegGuestWrite32(rdram, mpegAddr + 0x08u, haveFrame ? 0u : 1u); if (uint8_t *base = getMemPtr(rdram, mpegAddr)) { const uint32_t iVar1 = *reinterpret_cast(base + 0x40); if (uint8_t *inner = getMemPtr(rdram, iVar1)) { + // inner[0x00] is the end-of-stream flag the game polls; its + // sceMpegIsEnd equivalent is just `return **(mpeg+0x40)`. + // Without this the movie plays out and never terminates. + *reinterpret_cast(inner + 0x00) = movieEnded ? 1u : 0u; *reinterpret_cast(inner + 0xb0) = 1; *reinterpret_cast(inner + 0xd8) = (getRegU32(ctx, 5) & 0x0FFFFFFFu) | 0x20000000u; *reinterpret_cast(inner + 0xe4) = getRegU32(ctx, 6); @@ -2507,7 +2543,12 @@ namespace ps2_stubs std::lock_guard lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(param_1); MpegPlaybackState resetState = makeFreshPlaybackStatePreservingConfig(playback); - if (playback.streamEnded || playback.decoderFailed) + // Only meaningful while sceCdSt* drives the stream: appendPssBytes + // clears a carried streamEnded when the generation advances, and + // only notifyMpegCdStreamStart advances it. Carrying it for a game + // that streams with plain sceCdRead wedges every later movie. + const bool cdStreamDriven = g_mpeg_stub_state.cdStreamBytesProduced != 0u; + if (cdStreamDriven && (playback.streamEnded || playback.decoderFailed)) { resetState.sawInput = true; resetState.streamEnded = true; From 910a1182c40fbce41f580fab0d14a7b9d93fa2ff Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Thu, 20 Aug 2026 17:12:35 +0200 Subject: [PATCH 20/68] Recycle EE invocation stacks when a thread record goes away m_invocationStackTops keys a 16 KiB stack per (thread, depth) and is only ever find()-ed and emplace()-d, never erased. reserveAsyncCallbackStack behind it is a bump allocator that walks m_asyncCallbackStackTop downwards with no free path. So every guest thread that ever takes one invocation holds its stacks forever, and the pool -- 0x01F00000 to 0x01F40000, exactly 16 slots -- only ever drains. DQ8 starts a fresh thread per movie and deletes the old one, so it hit the wall on the second: "EE invocation stack space exhausted" the moment TAKA_N.MVI opened. Instrumenting the allocation showed 14 of 16 slots held by seven live threads before the second movie was even a factor, thread 12 alone holding four at depths 0 through 3. releaseInvocationStacks returns a thread's stacks to a free list when its record is erased, from both destruction paths, and invocationStackTop pops that list before falling back to the pool. reset() returns them too: it clears m_threads and restarts ids at kFirstThreadId, which would otherwise strand every existing entry permanently. Recycling at record-erase time is safe. deleteThread requires the thread to be Dormant already, and exitCurrent throws EeDispatcherTransfer immediately afterwards, so unwinding completes before another invocation can claim the slot. terminateThread deliberately does not release -- it only makes the thread dormant and the id can be restarted. Measured on DQ8: delete thread 9 and delete thread 13 both recycle, thread 13 reuses the freed slots, and the allocation that threw on every prior attempt succeeds. Exhaustion count zero across a full boot. --- ps2xRuntime/include/runtime/ee_scheduler.h | 4 +++ ps2xRuntime/src/lib/Kernel/EeScheduler.cpp | 40 +++++++++++++++++++++- 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/ps2xRuntime/include/runtime/ee_scheduler.h b/ps2xRuntime/include/runtime/ee_scheduler.h index ad937be85..cb0c1d7f2 100644 --- a/ps2xRuntime/include/runtime/ee_scheduler.h +++ b/ps2xRuntime/include/runtime/ee_scheduler.h @@ -315,6 +315,7 @@ class EeScheduler [[noreturn]] void invokeCurrentSequence(std::vector invocations); [[nodiscard]] bool hasInvocation(GuestInvocationKind kind, uint64_t tag) const; [[nodiscard]] uint32_t invocationStackTop(); + void releaseInvocationStacks(int threadId); int addIrqHandler(bool dmac, uint32_t cause, @@ -442,6 +443,9 @@ class EeScheduler uint32_t m_gsVSyncCallbackGp = 0; uint32_t m_gsVSyncCallbackSp = 0; std::unordered_map m_invocationStackTops; + // Freed by releaseInvocationStacks() when a thread record goes away; the + // underlying pool only bumps down and cannot hand memory back itself. + std::vector m_freeInvocationStacks; std::atomic m_nextDeadlineCycle{0}; mutable std::mutex m_snapshotMutex; diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index 3ecc929d1..e496576a1 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -87,6 +87,13 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) m_executorThread = std::this_thread::get_id(); m_rdram = rdram; m_readyQueues = {}; + // Thread ids restart here, so keyed stacks would otherwise be stranded: + // the pool below cannot reclaim them either. + for (const auto &[key, top] : m_invocationStackTops) + { + m_freeInvocationStacks.push_back(top); + } + m_invocationStackTops.clear(); m_threads.clear(); m_semaphores.clear(); m_eventFlags.clear(); @@ -491,6 +498,7 @@ int EeScheduler::deleteThread(int id, uint32_t &ownedStack) { ownedStack = it->second.stack; } + releaseInvocationStacks(id); m_threads.erase(it); publishSnapshot(); return KE_OK; @@ -539,6 +547,7 @@ int EeScheduler::startThread(int id, uint32_t arg, const R5900Context &caller, b m_currentThreadId = 0; if (deleteThreadRecord && id != kMainThreadId) { + releaseInvocationStacks(id); m_threads.erase(id); } if (ownedStack != 0u) @@ -1191,7 +1200,16 @@ uint32_t EeScheduler::invocationStackTop() return existing->second; } constexpr uint32_t kInvocationStackSize = 0x4000u; - const uint32_t top = m_runtime.reserveAsyncCallbackStack(kInvocationStackSize, 16u); + uint32_t top = 0u; + if (!m_freeInvocationStacks.empty()) + { + top = m_freeInvocationStacks.back(); + m_freeInvocationStacks.pop_back(); + } + else + { + top = m_runtime.reserveAsyncCallbackStack(kInvocationStackSize, 16u); + } if (top == 0u) { throw std::runtime_error("EE invocation stack space exhausted"); @@ -1200,6 +1218,26 @@ uint32_t EeScheduler::invocationStackTop() return top; } +// Stacks are keyed per (thread, depth) and the pool only bumps down, so without +// this every thread that ever took an invocation holds 16 KiB forever -- DQ8 +// starts one thread per movie and drained all 16 slots on the second one. +void EeScheduler::releaseInvocationStacks(int threadId) +{ + const uint64_t prefix = static_cast(static_cast(threadId)) << 32u; + for (auto it = m_invocationStackTops.begin(); it != m_invocationStackTops.end();) + { + if ((it->first & 0xFFFFFFFF00000000ull) == prefix) + { + m_freeInvocationStacks.push_back(it->second); + it = m_invocationStackTops.erase(it); + } + else + { + ++it; + } + } +} + int EeScheduler::addIrqHandler(bool dmac, uint32_t cause, uint32_t handler, From 266e367548b315fec164ad2c109618f665c9e854 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Sat, 22 Aug 2026 17:34:01 +0200 Subject: [PATCH 21/68] Make presentation pitch-explicit, and let a backend present natively PresentationFrame assumed the software backend's fixed 640-wide staging buffer: callers de-strided by that constant, so a backend returning tightly packed rows -- or rows at any other width -- was misread as skewed. Adds rowPitchBytes, and a GSPresentationMode distinguishing host pixels from a backend that presented through its own swapchain and has no pixel payload to copy. GS::copyLatchedHostPresentationFrame now uses the frame's own pitch and refuses one narrower than its width rather than reading past the end. Needed by a hardware backend, which presents at its own internal resolution: a 4x render target hands back 2048x1792 rows, not 640-strided ones. --- ps2xRuntime/include/runtime/gs/gs_frontend.h | 1 + ps2xRuntime/include/runtime/gs/gs_types.h | 37 +++++++++++++++++- ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp | 1 + ps2xRuntime/src/lib/gs/gs_frontend.cpp | 40 +++++++++++++++----- 4 files changed, 69 insertions(+), 10 deletions(-) diff --git a/ps2xRuntime/include/runtime/gs/gs_frontend.h b/ps2xRuntime/include/runtime/gs/gs_frontend.h index 4c3f6410a..b754133d5 100644 --- a/ps2xRuntime/include/runtime/gs/gs_frontend.h +++ b/ps2xRuntime/include/runtime/gs/gs_frontend.h @@ -225,6 +225,7 @@ class GS std::vector m_hostPresentationFrame; uint32_t m_hostPresentationWidth = 0; uint32_t m_hostPresentationHeight = 0; + uint32_t m_hostPresentationRowPitchBytes = 0; uint32_t m_hostPresentationDisplayFbp = 0; uint32_t m_hostPresentationSourceFbp = 0; bool m_hostPresentationUsedPreferred = false; diff --git a/ps2xRuntime/include/runtime/gs/gs_types.h b/ps2xRuntime/include/runtime/gs/gs_types.h index 836a3ff49..0e0b79b1d 100644 --- a/ps2xRuntime/include/runtime/gs/gs_types.h +++ b/ps2xRuntime/include/runtime/gs/gs_types.h @@ -3,6 +3,7 @@ #include #include #include +#include #include enum GSPrimType : uint8_t @@ -290,18 +291,52 @@ struct GSPresentationRequest bool hasPreferredSource = false; }; +enum class GSPresentationMode : uint8_t +{ + // The backend returned host-readable RGBA8 pixels. rowPitchBytes describes + // their layout and may be wider than width * 4. + HostPixels, + // The backend presented through its own native swapchain. There is no host + // pixel payload to copy through the legacy presentation path. + BackendNative, +}; + struct PresentationFrame { std::vector pixels; uint32_t width = 0; uint32_t height = 0; + uint32_t rowPitchBytes = 0; uint32_t displayFbp = 0; uint32_t sourceFbp = 0; bool usedPreferred = false; + GSPresentationMode mode = GSPresentationMode::HostPixels; + + bool HasHostPixels() const + { + if (mode != GSPresentationMode::HostPixels || width == 0u || height == 0u) + return false; + + if (static_cast(width) > std::numeric_limits::max() / 4u) + return false; + const size_t packedRowBytes = static_cast(width) * 4u; + const size_t sourceRowBytes = rowPitchBytes != 0u ? rowPitchBytes : packedRowBytes; + if (sourceRowBytes < packedRowBytes) + return false; + + if (height > 1u && + sourceRowBytes > (std::numeric_limits::max() - packedRowBytes) / + static_cast(height - 1u)) + return false; + const size_t requiredBytes = sourceRowBytes * static_cast(height - 1u) + packedRowBytes; + return pixels.size() >= requiredBytes; + } explicit operator bool() const { - return !pixels.empty() && width != 0u && height != 0u; + if (width == 0u || height == 0u) + return false; + return mode == GSPresentationMode::BackendNative || HasHostPixels(); } }; diff --git a/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp b/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp index 329f3da01..15ef50bcd 100644 --- a/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp @@ -1759,6 +1759,7 @@ PresentationFrame GSCpuBackend::Present(const GSPresentationRequest &request) PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationRequest &request) { PresentationFrame result{}; + result.rowPitchBytes = kHostFrameWidth * 4u; const GSPmodeState pmode = decodePmode(request.pmode); const GSSmode2State smode2 = decodeSMode2(request.smode2); // SMODE2 FFMD=1 (FRAME) reads a half-height buffer whole once per field, so the diff --git a/ps2xRuntime/src/lib/gs/gs_frontend.cpp b/ps2xRuntime/src/lib/gs/gs_frontend.cpp index fd5a90d28..645199a4c 100644 --- a/ps2xRuntime/src/lib/gs/gs_frontend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_frontend.cpp @@ -8,12 +8,11 @@ #include #include #include +#include #include namespace { - static constexpr uint32_t kHostFrameWidth = 640u; - GSPrimReg decodePrimRegister(uint64_t value) { GSPrimReg prim{}; @@ -170,6 +169,7 @@ void GS::reset() m_hostPresentationFrame.clear(); m_hostPresentationWidth = 0u; m_hostPresentationHeight = 0u; + m_hostPresentationRowPitchBytes = 0u; m_hostPresentationDisplayFbp = 0u; m_hostPresentationSourceFbp = 0u; m_hostPresentationUsedPreferred = false; @@ -535,6 +535,7 @@ void GS::latchHostPresentationFrame() m_hostPresentationFrame.clear(); m_hasHostPresentationFrame = false; m_hostPresentationWidth = m_hostPresentationHeight = 0u; + m_hostPresentationRowPitchBytes = 0u; return; } request = buildPresentationRequestUnlocked(); @@ -551,24 +552,32 @@ void GS::latchHostPresentationFrame() } } - const bool hasFrame = static_cast(frame); + const bool presented = static_cast(frame); + const bool hasHostFrame = presented && frame.mode == GSPresentationMode::HostPixels; const uint32_t displayFbp = frame.displayFbp; const uint32_t sourceFbp = frame.sourceFbp; const uint32_t width = frame.width; const uint32_t height = frame.height; + uint32_t rowPitchBytes = frame.rowPitchBytes; + if (rowPitchBytes == 0u && width <= std::numeric_limits::max() / 4u) + rowPitchBytes = width * 4u; const bool usedPreferred = frame.usedPreferred; { std::lock_guard presentationLock(m_presentationMutex); - m_hostPresentationFrame = std::move(frame.pixels); - m_hostPresentationWidth = width; - m_hostPresentationHeight = height; + if (hasHostFrame) + m_hostPresentationFrame = std::move(frame.pixels); + else + m_hostPresentationFrame.clear(); + m_hostPresentationWidth = hasHostFrame ? width : 0u; + m_hostPresentationHeight = hasHostFrame ? height : 0u; + m_hostPresentationRowPitchBytes = hasHostFrame ? rowPitchBytes : 0u; m_hostPresentationDisplayFbp = displayFbp; m_hostPresentationSourceFbp = sourceFbp; m_hostPresentationUsedPreferred = usedPreferred; - m_hasHostPresentationFrame = hasFrame; + m_hasHostPresentationFrame = hasHostFrame; } - if (hasFrame) + if (presented) { std::lock_guard lock(m_stateMutex); recordPresentDebugEventUnlocked(displayFbp, sourceFbp, width, height, usedPreferred); @@ -610,7 +619,20 @@ bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, outPixels.resize(packedRowBytes * static_cast(outHeight)); if (outWidth != 0u && outHeight != 0u) { - const size_t sourceRowBytes = static_cast(kHostFrameWidth) * 4u; + const size_t sourceRowBytes = m_hostPresentationRowPitchBytes; + if (sourceRowBytes < packedRowBytes) + { + outPixels.clear(); + outWidth = 0u; + outHeight = 0u; + if (outDisplayFbp) + *outDisplayFbp = 0u; + if (outSourceFbp) + *outSourceFbp = 0u; + if (outUsedPreferred) + *outUsedPreferred = false; + return false; + } for (uint32_t y = 0; y < outHeight; ++y) { const size_t srcOffset = static_cast(y) * sourceRowBytes; From 438d05edb3e988f102dd5dd86a2e3641fc173759 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Sat, 22 Aug 2026 21:56:21 +0200 Subject: [PATCH 22/68] Let a raster backend own the window and present itself PresentationFrame already had a BackendNative mode; nothing could reach it, because every frame went out through raylib. The runtime now takes an external presenter: with one installed it opens no window, creates no framebuffer texture, and does no drawing. It still latches each frame -- that is the call that reaches the backend's Present() -- and then hands control to the presenter to service its window. Between frames it sleeps briefly instead of spinning. raylib's frame limiter used to block in EndDrawing(); without a replacement the loop burns a core the EE thread wants. Audio moves out of the windowed branch. It has no window and an external presenter does not replace it, so initialising it only in the raylib path left the audio backend permanently not-ready. ps2PadPublishHostState() lets a host that is not raylib feed the same latching path, so press edges between two guest polls still register. --- ps2xRuntime/include/ps2_runtime.h | 13 ++++ ps2xRuntime/include/runtime/ps2_pad_host.h | 7 ++ ps2xRuntime/src/lib/gs/gs_frontend.cpp | 23 ++++++ ps2xRuntime/src/lib/ps2_pad.cpp | 35 +++++++++ ps2xRuntime/src/lib/ps2_runtime.cpp | 83 ++++++++++++++++++---- 5 files changed, 147 insertions(+), 14 deletions(-) diff --git a/ps2xRuntime/include/ps2_runtime.h b/ps2xRuntime/include/ps2_runtime.h index 15b106c98..44c10b4df 100644 --- a/ps2xRuntime/include/ps2_runtime.h +++ b/ps2xRuntime/include/ps2_runtime.h @@ -319,6 +319,17 @@ class PS2Runtime DebugUiCallback shutdownCallback, void *userData); + // A host that owns its own window and shows frames itself. When one is + // installed the runtime opens no window of its own and does no drawing: + // the raster backend has already presented by the time Present() returns, + // and this is called once per frame to service the window. Returning false + // asks the runtime to stop, the way a close button would. + // + // Set before initialize(), which is where the built-in window is created. + using FramePumpCallback = bool (*)(void *userData); + void setExternalPresenter(FramePumpCallback pump, void *userData); + [[nodiscard]] bool hasExternalPresenter() const { return m_framePump != nullptr; } + using RecompiledFunction = void (*)(uint8_t *, R5900Context *, PS2Runtime *); enum class GuestBranchKind @@ -553,6 +564,8 @@ class PS2Runtime std::atomic m_missingFunctionPolicy{static_cast(MissingFunctionPolicy::ContinueToTarget)}; std::atomic m_missingFunctionReported{false}; std::atomic m_stopRequested{false}; + FramePumpCallback m_framePump = nullptr; + void *m_framePumpUserData = nullptr; DebugUiCallback m_debugUiInitCallback = nullptr; DebugUiCallback m_debugUiDrawCallback = nullptr; DebugUiCallback m_debugUiShutdownCallback = nullptr; diff --git a/ps2xRuntime/include/runtime/ps2_pad_host.h b/ps2xRuntime/include/runtime/ps2_pad_host.h index b7652de89..74ed9edbf 100644 --- a/ps2xRuntime/include/runtime/ps2_pad_host.h +++ b/ps2xRuntime/include/runtime/ps2_pad_host.h @@ -1,10 +1,17 @@ #ifndef PS2_PAD_HOST_H #define PS2_PAD_HOST_H +#include + // Render-thread half of the pad backend. Samples raylib input once per presented // frame and latches press edges so a tap between two guest polls still registers. // Declared here rather than on PSPadBackend so the recompiled corpus, which // includes ps2_pad.h through ps2_runtime.h, does not rebuild for input changes. void ps2PadPollHost(); +// The same latching, driven by a host that is not raylib. A backend owning its +// own window samples input itself and publishes it here, in the pad report's +// button encoding, with `pressed` carrying edges seen since the last call. +void ps2PadPublishHostState(uint32_t held, uint32_t pressed, uint32_t sticks); + #endif diff --git a/ps2xRuntime/src/lib/gs/gs_frontend.cpp b/ps2xRuntime/src/lib/gs/gs_frontend.cpp index 645199a4c..fc27b7be1 100644 --- a/ps2xRuntime/src/lib/gs/gs_frontend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_frontend.cpp @@ -4,6 +4,7 @@ #include "runtime/ps2_memory.h" #include #include +#include #include #include #include @@ -660,8 +661,30 @@ bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, return true; } +// Debug accounting: total time decoding GIF packets, and how many arrived. +// A backend can subtract its own time from this to see what packet decoding +// costs on its own, which is the difference between "the renderer is slow" and +// "the game is sending an enormous number of tiny packets". +std::atomic g_gsFrontendPacketNanos{0}; +std::atomic g_gsFrontendPacketCount{0}; + void GS::processGIFPacket(const uint8_t *data, uint32_t sizeBytes) { + const auto packetStart = std::chrono::steady_clock::now(); + struct PacketTimer + { + std::chrono::steady_clock::time_point start; + ~PacketTimer() + { + g_gsFrontendPacketNanos.fetch_add( + static_cast(std::chrono::duration_cast( + std::chrono::steady_clock::now() - start) + .count()), + std::memory_order_relaxed); + g_gsFrontendPacketCount.fetch_add(1, std::memory_order_relaxed); + } + } packetTimer{packetStart}; + std::lock_guard lock(m_stateMutex); if (!data || sizeBytes < 16 || !m_backend) return; diff --git a/ps2xRuntime/src/lib/ps2_pad.cpp b/ps2xRuntime/src/lib/ps2_pad.cpp index e5a26e5ca..db37b887e 100644 --- a/ps2xRuntime/src/lib/ps2_pad.cpp +++ b/ps2xRuntime/src/lib/ps2_pad.cpp @@ -165,6 +165,41 @@ namespace } } +namespace +{ + // Shared tail of both host paths: publish held state and latch edges long + // enough that a tap between two guest polls is not lost. + void publishHostState(uint32_t held, uint32_t pressed, uint32_t sticks) + { + g_held.store(held); + g_sticks.store(sticks); + + const uint64_t now = nowMs(); + if (pressed != 0u) + { + if (g_latched.fetch_or(pressed) == 0u) + { + g_latchStampMs.store(now); + } + } + else + { + const uint32_t stale = g_latched.load(); + if (stale != 0u && (now - g_latchStampMs.load()) > kLatchTimeoutMs) + { + g_latched.fetch_and(~stale); + } + } + + g_hostPolled.store(true); + } +} + +void ps2PadPublishHostState(uint32_t held, uint32_t pressed, uint32_t sticks) +{ + publishHostState(held, pressed, sticks); +} + void ps2PadPollHost() { if (!IsWindowReady()) diff --git a/ps2xRuntime/src/lib/ps2_runtime.cpp b/ps2xRuntime/src/lib/ps2_runtime.cpp index 345a30b51..f50afc0cf 100644 --- a/ps2xRuntime/src/lib/ps2_runtime.cpp +++ b/ps2xRuntime/src/lib/ps2_runtime.cpp @@ -522,6 +522,12 @@ PS2Runtime::PS2Runtime() m_asyncCallbackStackTop = PS2_RAM_SIZE; } +void PS2Runtime::setExternalPresenter(FramePumpCallback pump, void *userData) +{ + m_framePump = pump; + m_framePumpUserData = userData; +} + void PS2Runtime::setDebugUiCallbacks(DebugUiCallback initCallback, DebugUiCallback drawCallback, DebugUiCallback shutdownCallback, @@ -721,18 +727,29 @@ bool PS2Runtime::initialize(const char *title) return false; } #endif + // An external presenter owns the window, so opening one here would put + // a second, empty window on screen and pay for its swap every frame. + if (!hasExternalPresenter()) + { #if defined(PLATFORM_VITA) - InitWindow(HOST_WINDOW_WIDTH, HOST_WINDOW_HEIGHT, title); // raylib vita does not support audio + InitWindow(HOST_WINDOW_WIDTH, HOST_WINDOW_HEIGHT, title); // raylib vita does not support audio #else - SetConfigFlags(FLAG_WINDOW_RESIZABLE); - InitWindow(HOST_WINDOW_WIDTH, HOST_WINDOW_HEIGHT, title); - // Escape is bound to Circle; without this raylib also closes on it. - SetExitKey(KEY_NULL); - applyDefaultWindowScale(); + SetConfigFlags(FLAG_WINDOW_RESIZABLE); + InitWindow(HOST_WINDOW_WIDTH, HOST_WINDOW_HEIGHT, title); + // Escape is bound to Circle; without this raylib also closes on it. + SetExitKey(KEY_NULL); + applyDefaultWindowScale(); +#endif + SetTargetFPS(60); + } +#if !defined(PLATFORM_VITA) + // Audio is not presentation: it has no window and an external presenter + // does not replace it. Initialising it only in the built-in path left + // the audio backend permanently not-ready, which anything waiting on + // sound never recovers from. InitAudioDevice(); m_audioBackend.setAudioReady(IsAudioDeviceReady()); #endif - SetTargetFPS(60); if (m_debugUiInitCallback) { m_debugUiInitCallback(*this, m_debugUiUserData); @@ -2414,12 +2431,17 @@ void PS2Runtime::run() RUNTIME_LOG("Starting execution at address 0x" << std::hex << m_cpuContext.pc << std::dec); - // A blank image to use as a framebuffer - Image blank = GenImageColor(FB_WIDTH, FB_HEIGHT, BLANK); - Texture2D frameTex = LoadTextureFromImage(blank); - UnloadImage(blank); + // A blank image to use as a framebuffer. Not needed when something else + // owns the window -- there is no GL context to create a texture in. + Texture2D frameTex{}; int frameTexFilter = TEXTURE_FILTER_POINT; - SetTextureFilter(frameTex, frameTexFilter); + if (!hasExternalPresenter()) + { + Image blank = GenImageColor(FB_WIDTH, FB_HEIGHT, BLANK); + frameTex = LoadTextureFromImage(blank); + UnloadImage(blank); + SetTextureFilter(frameTex, frameTexFilter); + } std::atomic gameThreadFinished{false}; @@ -2478,6 +2500,36 @@ void PS2Runtime::run() << std::endl); } }); + // With an external presenter the backend draws to its own swapchain, + // so the frame still has to be latched -- that is what calls into the + // backend's Present() -- but nothing is uploaded or drawn here. + if (m_framePump) + { + static uint64_t s_lastNativeTick = std::numeric_limits::max(); + const uint64_t currentTick = eeScheduler().currentVSyncTick(); + const bool newFrame = currentTick != s_lastNativeTick; + if (newFrame) + { + gs().latchHostPresentationFrame(); + s_lastNativeTick = currentTick; + } + else + { + // Nothing new to show. Without this the loop spins a core flat + // out waiting for the guest, which is a core the EE thread + // wants; the raylib path got the same effect from its frame + // limiter blocking in EndDrawing(). + std::this_thread::sleep_for(std::chrono::microseconds(250)); + } + if (!m_framePump(m_framePumpUserData)) + { + RUNTIME_LOG("[run] external presenter requested stop"); + requestStop(); + break; + } + continue; + } + uint32_t presentWidth = FB_WIDTH; uint32_t presentHeight = DEFAULT_DISPLAY_HEIGHT; UploadFrame(frameTex, this, presentWidth, presentHeight); @@ -2535,8 +2587,11 @@ void PS2Runtime::run() m_debugUiShutdownCallback(*this, m_debugUiUserData); m_debugUiInitialized = false; } - UnloadTexture(frameTex); - CloseWindow(); + if (!hasExternalPresenter()) + { + UnloadTexture(frameTex); + CloseWindow(); + } RUNTIME_LOG("[run] exiting loop"); } From 3ae580597d6965ede4796e7d1952b0afcf1dcba8 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Sat, 22 Aug 2026 23:43:41 +0200 Subject: [PATCH 23/68] Stop the guest busy-waiting when MPEG demux is backpressured mpegDemuxBackpressured() returns true once the decoder is 8 pictures ahead, and the demux stubs then returned 0 consumed and nothing else. DQ8's producer loop reads that as "not now" and asks again immediately, so the movie thread span: 670,171 calls to sceMpegDemuxPssRing per 30 presented frames -- about 22,300 per frame -- against 0.5 sceMpegGetPicture calls per frame of actual work. That was the movie's frame time. Measured with the graphics side fully accounted for: the GS backend was 6% of wall, GIF packet decode ~0%, and the 896 tile transfers per frame cost almost nothing. Both demux entry points now rotate the ready queue before returning. This is deliberately a yield and not a park: the comment on mpegDemuxBackpressured warns that parking the producer can leave a consumer asleep with nobody to wake it, which is real -- Code Veronica wakes its video thread before every demux call. Rotating the queue lets the consumer run and comes back, so the spin becomes a scheduling point without changing who wakes whom. DQ8's attract movie goes from 0.8 fps to 9.1, and demux calls per frame from ~22,300 to ~1,025. Still 1,025 spins per frame, so this is a large improvement rather than a cure. The remaining fix is a real producer/consumer handshake -- park the producer on space-available and wake it from sceMpegGetPicture -- which needs care for exactly the reason the yield was chosen here. Also adds timers on sceMpegGetPicture, writeDecodedFrameToGuest and the demux entry points, and on the GIF frontend's packet and native-image-upload paths. None of this was visible before; the demux call count is what identified the bug, and it was only findable because the graphics side had already been measured and eliminated. --- ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp | 49 +++++++++++++++++++++++ ps2xRuntime/src/lib/gs/gs_frontend.cpp | 18 ++++++++- ps2xRuntime/src/lib/ps2_runtime.cpp | 12 +++++- 3 files changed, 76 insertions(+), 3 deletions(-) diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp index 4e2abd912..d145eb86e 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp @@ -16,11 +16,44 @@ extern "C" } #endif +#include +#include #include #include #include "Syscalls/Helpers/State.h" +// Debug accounting for the movie path, read by whichever backend reports stats. +// A movie frame arrives as ~900 tile transfers, but those turned out to cost +// almost nothing; these say what the rest of the frame is doing. +std::atomic g_mpegGetPictureNanos{0}; +std::atomic g_mpegGetPictureCount{0}; +std::atomic g_mpegWriteFrameNanos{0}; +std::atomic g_mpegWriteFrameCount{0}; +std::atomic g_mpegDemuxNanos{0}; +std::atomic g_mpegDemuxCount{0}; + +namespace +{ + struct MpegScopedTimer + { + std::atomic &sink; + std::atomic &counter; + std::chrono::steady_clock::time_point start; + MpegScopedTimer(std::atomic &nanos, std::atomic &count) + : sink(nanos), counter(count), start(std::chrono::steady_clock::now()) {} + ~MpegScopedTimer() + { + sink.fetch_add(static_cast( + std::chrono::duration_cast( + std::chrono::steady_clock::now() - start) + .count()), + std::memory_order_relaxed); + counter.fetch_add(1, std::memory_order_relaxed); + } + }; +} + namespace ps2_stubs { namespace @@ -1706,6 +1739,7 @@ namespace ps2_stubs { return; } + MpegScopedTimer timer(g_mpegWriteFrameNanos, g_mpegWriteFrameCount); const uint32_t width = static_cast(frame.width); const uint32_t height = static_cast(frame.height); @@ -2128,6 +2162,9 @@ namespace ps2_stubs std::cerr << "[MPEG:DemuxPss:BACKPRESSURE] mpeg=0x" << std::hex << mpegAddr << std::dec << " decoded=" << decodedCount << std::endl; }); } + // Same reasoning as the ring variant: yield so the consumer can + // drain a picture, rather than letting the producer spin. + runtime->eeScheduler().rotateReadyQueue(0, false); setReturnS32(ctx, 0); return; } @@ -2163,6 +2200,7 @@ namespace ps2_stubs void sceMpegDemuxPssRing(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { + MpegScopedTimer timer(g_mpegDemuxNanos, g_mpegDemuxCount); static std::atomic s_demuxRingEntryCount{0u}; const uint32_t entryIdx = s_demuxRingEntryCount.fetch_add(1u, std::memory_order_relaxed); if (entryIdx < 4u) @@ -2223,6 +2261,16 @@ namespace ps2_stubs << " avail=" << availableBytes << std::endl; }); } + // Returning 0 consumed tells the producer "not now", and its loop + // asks again immediately -- DQ8 called this 22,000 times per + // presented frame, which is where the movie's frame time went. + // + // Yield rather than park. Parking is what the comment on + // mpegDemuxBackpressured warns against: the consumer may be asleep + // waiting for this very thread to wake it. Rotating the ready queue + // just lets the consumer run and comes back, so the spin becomes a + // scheduling point without changing who wakes whom. + runtime->eeScheduler().rotateReadyQueue(0, false); setReturnS32(ctx, 0); return; } @@ -2300,6 +2348,7 @@ namespace ps2_stubs void sceMpegGetPicture(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { + MpegScopedTimer timer(g_mpegGetPictureNanos, g_mpegGetPictureCount); const uint32_t mpegAddr = getRegU32(ctx, 4); const uint32_t imageAddr = getRegU32(ctx, 5); uint32_t width = kStubMovieWidth; diff --git a/ps2xRuntime/src/lib/gs/gs_frontend.cpp b/ps2xRuntime/src/lib/gs/gs_frontend.cpp index fc27b7be1..6d99b04c9 100644 --- a/ps2xRuntime/src/lib/gs/gs_frontend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_frontend.cpp @@ -667,6 +667,11 @@ bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, // "the game is sending an enormous number of tiny packets". std::atomic g_gsFrontendPacketNanos{0}; std::atomic g_gsFrontendPacketCount{0}; +// The same for the native image-upload fast path, which bypasses +// processGIFPacket entirely -- DQ8's movie tiles arrive this way, one DMA +// chain per 16x16 tile, so without a separate counter they are invisible. +std::atomic g_gsUploadNativeNanos{0}; +std::atomic g_gsUploadNativeCount{0}; void GS::processGIFPacket(const uint8_t *data, uint32_t sizeBytes) { @@ -828,8 +833,17 @@ void GS::uploadImageNative(uint64_t bitbltbuf, const uint8_t *data, uint32_t sizeBytes) { - std::lock_guard lock(m_stateMutex); - uploadImageNativeUnlocked(bitbltbuf, trxpos, trxreg, trxdir, data, sizeBytes); + const auto start = std::chrono::steady_clock::now(); + { + std::lock_guard lock(m_stateMutex); + uploadImageNativeUnlocked(bitbltbuf, trxpos, trxreg, trxdir, data, sizeBytes); + } + g_gsUploadNativeNanos.fetch_add( + static_cast(std::chrono::duration_cast( + std::chrono::steady_clock::now() - start) + .count()), + std::memory_order_relaxed); + g_gsUploadNativeCount.fetch_add(1, std::memory_order_relaxed); } void GS::uploadImageNativeUnlocked(uint64_t bitbltbuf, diff --git a/ps2xRuntime/src/lib/ps2_runtime.cpp b/ps2xRuntime/src/lib/ps2_runtime.cpp index f50afc0cf..670587616 100644 --- a/ps2xRuntime/src/lib/ps2_runtime.cpp +++ b/ps2xRuntime/src/lib/ps2_runtime.cpp @@ -2519,7 +2519,17 @@ void PS2Runtime::run() // out waiting for the guest, which is a core the EE thread // wants; the raylib path got the same effect from its frame // limiter blocking in EndDrawing(). - std::this_thread::sleep_for(std::chrono::microseconds(250)); + // + // Tunable because how often this thread wakes changes the + // interleaving the EE thread sees, and the memory-card boot + // race is sensitive to exactly that + // (docs/notes/memory-card-boot-race.md). + static const int idleMicros = [] { + const char *value = std::getenv("DQ8_PRESENT_IDLE_US"); + const int parsed = value ? std::atoi(value) : 0; + return parsed > 0 ? parsed : 1000; + }(); + std::this_thread::sleep_for(std::chrono::microseconds(idleMicros)); } if (!m_framePump(m_framePumpUserData)) { From e3f7eb29b98eb5baf45fdca22ec877973bf63499 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Sun, 23 Aug 2026 00:36:29 +0200 Subject: [PATCH 24/68] Complete the demux yield protocol, and stop crashing on every movie The previous commit called rotateReadyQueue from the demux stub and returned normally. That clears the current thread and requests a reschedule, so the next syscall reached bindMainContextForSyscall with no thread running and tripped its assert -- the game aborted on every movie. transferIfRequested must follow, which is what the RotateThreadReadyQueue syscall does, and the return value has to be set before it because the transfer throws. Verified: 0 aborts across a 700-second run that plays the attract movie through, 582 pictures served. Also tried and reverted: parking the producer on a space-available wait and completing it from sceMpegGetPicture. That deadlocks -- 4 pictures served instead of 658 -- which is exactly what the comment on mpegDemuxBackpressured predicts. Both the attempt and its result are recorded in docs/notes/movie-decode.md so it is not tried a third time. Honest numbers: 0.8 fps before, 1.2-1.7 fps now, demux calls halved from 7.6M to 3.4M over the run. That is an improvement, not a fix. Demux is now 6% of wall and the backend 6%; the other ~88% is still unattributed, and the principled next step -- buffering PSS bytes and decoding lazily so the producer is never refused -- has not been attempted. --- ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp | 26 +++++++++++++++-------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp index d145eb86e..b7114b757 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp @@ -2162,10 +2162,10 @@ namespace ps2_stubs std::cerr << "[MPEG:DemuxPss:BACKPRESSURE] mpeg=0x" << std::hex << mpegAddr << std::dec << " decoded=" << decodedCount << std::endl; }); } - // Same reasoning as the ring variant: yield so the consumer can - // drain a picture, rather than letting the producer spin. - runtime->eeScheduler().rotateReadyQueue(0, false); + // Same reasoning and the same order as the ring variant. setReturnS32(ctx, 0); + runtime->eeScheduler().rotateReadyQueue(0, false); + runtime->eeScheduler().transferIfRequested(false); return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); @@ -2265,13 +2265,21 @@ namespace ps2_stubs // asks again immediately -- DQ8 called this 22,000 times per // presented frame, which is where the movie's frame time went. // - // Yield rather than park. Parking is what the comment on - // mpegDemuxBackpressured warns against: the consumer may be asleep - // waiting for this very thread to wake it. Rotating the ready queue - // just lets the consumer run and comes back, so the spin becomes a - // scheduling point without changing who wakes whom. - runtime->eeScheduler().rotateReadyQueue(0, false); + // Yield, do not park. Parking the producer on a space-available + // wait and waking it from sceMpegGetPicture was tried and + // deadlocks the movie: DQ8 served 4 pictures instead of 658. That + // is exactly what the comment on mpegDemuxBackpressured predicts, + // so it is recorded here rather than left to be rediscovered. + // + // The return value has to be set before the transfer, and the + // transfer is not optional: rotateReadyQueue clears the current + // thread and asks for a reschedule, so returning normally after it + // leaves the scheduler with no running thread and trips + // bindMainContextForSyscall's assert on the next syscall. Same + // order as the RotateThreadReadyQueue syscall. setReturnS32(ctx, 0); + runtime->eeScheduler().rotateReadyQueue(0, false); + runtime->eeScheduler().transferIfRequested(false); return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); From 052915dfd77ff953b9196d15d4e96e8503483d97 Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Sun, 23 Aug 2026 02:41:26 +0200 Subject: [PATCH 25/68] Stop paying for the movie's yields: lazy snapshot, lazy decode, cheap event pump DQ8's movie producer thread is a spin loop, and correctly so -- it feeds audio and waits for its CD ring, ~5,150 iterations per movie frame, which is about what a real 294 MHz EE would manage. Our cost per iteration was the problem. Three measured costs, each removed: publishSnapshot() built and sorted three vectors of kernel state on every one of ~40 scheduler operations, for state only the debug panel and an aggressive-log tick ever read. 45% of EE thread time during a movie. It now builds only when snapshot() has asked; the idle path publishes unconditionally so a quiet scheduler still converges for a poller. The MPEG stub refused demux bytes once 8 pictures were decoded ahead -- 134,563 refusals out of 134,569 calls per 30 frames -- and the guest's loop (remaining -= consumed; bgtz remaining) cannot terminate on a zero count. Demux now queues elementary stream and sceMpegGetPicture pulls decode from it, so lookahead is bounded by buffered bytes rather than decoded pictures: ~25 KB per frame on the wire against ~917 KB decoded. Refusals go to zero and demux calls per frame from 4,551 to 13. flushDecoderIfEnded() has to wait for the queue to drain, or a program-end code drains the decoder while the whole movie is still buffered and every later packet fails with AVERROR_EOF. processPendingEvents() ran after every guest dispatch and took three mutexes, read the clock, walked m_deadlines and heap-allocated a deque to find nothing. 24% of EE thread time, now 6%. Also: transferIfRequested() skips the unwind when the thread it would reschedule is the one already executing, and counters for dispatches, transfers and refusals so the remaining cost stays visible. DQ8_EE_DISPATCH_HISTOGRAM names the guest addresses the executor re-enters most. Logo movie 1.4 -> 3.3 fps. Still not smooth: 30,908 context transfers per movie frame at ~10 us each, where 60 fps needs ~540 ns. That is the exception-based transfer itself, and fixing it means fibers rather than throw/unwind. --- ps2xRuntime/include/runtime/ee_scheduler.h | 14 ++ ps2xRuntime/src/lib/Kernel/EeScheduler.cpp | 188 +++++++++++++++++++-- ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp | 146 +++++++++++++++- 3 files changed, 329 insertions(+), 19 deletions(-) diff --git a/ps2xRuntime/include/runtime/ee_scheduler.h b/ps2xRuntime/include/runtime/ee_scheduler.h index cb0c1d7f2..d6dce5e68 100644 --- a/ps2xRuntime/include/runtime/ee_scheduler.h +++ b/ps2xRuntime/include/runtime/ee_scheduler.h @@ -354,7 +354,11 @@ class EeScheduler void bindMainContextForSyscall(R5900Context &ctx, uint8_t *rdram); [[nodiscard]] EeKernelSnapshot snapshot() const; + // Cheap unless snapshot() has asked for a refresh since the last build. void publishSnapshot(); + // Builds unconditionally; for the idle path, where no later scheduler + // operation is guaranteed to service a pending request. + void publishSnapshotNow(); private: struct ScheduledEvent @@ -372,6 +376,10 @@ class EeScheduler void enqueueReady(GuestThread &thread, bool front = false); void removeReady(GuestThread &thread); [[nodiscard]] GuestThread *selectReady(); + // What selectReady() would return, without dequeuing it. 0 when nothing is ready. + [[nodiscard]] int peekReadyId() const; + // Whether the dispatch loop has work that a same-thread yield must not skip. + [[nodiscard]] bool mustReturnToDispatcher() const noexcept; void makeRunning(GuestThread &thread); void makeDormant(GuestThread &thread); void removeFromWaitObject(GuestThread &thread); @@ -428,6 +436,9 @@ class EeScheduler std::atomic m_stopRequested{false}; std::atomic m_checkpointPending{false}; uint32_t m_debugPublishCountdown = 0u; + // The thread whose C++ stack the executor is standing in, or 0 outside a + // guest dispatch. A yield back onto this thread can skip the unwind. + int m_dispatchedThreadId = 0; mutable std::mutex m_eventMutex; std::condition_variable m_eventCv; @@ -451,4 +462,7 @@ class EeScheduler mutable std::mutex m_snapshotMutex; EeKernelSnapshot m_snapshot; uint64_t m_snapshotSequence = 0; + // Set by snapshot(), cleared by publishSnapshot(). The snapshot is debug + // state, so it is only worth building when someone has asked for it. + mutable std::atomic m_snapshotWanted{false}; }; diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index e496576a1..cbaf8966c 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -4,10 +4,39 @@ #include "ps2_runtime_macros.h" #include +#include #include +#include +#include #include #include #include +#include +#include +#include + +// How many times the executor re-entered guest code. The EE clock advances +// kGuestDispatchCycles per dispatch, so this is what paces the emulated frame. +std::atomic g_eeGuestDispatchCount{0}; +// Thrown EeDispatcherTransfers. Against g_eeGuestDispatchCount this says +// whether the executor is re-entering guest code because a transfer asked it +// to, or because the call chain has to be rebuilt after one. +std::atomic g_eeTransferThrowCount{0}; + +namespace +{ + const uint64_t g_eeCycleScale = [] { + const char *value = std::getenv("DQ8_EE_CYCLE_SCALE"); + const long long parsed = value ? std::atoll(value) : 0; + return parsed > 0 ? static_cast(parsed) : 1ull; + }(); + + const uint64_t g_eeDispatchHistogramInterval = [] { + const char *value = std::getenv("DQ8_EE_DISPATCH_HISTOGRAM"); + const long long parsed = value ? std::atoll(value) : 0; + return parsed > 0 ? static_cast(parsed) : 0ull; + }(); +} namespace { @@ -123,6 +152,7 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) m_stopRequested.store(false, std::memory_order_release); m_checkpointPending.store(false, std::memory_order_release); m_debugPublishCountdown = 0u; + m_dispatchedThreadId = 0; { std::lock_guard lock(m_eventMutex); m_events.clear(); @@ -156,7 +186,7 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) scheduleEvent(m_eeCycle + kVBlankPeriodCycles, std::chrono::steady_clock::now() + kVBlankPeriod, EeEvent{EeEventType::VBlankStart, 0, 0}); - publishSnapshot(); + publishSnapshotNow(); } void EeScheduler::run() @@ -177,7 +207,9 @@ void EeScheduler::run() GuestThread *next = selectReady(); if (!next && m_pendingInvocations.empty()) { - publishSnapshot(); + // About to sleep: nothing after this would service a pending + // snapshot request until the guest runs again. + publishSnapshotNow(); waitForEvent(); continue; } @@ -209,10 +241,13 @@ void EeScheduler::run() running->resumeCompletion = {}; try { + m_dispatchedThreadId = running->id; completion(running->activeContext()); + m_dispatchedThreadId = 0; } catch (const EeDispatcherTransfer &) { + m_dispatchedThreadId = 0; } if (m_currentThreadId == 0) { @@ -297,6 +332,32 @@ void EeScheduler::run() } PS2Runtime::RecompiledFunction function = m_runtime.lookupFunction(context.pc); + g_eeGuestDispatchCount.fetch_add(1u, std::memory_order_relaxed); + // DQ8_EE_DISPATCH_HISTOGRAM=N reports the guest addresses the executor + // re-enters most often, every N dispatches. A spinning guest loop shows + // up here as one address with a huge count. + if (g_eeDispatchHistogramInterval != 0u) + { + static std::unordered_map s_histogram; + static uint64_t s_seen = 0u; + ++s_histogram[context.pc]; + if (++s_seen >= g_eeDispatchHistogramInterval) + { + std::vector> top(s_histogram.begin(), s_histogram.end()); + std::partial_sort(top.begin(), top.begin() + std::min(8u, top.size()), top.end(), + [](const auto &l, const auto &r) { return l.second > r.second; }); + std::fprintf(stderr, "[ee] dispatch histogram over %llu:", + static_cast(s_seen)); + for (size_t i = 0u; i < std::min(8u, top.size()); ++i) + { + std::fprintf(stderr, " 0x%08x=%llu", top[i].first, + static_cast(top[i].second)); + } + std::fprintf(stderr, "\n"); + s_histogram.clear(); + s_seen = 0u; + } + } if (checkpointDue(kGuestDispatchCycles)) { continue; @@ -306,17 +367,21 @@ void EeScheduler::run() { m_insideInterrupt = !running->invocations.empty() && running->invocations.back().kind == GuestInvocationKind::Interrupt; m_guestExecuting.store(true, std::memory_order_release); + m_dispatchedThreadId = running->id; function(m_rdram, &context, &m_runtime); + m_dispatchedThreadId = 0; m_guestExecuting.store(false, std::memory_order_release); m_insideInterrupt = false; } catch (const EeDispatcherTransfer &) { + m_dispatchedThreadId = 0; m_guestExecuting.store(false, std::memory_order_release); m_insideInterrupt = false; } catch (...) { + m_dispatchedThreadId = 0; m_guestExecuting.store(false, std::memory_order_release); m_running.store(false, std::memory_order_release); publishSnapshot(); @@ -400,7 +465,13 @@ bool EeScheduler::checkpointDue(uint32_t cycles) noexcept void EeScheduler::accountCycles(uint32_t cycles) noexcept { - const uint64_t elapsed = std::max(1u, cycles); + // DQ8_EE_CYCLE_SCALE multiplies the charge per guest safe point. The model + // charges per inter-function branch and ignores the instructions between, + // so it undercounts badly against a real 294 MHz EE; a spinning guest loop + // then gets far more iterations per emulated frame than hardware allows. + // Only raises the floor: a frame still cannot retire before its host + // deadline, so this can never run the game faster than real time. + const uint64_t elapsed = std::max(1u, cycles) * g_eeCycleScale; m_eeCycle += elapsed; m_pendingEeTimerInterrupts |= m_runtime.memory().advanceEeTimers(elapsed); if (m_pendingEeTimerInterrupts != 0u) @@ -555,6 +626,7 @@ int EeScheduler::startThread(int id, uint32_t arg, const R5900Context &caller, b m_runtime.guestFree(ownedStack); } publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -841,9 +913,41 @@ void EeScheduler::transferIfRequested(bool interruptSafe) enqueueReady(*self, true); m_currentThreadId = 0; } + + // A yield that lands back on the thread we are already executing does not + // need the C++ stack torn down and rebuilt. Unwinding EeDispatcherTransfer + // through the guest's frames was ~40% of EE time during a movie, because + // DQ8's movie producer yields thousands of times per frame and almost + // always ends up running again immediately. + if (m_dispatchedThreadId != 0 && + m_pendingInvocations.empty() && + peekReadyId() == m_dispatchedThreadId && + !mustReturnToDispatcher()) + { + // Charge what the dispatch loop would have charged, so the time slice + // and the EE timers still advance and this cannot spin forever. + accountCycles(kGuestDispatchCycles); + GuestThread *next = selectReady(); + assert(next != nullptr && next->id == m_dispatchedThreadId); + if (next != nullptr && next->id == m_dispatchedThreadId) + { + next->status = EeThreadStatus::Running; + m_currentThreadId = next->id; + m_rescheduleRequested = false; + m_timeSliceExpired = false; + return; + } + // peekReadyId lied, so put it back and take the slow path. + if (next != nullptr) + { + enqueueReady(*next, true); + } + } + m_rescheduleRequested = false; m_timeSliceExpired = false; publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1147,6 +1251,7 @@ void EeScheduler::queueInvocation(GuestInvocation invocation) invocation.sequence = ++m_invocationSequence; owner->invocations.push_back(std::move(invocation)); publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1166,6 +1271,7 @@ void EeScheduler::queueInvocation(GuestInvocation invocation) owner->invocations.push_back(std::move(*it)); } publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1538,11 +1644,28 @@ void EeScheduler::bindMainContextForSyscall(R5900Context &ctx, uint8_t *rdram) EeKernelSnapshot EeScheduler::snapshot() const { + // Ask the executor to refresh, then return what it published last. One + // scheduler operation of staleness is invisible to a debug panel, and the + // idle path publishes unconditionally so a quiet scheduler still converges. + m_snapshotWanted.store(true, std::memory_order_relaxed); std::lock_guard lock(m_snapshotMutex); return m_snapshot; } void EeScheduler::publishSnapshot() +{ + // Building this walks every thread, semaphore and event flag and sorts all + // three. Called from ~40 scheduler operations it was 45% of EE thread time + // during a movie, for state whose only readers are the debug panel and an + // aggressive-log tick. + if (!m_snapshotWanted.exchange(false, std::memory_order_relaxed)) + { + return; + } + publishSnapshotNow(); +} + +void EeScheduler::publishSnapshotNow() { EeKernelSnapshot next{}; next.sequence = ++m_snapshotSequence; @@ -1684,6 +1807,34 @@ GuestThread *EeScheduler::selectReady() return nullptr; } +int EeScheduler::peekReadyId() const +{ + for (const auto &queue : m_readyQueues) + { + if (!queue.empty()) + { + return queue.front(); + } + } + return 0; +} + +bool EeScheduler::mustReturnToDispatcher() const noexcept +{ + // The same conditions checkpointDue() uses, minus its side effects. + if (m_checkpointPending.load(std::memory_order_acquire) || + m_stopRequested.load(std::memory_order_acquire)) + { + return true; + } + const uint64_t nextEventCycle = m_nextDeadlineCycle.load(std::memory_order_acquire); + if (nextEventCycle != 0u && m_eeCycle >= nextEventCycle) + { + return true; + } + return m_eeCycle >= m_sliceEndCycle; +} + void EeScheduler::makeRunning(GuestThread &item) { assert(m_currentThreadId == 0); @@ -1743,6 +1894,7 @@ void EeScheduler::blockCurrent(EeWaitState wait) self->status = self->suspendCount == 0 ? EeThreadStatus::Waiting : EeThreadStatus::WaitingSuspended; m_currentThreadId = 0; publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1819,14 +1971,20 @@ void EeScheduler::processPendingEvents() dispatchIrq(false, 5u); } - std::deque pending; + // A default-constructed deque still allocates, and this path is almost + // always empty, so only build one when there is something to take. { - std::lock_guard lock(m_eventMutex); - pending.swap(m_events); - } - for (const EeEvent &event : pending) - { - processEvent(event); + std::unique_lock lock(m_eventMutex); + if (!m_events.empty()) + { + std::deque pending; + pending.swap(m_events); + lock.unlock(); + for (const EeEvent &event : pending) + { + processEvent(event); + } + } } { @@ -1841,6 +1999,16 @@ void EeScheduler::processPendingEvents() void EeScheduler::processDueDeadlines() { + // Runs after every guest dispatch, so the common "nothing is due" case must + // not cost a mutex, a clock read and a walk of m_deadlines. m_nextDeadlineCycle + // is the minimum deadlineCycle, or 0 when there are none -- exactly the + // condition the loop below would fail on. + const uint64_t nextDeadline = m_nextDeadlineCycle.load(std::memory_order_acquire); + if (nextDeadline == 0u || m_eeCycle < nextDeadline) + { + return; + } + for (;;) { std::vector due; diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp index b7114b757..4cf554b36 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp @@ -32,6 +32,13 @@ std::atomic g_mpegWriteFrameNanos{0}; std::atomic g_mpegWriteFrameCount{0}; std::atomic g_mpegDemuxNanos{0}; std::atomic g_mpegDemuxCount{0}; +// How many demux calls were refused, and which guest threads drive each side: +// whether the producer and the consumer are the same thread decides whether +// parking the producer is even possible. +std::atomic g_mpegDemuxRefusedCount{0}; +std::atomic g_mpegPendingEsPeakBytes{0}; +std::atomic g_mpegDemuxThreadId{0}; +std::atomic g_mpegGetPictureThreadId{0}; namespace { @@ -521,6 +528,28 @@ namespace ps2_stubs constexpr size_t kMpegTimingScanLimit = 4096u; constexpr size_t kMaxDecodedPicturesAhead = 8u; + // Lookahead is held as compressed elementary stream, not as decoded + // pictures: a 512x448 frame costs ~917 KB decoded against ~25 KB on the + // wire, so the same memory buys ~36x more of it. Refusing a demux call + // is what makes the guest's producer loop spin, and every spin costs a + // guest RotateThreadReadyQueue, so the lookahead wants to be deep. + size_t mpegMaxPendingEsBytes() + { + static const size_t bytes = [] { + const char *value = std::getenv("DQ8_MPEG_ES_BUFFER_MB"); + const long parsed = value ? std::atol(value) : 0; + return (parsed > 0 ? static_cast(parsed) : 32u) * 1024u * 1024u; + }(); + return bytes; + } + + struct MpegPendingEs + { + std::vector data; + int64_t pts90k = -1; + int64_t dts90k = -1; + }; + struct MpegPlaybackState { uint32_t picturesServed = 0u; @@ -538,6 +567,11 @@ namespace ps2_stubs std::vector pssBuffer; std::vector pssGuestAddrs; std::deque decodedFrames; + // Demuxed but not yet decoded. Decode is pulled from here by + // sceMpegGetPicture rather than pushed by the demux call. + std::deque pendingEs; + size_t pendingEsBytes = 0u; + size_t pendingEsPeakBytes = 0u; std::unique_ptr decoder; uint8_t frameRateCode = 0u; uint8_t frameRateExtensionN = 0u; @@ -583,6 +617,21 @@ namespace ps2_stubs std::mutex g_mpeg_stub_mutex; constexpr uint32_t kMpegPictureWaitType = 1u; + + // A refused demux call only has to give the consumer thread a turn -- + // it does not need one context transfer per refusal, and a transfer is + // a thrown EeDispatcherTransfer unwound through the guest's whole call + // stack. Yield on every Nth refusal instead; the spins in between are + // just a stub entry and a mutex. + size_t mpegDemuxYieldInterval() + { + static const size_t interval = [] { + const char *value = std::getenv("DQ8_MPEG_YIELD_EVERY"); + const long parsed = value ? std::atol(value) : 0; + return parsed > 0 ? static_cast(parsed) : 64u; + }(); + return interval; + } MpegStubState g_mpeg_stub_state; // TODO this resolution should follow runtime resolution @@ -1043,7 +1092,10 @@ namespace ps2_stubs void flushDecoderIfEnded(MpegPlaybackState &playback) { - if (playback.streamEnded && playback.decoder) + // Not while bytes are still queued. A program-end code arrives with + // the whole movie still buffered, and draining the decoder there + // makes every packet after it fail with AVERROR_EOF. + if (playback.streamEnded && playback.pendingEs.empty() && playback.decoder) { playback.decoder->flush(playback.decodedFrames); } @@ -1157,6 +1209,48 @@ namespace ps2_stubs flushDecoderIfEnded(playback); } + // Demux hands video payload here instead of straight to the decoder, so + // accepting the guest's bytes costs a memcpy rather than a decode. + void queueElementaryStream(MpegPlaybackState &playback, const uint8_t *data, size_t size, + int64_t pts90k, int64_t dts90k) + { + if (!data || size == 0u) + { + return; + } + playback.sawInput = true; + playback.pendingEs.push_back( + MpegPendingEs{std::vector(data, data + size), pts90k, dts90k}); + playback.pendingEsBytes += size; + playback.pendingEsPeakBytes = std::max(playback.pendingEsPeakBytes, playback.pendingEsBytes); + if (playback.pendingEsPeakBytes > g_mpegPendingEsPeakBytes.load(std::memory_order_relaxed)) + { + g_mpegPendingEsPeakBytes.store(playback.pendingEsPeakBytes, std::memory_order_relaxed); + } + } + + // Decode forward until there is something to show, or the buffer runs + // dry. One chunk is a single PES payload, so several are usually needed + // before the decoder emits a picture. + void decodePendingElementaryStream(MpegPlaybackState &playback, bool drainAll = false) + { + while (!playback.pendingEs.empty() && + (drainAll || playback.decodedFrames.empty())) + { + MpegPendingEs chunk = std::move(playback.pendingEs.front()); + playback.pendingEs.pop_front(); + playback.pendingEsBytes -= std::min(playback.pendingEsBytes, chunk.data.size()); + feedElementaryStream(playback, chunk.data.data(), chunk.data.size(), + chunk.pts90k, chunk.dts90k); + if (playback.decoderFailed) + { + break; + } + } + // The queue emptying is what makes a already-ended stream flushable. + flushDecoderIfEnded(playback); + } + void erasePssPrefix(MpegPlaybackState &playback, size_t count) { std::vector &buffer = playback.pssBuffer; @@ -1378,7 +1472,7 @@ namespace ps2_stubs pes.pts90k, pes.dts90k); } - feedElementaryStream( + queueElementaryStream( playback, buffer.data() + payloadStart, packetEnd - payloadStart, @@ -1419,6 +1513,9 @@ namespace ps2_stubs { std::vector ignoredCallbacks; processPssBuffer(mpegAddr, playback, ignoredCallbacks, true); + // Everything still buffered has to reach the decoder before the + // stream can be called ended, or the tail of the movie is dropped. + decodePendingElementaryStream(playback, true); playback.streamEnded = true; playback.cdStreamGeneration = g_mpeg_stub_state.cdStreamGeneration; flushDecoderIfEnded(playback); @@ -1456,8 +1553,13 @@ namespace ps2_stubs // lets the game's producer loop wake the consumer again. Backpressure // still propagates naturally to sceCdStRead because the ring does not // advance while this is true. + // Bounded by buffered bytes now, not by decoded pictures: decode is + // pulled by the consumer, so pictures no longer pile up ahead of it + // and the old limit would fire on a queue that is nearly always + // empty. return !g_mpeg_stub_state.currentCdStreamEofSeen && - playback.decodedFrames.size() >= kMaxDecodedPicturesAhead; + (playback.pendingEsBytes >= mpegMaxPendingEsBytes() || + playback.decodedFrames.size() >= kMaxDecodedPicturesAhead); } void recordCdStreamBytesDemuxedUnlocked( @@ -2148,6 +2250,10 @@ namespace ps2_stubs { consumed = appendGuestBytes(mpegAddr, playback, rdram, dataAddr, byteCount, callbackEvents); recordCdStreamBytesDemuxedUnlocked(consumed, completedMpegIds, eofChanged); + // Enough to wake a consumer parked in sceMpegGetPicture; it + // stops as soon as one picture exists, so the decode rate still + // follows consumption. + decodePendingElementaryStream(playback); } decodedCount = playback.decodedFrames.size(); decoderFailed = playback.decoderFailed; @@ -2156,6 +2262,7 @@ namespace ps2_stubs if (backpressured) { + const uint64_t refusals = g_mpegDemuxRefusedCount.fetch_add(1u, std::memory_order_relaxed); if (traceIdx < 32u) { PS2_IF_AGRESSIVE_LOGS({ @@ -2164,8 +2271,11 @@ namespace ps2_stubs } // Same reasoning and the same order as the ring variant. setReturnS32(ctx, 0); - runtime->eeScheduler().rotateReadyQueue(0, false); - runtime->eeScheduler().transferIfRequested(false); + if ((refusals % mpegDemuxYieldInterval()) == 0u) + { + runtime->eeScheduler().rotateReadyQueue(0, false); + runtime->eeScheduler().transferIfRequested(false); + } return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); @@ -2201,6 +2311,9 @@ namespace ps2_stubs void sceMpegDemuxPssRing(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { MpegScopedTimer timer(g_mpegDemuxNanos, g_mpegDemuxCount); + g_mpegDemuxThreadId.store( + static_cast(runtime->eeScheduler().currentThreadId()), + std::memory_order_relaxed); static std::atomic s_demuxRingEntryCount{0u}; const uint32_t entryIdx = s_demuxRingEntryCount.fetch_add(1u, std::memory_order_relaxed); if (entryIdx < 4u) @@ -2245,6 +2358,9 @@ namespace ps2_stubs ringSize, callbackEvents); recordCdStreamBytesDemuxedUnlocked(consumed, completedMpegIds, eofChanged); + // Same reason as the non-ring variant: keep a parked + // sceMpegGetPicture wakeable without decoding ahead. + decodePendingElementaryStream(playback); } decodedCount = playback.decodedFrames.size(); decoderFailed = playback.decoderFailed; @@ -2253,6 +2369,7 @@ namespace ps2_stubs if (backpressured) { + const uint64_t refusals = g_mpegDemuxRefusedCount.fetch_add(1u, std::memory_order_relaxed); if (traceIdx < 32u) { PS2_IF_AGRESSIVE_LOGS({ @@ -2278,8 +2395,11 @@ namespace ps2_stubs // bindMainContextForSyscall's assert on the next syscall. Same // order as the RotateThreadReadyQueue syscall. setReturnS32(ctx, 0); - runtime->eeScheduler().rotateReadyQueue(0, false); - runtime->eeScheduler().transferIfRequested(false); + if ((refusals % mpegDemuxYieldInterval()) == 0u) + { + runtime->eeScheduler().rotateReadyQueue(0, false); + runtime->eeScheduler().transferIfRequested(false); + } return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); @@ -2357,6 +2477,9 @@ namespace ps2_stubs void sceMpegGetPicture(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { MpegScopedTimer timer(g_mpegGetPictureNanos, g_mpegGetPictureCount); + g_mpegGetPictureThreadId.store( + static_cast(runtime->eeScheduler().currentThreadId()), + std::memory_order_relaxed); const uint32_t mpegAddr = getRegU32(ctx, 4); const uint32_t imageAddr = getRegU32(ctx, 5); uint32_t width = kStubMovieWidth; @@ -2368,6 +2491,8 @@ namespace ps2_stubs { std::unique_lock lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); + // The consumer drives decode: demux only buffers. + decodePendingElementaryStream(playback); // sawSequenceEnd is the only end-of-video signal a game streaming // with plain sceCdRead can raise: the CD-stream EOF path belongs to // sceCdSt*, and a PSS need not carry a program-end code. @@ -2470,7 +2595,7 @@ namespace ps2_stubs // Not on the call that serves the final frame -- the caller would // tear the movie down before uploading it. - movieEnded = !haveFrame && playback.decodedFrames.empty() && + movieEnded = !haveFrame && playback.decodedFrames.empty() && playback.pendingEs.empty() && (playback.sawSequenceEnd || playback.streamEnded || playback.decoderFailed || g_mpeg_stub_state.currentCdStreamEofSeen); } @@ -2579,7 +2704,10 @@ namespace ps2_stubs ++g_mpeg_stub_state.isEndTraceCount; } - setReturnS32(ctx, (ended && playback.decodedFrames.empty() && presentationComplete) ? 1 : 0); + setReturnS32(ctx, (ended && playback.decodedFrames.empty() && playback.pendingEs.empty() && + presentationComplete) + ? 1 + : 0); } void sceMpegIsRefBuffEmpty(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) From 8b83b2bc28285939068e775855c0a4672959405d Mon Sep 17 00:00:00 2001 From: Sinan KARAKAYA Date: Sun, 23 Aug 2026 13:31:39 +0200 Subject: [PATCH 26/68] Run guest threads on their own stacks instead of unwinding them to switch The EE scheduler transferred control by throwing EeDispatcherTransfer, which destroyed the guest's whole C++ call chain and rebuilt it on the way back -- about 2.7 executor re-entries per transfer, ~10us each, where 60fps allows ~540ns. DQ8's movie thread needs ~31,000 transfers per movie frame, so the movie ran at 1.4fps and _Unwind was 43% of EE thread time. EeFiber gives each GuestThread its own stack. A yield now saves the callee-saved registers and swaps the stack pointer: 22ns per round trip, measured. Guest dispatches per movie frame fall 82,441 -> 353 and thrown transfers per 30 frames 927,230 -> 2; the movie runs at 7.0fps. transferIfRequested suspends, and so does blockCurrentResumable -- semaphore, event flag and sleep waits, which only take a return value from makeReady() and whose callers have nothing left to do. Waits carrying a completion still throw, because the completion re-invokes its syscall on wake and would run twice against a stack that is still there. blockCurrent stays [[noreturn]]: giving it a return path makes the compiler emit ud2, and waitVSync/waitExternal take an optional completion, so a flag on blockCurrent turns those into SIGILL. x86-64 gets a hand-written switch, everything else falls back to ucontext (correct, but a sigprocmask syscall: 461ns against 22ns). setjmp/longjmp across stacks is faster still and deliberately unused -- _FORTIFY_SOURCE turns it into an abort. The trampoline realigns the stack before calling, or the first std::string inside a fiber faults on an aligned SSE spill. Stacks are mmapped with a guard page so an overflow faults instead of corrupting its neighbour. Also here: DQ8_PAD_SCRIPT replays input timed in guest vsync ticks so a script lands in the same place whatever the frame rate; DQ8_GFX_SCREENSHOT_DIR writes the presented frame as a PPM named after that same tick; DQ8_SKIP_MOVIES taps START while a movie plays, which is the game's own skip -- reporting end-of-stream from the MPEG stub instead leaves its streaming thread spinning on a black screen. --- ps2xRuntime/include/runtime/ee_fiber.h | 48 ++++ ps2xRuntime/include/runtime/ee_scheduler.h | 33 ++- ps2xRuntime/include/runtime/ps2_pad_host.h | 8 + ps2xRuntime/src/lib/Kernel/EeFiber.cpp | 274 +++++++++++++++++++++ ps2xRuntime/src/lib/Kernel/EeScheduler.cpp | 213 ++++++++++------ ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp | 3 + ps2xRuntime/src/lib/gs/gs_frontend.cpp | 88 +++++++ ps2xRuntime/src/lib/ps2_pad.cpp | 236 ++++++++++++++++-- ps2xRuntime/src/lib/ps2_runtime.cpp | 1 + ps2xTest/CMakeLists.txt | 1 + ps2xTest/src/ee_fiber_tests.cpp | 147 +++++++++++ ps2xTest/src/main.cpp | 2 + 12 files changed, 959 insertions(+), 95 deletions(-) create mode 100644 ps2xRuntime/include/runtime/ee_fiber.h create mode 100644 ps2xRuntime/src/lib/Kernel/EeFiber.cpp create mode 100644 ps2xTest/src/ee_fiber_tests.cpp diff --git a/ps2xRuntime/include/runtime/ee_fiber.h b/ps2xRuntime/include/runtime/ee_fiber.h new file mode 100644 index 000000000..a29f2ac19 --- /dev/null +++ b/ps2xRuntime/include/runtime/ee_fiber.h @@ -0,0 +1,48 @@ +#pragma once + +#include +#include + +// A stack a guest thread can be suspended on and resumed into. +// +// The EE scheduler used to transfer control by throwing EeDispatcherTransfer, +// which unwinds the guest's whole C++ call chain and then rebuilds it on the +// way back. That cost ~10us per switch against the ~540ns a 60fps movie frame +// can afford, because DQ8's movie thread yields ~31,000 times per frame. +// Suspending the stack instead makes a switch a register save and a stack +// pointer swap. +// +// x86-64 gets a hand-written switch; everything else falls back to ucontext, +// which is correct but pays a sigprocmask syscall per switch. setjmp/longjmp +// across stacks is deliberately not used: glibc's _FORTIFY_SOURCE turns it +// into an abort ("longjmp causes uninitialized stack frame"). +class EeFiber +{ +public: + using EntryFn = void (*)(void *user); + + EeFiber() = default; + ~EeFiber(); + + EeFiber(const EeFiber &) = delete; + EeFiber &operator=(const EeFiber &) = delete; + + // Allocates the stack and arms `entry`; entry must never return. + bool create(EntryFn entry, void *user, size_t stackBytes); + [[nodiscard]] bool valid() const noexcept { return m_stack != nullptr; } + void destroy(); + + // Called from the scheduler: run this fiber until it switches back. + void resume(); + // Called from inside the fiber: hand control back to whoever resumed it. + void suspend(); + + [[nodiscard]] static bool usingFastSwitch() noexcept; + +private: + void *m_stack = nullptr; // lowest address of the allocation + size_t m_stackBytes = 0u; + void *m_fiberSp = nullptr; // suspended fiber stack pointer + void *m_returnSp = nullptr; // stack pointer of whoever called resume() + void *m_platform = nullptr; // ucontext pair on the fallback path +}; diff --git a/ps2xRuntime/include/runtime/ee_scheduler.h b/ps2xRuntime/include/runtime/ee_scheduler.h index d6dce5e68..fe3263c13 100644 --- a/ps2xRuntime/include/runtime/ee_scheduler.h +++ b/ps2xRuntime/include/runtime/ee_scheduler.h @@ -1,6 +1,9 @@ #pragma once #include "ps2_runtime.h" +#include "runtime/ee_fiber.h" + +#include #include #include @@ -122,6 +125,11 @@ struct GuestThread EeWaitState wait{}; std::function resumeCompletion; std::vector invocations; + // Host stack this thread's guest code runs on, created on first dispatch. + // While inGuestCall is set the stack is suspended part way through a guest + // call and must be resumed, not re-entered from the saved pc. + std::unique_ptr fiber; + bool inGuestCall = false; [[nodiscard]] R5900Context &activeContext() { @@ -375,15 +383,21 @@ class EeScheduler void reserveGuestStackFromAsyncPool(uint32_t guestStackBase); void enqueueReady(GuestThread &thread, bool front = false); void removeReady(GuestThread &thread); + // Runs one guest dispatch on `thread`'s fiber, creating it if needed, and + // returns when the fiber suspends or the call completes. + void enterGuest(GuestThread &thread); + // The fiber's body: run the pending dispatch, hand control back, repeat. + static void fiberEntry(void *user); + void runPendingGuestCall(); + [[nodiscard]] GuestThread *selectReady(); - // What selectReady() would return, without dequeuing it. 0 when nothing is ready. - [[nodiscard]] int peekReadyId() const; - // Whether the dispatch loop has work that a same-thread yield must not skip. - [[nodiscard]] bool mustReturnToDispatcher() const noexcept; void makeRunning(GuestThread &thread); void makeDormant(GuestThread &thread); void removeFromWaitObject(GuestThread &thread); [[noreturn]] void blockCurrent(EeWaitState wait); + // Suspends the guest stack instead of unwinding it; see the definition for + // when that is allowed. + void blockCurrentResumable(EeWaitState wait); void makeReady(GuestThread &thread, int result, bool interruptSafe); void requestPreemptionIfHigher(const GuestThread &readyThread, bool interruptSafe); void applyPendingPreemption(); @@ -440,6 +454,17 @@ class EeScheduler // guest dispatch. A yield back onto this thread can skip the unwind. int m_dispatchedThreadId = 0; + // Guest code runs on a per-thread fiber so a yield can suspend the stack + // instead of unwinding it. Null while the executor is on its own stack. + EeFiber *m_activeFiber = nullptr; + // Handed to the fiber for the dispatch it is about to run. + PS2Runtime::RecompiledFunction m_pendingFunction = nullptr; + R5900Context *m_pendingContext = nullptr; + bool m_pendingInsideInterrupt = false; + // A non-transfer exception escaping guest code, rethrown by the executor + // once it is back on its own stack. + std::exception_ptr m_fiberException; + mutable std::mutex m_eventMutex; std::condition_variable m_eventCv; std::deque m_events; diff --git a/ps2xRuntime/include/runtime/ps2_pad_host.h b/ps2xRuntime/include/runtime/ps2_pad_host.h index 74ed9edbf..3cbd06994 100644 --- a/ps2xRuntime/include/runtime/ps2_pad_host.h +++ b/ps2xRuntime/include/runtime/ps2_pad_host.h @@ -14,4 +14,12 @@ void ps2PadPollHost(); // button encoding, with `pressed` carrying edges seen since the last call. void ps2PadPublishHostState(uint32_t held, uint32_t pressed, uint32_t sticks); +// Clock for DQ8_PAD_SCRIPT, in guest vsync ticks rather than host frames: the +// game runs well under 60 fps, so a script timed in host frames would fire at +// the wrong point in the game and replay differently on every run. +void ps2PadSetGuestFrame(uint64_t frame); + +// The same clock, for anything that wants to name its output in script time. +uint64_t ps2PadCurrentGuestFrame(); + #endif diff --git a/ps2xRuntime/src/lib/Kernel/EeFiber.cpp b/ps2xRuntime/src/lib/Kernel/EeFiber.cpp new file mode 100644 index 000000000..5338f4b3f --- /dev/null +++ b/ps2xRuntime/src/lib/Kernel/EeFiber.cpp @@ -0,0 +1,274 @@ +#include "runtime/ee_fiber.h" + +#include +#include +#include + +#if defined(__unix__) || defined(__APPLE__) +#include +#define EE_FIBER_HAVE_MMAP 1 +#else +#define EE_FIBER_HAVE_MMAP 0 +#endif + +namespace +{ + // Whole mapping including one guard page at the low end. + void *mmapStack(size_t bytes) + { +#if EE_FIBER_HAVE_MMAP + void *base = mmap(nullptr, bytes, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (base == MAP_FAILED) + { + return nullptr; + } + if (mprotect(base, 4096u, PROT_NONE) != 0) + { + munmap(base, bytes); + return nullptr; + } + return base; +#else + return std::aligned_alloc(64u, bytes); +#endif + } + + void munmapStack(void *base, size_t bytes) + { +#if EE_FIBER_HAVE_MMAP + if (base != nullptr) + { + munmap(base, bytes); + } +#else + (void)bytes; + std::free(base); +#endif + } +} // namespace + +#if defined(__x86_64__) && (defined(__linux__) || defined(__unix__) || defined(__APPLE__)) +#define EE_FIBER_FAST_X86_64 1 +#else +#define EE_FIBER_FAST_X86_64 0 +#include +#endif + +namespace +{ +#if EE_FIBER_FAST_X86_64 + // void eeFiberSwitch(void **saveSp, void *targetSp) + // + // Saves the callee-saved registers and the FP control words on the current + // stack, records the resulting stack pointer through saveSp, switches to + // targetSp and restores the same set from there. The final `ret` lands on + // whatever return address that stack was suspended at -- for a fresh fiber, + // the trampoline create() planted. + extern "C" void eeFiberSwitch(void **saveSp, void *targetSp); + __asm__( + ".text\n" + ".globl eeFiberSwitch\n" + ".hidden eeFiberSwitch\n" + ".type eeFiberSwitch,@function\n" + ".align 16\n" + "eeFiberSwitch:\n" + " pushq %rbp\n" + " pushq %rbx\n" + " pushq %r12\n" + " pushq %r13\n" + " pushq %r14\n" + " pushq %r15\n" + " subq $8, %rsp\n" + " stmxcsr (%rsp)\n" + " fnstcw 4(%rsp)\n" + " movq %rsp, (%rdi)\n" + " movq %rsi, %rsp\n" + " ldmxcsr (%rsp)\n" + " fldcw 4(%rsp)\n" + " addq $8, %rsp\n" + " popq %r15\n" + " popq %r14\n" + " popq %r13\n" + " popq %r12\n" + " popq %rbx\n" + " popq %rbp\n" + " ret\n" + ".size eeFiberSwitch,.-eeFiberSwitch\n"); + + struct FiberBootstrap + { + EeFiber::EntryFn entry; + void *user; + }; + + // Reached by the first `ret` out of eeFiberSwitch. %rbx carries the + // bootstrap because create() planted it in the saved-register slot. + extern "C" void eeFiberTrampoline(); + __asm__( + ".text\n" + ".globl eeFiberTrampoline\n" + ".hidden eeFiberTrampoline\n" + ".type eeFiberTrampoline,@function\n" + ".align 16\n" + "eeFiberTrampoline:\n" + // Entered by `ret`, so rsp%16 == 8 as at any function entry. SysV wants + // rsp%16 == 0 immediately before a call; without this the callee's + // 16-byte SSE spills fault. + " subq $8, %rsp\n" + " movq %rbx, %rdi\n" + " call eeFiberEnter\n" + " hlt\n" // eeFiberEnter never returns + ".size eeFiberTrampoline,.-eeFiberTrampoline\n"); + + extern "C" void eeFiberEnter(void *bootstrap) + { + auto *boot = static_cast(bootstrap); + boot->entry(boot->user); + std::abort(); // entry must never return + } +#else + struct FiberPlatform + { + ucontext_t fiber{}; + ucontext_t caller{}; + EeFiber::EntryFn entry = nullptr; + void *user = nullptr; + }; + + FiberPlatform *g_startingFiber = nullptr; + + void fiberTrampoline() + { + FiberPlatform *self = g_startingFiber; + self->entry(self->user); + std::abort(); + } +#endif +} // namespace + +bool EeFiber::usingFastSwitch() noexcept +{ + return EE_FIBER_FAST_X86_64 != 0; +} + +EeFiber::~EeFiber() +{ + destroy(); +} + +void EeFiber::destroy() +{ +#if !EE_FIBER_FAST_X86_64 + delete static_cast(m_platform); +#else + std::free(m_platform); +#endif + m_platform = nullptr; + munmapStack(m_stack, m_stackBytes); + m_stack = nullptr; + m_stackBytes = 0u; + m_fiberSp = nullptr; + m_returnSp = nullptr; +} + +bool EeFiber::create(EntryFn entry, void *user, size_t stackBytes) +{ + destroy(); + if (entry == nullptr || stackBytes < 64u * 1024u) + { + return false; + } + // Guest call chains nest in C++ through dispatchGuestBranch, so a fiber + // stack can run deep. Map it with a PROT_NONE guard page at the low end: + // an overflow then faults on the guard instead of quietly writing into + // whatever the allocator put underneath. + const size_t pageSize = 4096u; + const size_t mapped = ((stackBytes + pageSize - 1u) / pageSize) * pageSize + pageSize; + void *base = mmapStack(mapped); + if (base == nullptr) + { + return false; + } + void *stack = static_cast(base) + pageSize; + m_stack = base; + m_stackBytes = mapped; + stackBytes = mapped - pageSize; + +#if EE_FIBER_FAST_X86_64 + auto *boot = static_cast(std::malloc(sizeof(FiberBootstrap))); + if (boot == nullptr) + { + destroy(); + return false; + } + boot->entry = entry; + boot->user = user; + m_platform = boot; + + // Build the frame eeFiberSwitch will pop: FP words, r15..rbx, rbp, then the + // return address it rets to. rbx carries the bootstrap into the trampoline. + auto *top = reinterpret_cast(static_cast(stack) + stackBytes); + top = reinterpret_cast(reinterpret_cast(top) & ~static_cast(15)); + // SysV wants rsp % 16 == 8 on entry, i.e. the return-address slot 16-aligned. + *--top = 0u; // alignment padding + *--top = reinterpret_cast(&eeFiberTrampoline); // ret target + *--top = 0u; // rbp + *--top = reinterpret_cast(boot); // rbx + *--top = 0u; // r12 + *--top = 0u; // r13 + *--top = 0u; // r14 + *--top = 0u; // r15 + // mxcsr (low 32) + x87 control word (next 16), matching the switch's layout. + uint32_t mxcsr = 0x1F80u; + uint16_t fcw = 0x037Fu; + __asm__ __volatile__("stmxcsr %0" : "=m"(mxcsr)); + __asm__ __volatile__("fnstcw %0" : "=m"(fcw)); + --top; + std::memcpy(top, &mxcsr, sizeof(mxcsr)); + std::memcpy(reinterpret_cast(top) + 4, &fcw, sizeof(fcw)); + m_fiberSp = top; +#else + auto *platform = new (std::nothrow) FiberPlatform(); + if (platform == nullptr) + { + destroy(); + return false; + } + platform->entry = entry; + platform->user = user; + if (getcontext(&platform->fiber) != 0) + { + delete platform; + destroy(); + return false; + } + platform->fiber.uc_stack.ss_sp = stack; + platform->fiber.uc_stack.ss_size = stackBytes; + platform->fiber.uc_link = nullptr; + makecontext(&platform->fiber, fiberTrampoline, 0); + m_platform = platform; +#endif + return true; +} + +void EeFiber::resume() +{ +#if EE_FIBER_FAST_X86_64 + eeFiberSwitch(&m_returnSp, m_fiberSp); +#else + auto *platform = static_cast(m_platform); + g_startingFiber = platform; + swapcontext(&platform->caller, &platform->fiber); +#endif +} + +void EeFiber::suspend() +{ +#if EE_FIBER_FAST_X86_64 + eeFiberSwitch(&m_fiberSp, m_returnSp); +#else + auto *platform = static_cast(m_platform); + swapcontext(&platform->fiber, &platform->caller); +#endif +} diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index cbaf8966c..81f14208f 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -22,6 +22,8 @@ std::atomic g_eeGuestDispatchCount{0}; // whether the executor is re-entering guest code because a transfer asked it // to, or because the call chain has to be rebuilt after one. std::atomic g_eeTransferThrowCount{0}; +// Transfers served by suspending the guest stack instead of unwinding it. +std::atomic g_eeTransferSuspendCount{0}; namespace { @@ -235,6 +237,34 @@ void EeScheduler::run() GuestThread *running = currentThread(); assert(running != nullptr); + + // Suspended part way through a guest call: its stack is still live, so + // resume it rather than dispatching the saved pc, which points into the + // middle of a function the fiber is already executing. + if (running->inGuestCall) + { + enterGuest(*running); + if (m_fiberException) + { + std::exception_ptr escaped; + escaped.swap(m_fiberException); + m_running.store(false, std::memory_order_release); + publishSnapshotNow(); + std::rethrow_exception(escaped); + } + processPendingEvents(); + if (m_rescheduleRequested && m_currentThreadId != 0) + { + GuestThread *preempted = currentThread(); + assert(preempted != nullptr); + enqueueReady(*preempted, !m_timeSliceExpired); + m_currentThreadId = 0; + m_rescheduleRequested = false; + m_timeSliceExpired = false; + } + continue; + } + if (running->resumeCompletion) { auto completion = std::move(running->resumeCompletion); @@ -363,29 +393,19 @@ void EeScheduler::run() continue; } - try - { - m_insideInterrupt = !running->invocations.empty() && running->invocations.back().kind == GuestInvocationKind::Interrupt; - m_guestExecuting.store(true, std::memory_order_release); - m_dispatchedThreadId = running->id; - function(m_rdram, &context, &m_runtime); - m_dispatchedThreadId = 0; - m_guestExecuting.store(false, std::memory_order_release); - m_insideInterrupt = false; - } - catch (const EeDispatcherTransfer &) - { - m_dispatchedThreadId = 0; - m_guestExecuting.store(false, std::memory_order_release); - m_insideInterrupt = false; - } - catch (...) + m_pendingFunction = function; + m_pendingContext = &context; + m_pendingInsideInterrupt = + !running->invocations.empty() && + running->invocations.back().kind == GuestInvocationKind::Interrupt; + enterGuest(*running); + if (m_fiberException) { - m_dispatchedThreadId = 0; - m_guestExecuting.store(false, std::memory_order_release); + std::exception_ptr escaped; + escaped.swap(m_fiberException); m_running.store(false, std::memory_order_release); - publishSnapshot(); - throw; + publishSnapshotNow(); + std::rethrow_exception(escaped); } processPendingEvents(); @@ -749,7 +769,7 @@ void EeScheduler::sleepCurrent() setReturnS32(&self->activeContext(), KE_OK); return; } - blockCurrent(EeWaitState{EeWaitReason::Sleep, std::monostate{}}); + blockCurrentResumable(EeWaitState{EeWaitReason::Sleep, std::monostate{}}); } int EeScheduler::wakeupThread(int id, bool interruptSafe) @@ -914,39 +934,21 @@ void EeScheduler::transferIfRequested(bool interruptSafe) m_currentThreadId = 0; } - // A yield that lands back on the thread we are already executing does not - // need the C++ stack torn down and rebuilt. Unwinding EeDispatcherTransfer - // through the guest's frames was ~40% of EE time during a movie, because - // DQ8's movie producer yields thousands of times per frame and almost - // always ends up running again immediately. - if (m_dispatchedThreadId != 0 && - m_pendingInvocations.empty() && - peekReadyId() == m_dispatchedThreadId && - !mustReturnToDispatcher()) - { - // Charge what the dispatch loop would have charged, so the time slice - // and the EE timers still advance and this cannot spin forever. - accountCycles(kGuestDispatchCycles); - GuestThread *next = selectReady(); - assert(next != nullptr && next->id == m_dispatchedThreadId); - if (next != nullptr && next->id == m_dispatchedThreadId) - { - next->status = EeThreadStatus::Running; - m_currentThreadId = next->id; - m_rescheduleRequested = false; - m_timeSliceExpired = false; - return; - } - // peekReadyId lied, so put it back and take the slow path. - if (next != nullptr) - { - enqueueReady(*next, true); - } - } - m_rescheduleRequested = false; m_timeSliceExpired = false; publishSnapshot(); + + // The thread stays runnable, so its C++ stack is still wanted: suspend it + // rather than unwinding and rebuilding it. Only the executor's own stack + // has nowhere to suspend to (direct-syscall tests, embedders), and it still + // throws. + if (m_activeFiber != nullptr) + { + g_eeTransferSuspendCount.fetch_add(1u, std::memory_order_relaxed); + m_activeFiber->suspend(); + return; + } + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1063,7 +1065,7 @@ void EeScheduler::waitSemaphore(int id) GuestThread *self = currentThread(); assert(self != nullptr); object->waiters.push_back(self->id); - blockCurrent(EeWaitState{EeWaitReason::Semaphore, EeSemaphoreWait{id}}); + blockCurrentResumable(EeWaitState{EeWaitReason::Semaphore, EeSemaphoreWait{id}}); } int EeScheduler::createEventFlag(uint32_t initialBits, uint32_t attr, uint32_t option) @@ -1186,8 +1188,8 @@ void EeScheduler::waitEventFlag(int id, uint32_t bits, uint32_t mode, uint32_t r return; } flag->waiters.push_back(self->id); - blockCurrent(EeWaitState{EeWaitReason::EventFlag, - EeEventFlagWait{id, bits, mode, resultAddress}}); + blockCurrentResumable(EeWaitState{EeWaitReason::EventFlag, + EeEventFlagWait{id, bits, mode, resultAddress}}); } int EeScheduler::setAlarm(uint16_t ticks, @@ -1807,32 +1809,82 @@ GuestThread *EeScheduler::selectReady() return nullptr; } -int EeScheduler::peekReadyId() const +void EeScheduler::fiberEntry(void *user) +{ + auto *self = static_cast(user); + for (;;) + { + self->runPendingGuestCall(); + // The dispatch is over; hand the executor back its stack. Resumed when + // this thread is scheduled again with a new pending call. + self->m_activeFiber->suspend(); + } +} + +// Runs on the fiber's stack. Nothing may escape it: an exception unwinding past +// the fiber entry has no frame to land on. +void EeScheduler::runPendingGuestCall() { - for (const auto &queue : m_readyQueues) + PS2Runtime::RecompiledFunction function = m_pendingFunction; + R5900Context *context = m_pendingContext; + m_pendingFunction = nullptr; + m_pendingContext = nullptr; + try { - if (!queue.empty()) + m_insideInterrupt = m_pendingInsideInterrupt; + m_guestExecuting.store(true, std::memory_order_release); + if (function != nullptr && context != nullptr) { - return queue.front(); + function(m_rdram, context, &m_runtime); } } - return 0; + catch (const EeDispatcherTransfer &) + { + // A block, an invocation push or a thread exit. The stack is unwound, + // so the next dispatch starts from the thread's saved pc again. + } + catch (...) + { + m_fiberException = std::current_exception(); + } + m_guestExecuting.store(false, std::memory_order_release); + m_insideInterrupt = false; + m_dispatchedThreadId = 0; } -bool EeScheduler::mustReturnToDispatcher() const noexcept +void EeScheduler::enterGuest(GuestThread &thread) { - // The same conditions checkpointDue() uses, minus its side effects. - if (m_checkpointPending.load(std::memory_order_acquire) || - m_stopRequested.load(std::memory_order_acquire)) + if (!thread.fiber) { - return true; + // Sized for deeply nested guest call chains; mapped lazily, so the + // resident cost is only the pages a thread actually touches. + static const size_t stackBytes = [] { + const char *value = std::getenv("DQ8_EE_FIBER_STACK_KB"); + const long long parsed = value ? std::atoll(value) : 0; + const size_t kb = parsed > 0 ? static_cast(parsed) : 8192u; + return kb * 1024u; + }(); + thread.fiber = std::make_unique(); + if (!thread.fiber->create(&EeScheduler::fiberEntry, this, stackBytes)) + { + thread.fiber.reset(); + throw std::runtime_error("EE scheduler could not allocate a guest fiber stack"); + } } - const uint64_t nextEventCycle = m_nextDeadlineCycle.load(std::memory_order_acquire); - if (nextEventCycle != 0u && m_eeCycle >= nextEventCycle) + + assert(m_activeFiber == nullptr && "guest fibers must not nest"); + EeFiber *previous = m_activeFiber; + m_activeFiber = thread.fiber.get(); + m_dispatchedThreadId = thread.id; + thread.inGuestCall = true; + m_activeFiber->resume(); + m_activeFiber = previous; + // Cleared by the fiber when a dispatch finishes; still set means it + // suspended part way through and its stack is waiting to be resumed. + if (m_dispatchedThreadId == 0) { - return true; + thread.inGuestCall = false; } - return m_eeCycle >= m_sliceEndCycle; } void EeScheduler::makeRunning(GuestThread &item) @@ -1898,6 +1950,29 @@ void EeScheduler::blockCurrent(EeWaitState wait) throw EeDispatcherTransfer{}; } +// Blocks and comes back where it left off. Only for waits that carry no +// completion -- a completion re-invokes its syscall on wake, which would run +// twice if the stack were still there -- and whose caller has nothing left to +// do afterwards, so returning here returns into the guest. blockCurrent() +// stays [[noreturn]] for everyone else, including the [[noreturn]] waitVSync +// and waitExternal, whose completions are optional. +void EeScheduler::blockCurrentResumable(EeWaitState wait) +{ + assert(!wait.completion && "a resumable block must not carry a completion"); + if (m_activeFiber == nullptr) + { + blockCurrent(std::move(wait)); + } + GuestThread *self = currentThread(); + assert(self != nullptr); + self->wait = std::move(wait); + self->status = self->suspendCount == 0 ? EeThreadStatus::Waiting : EeThreadStatus::WaitingSuspended; + m_currentThreadId = 0; + publishSnapshot(); + g_eeTransferSuspendCount.fetch_add(1u, std::memory_order_relaxed); + m_activeFiber->suspend(); +} + void EeScheduler::makeReady(GuestThread &item, int result, bool interruptSafe) { auto completion = std::move(item.wait.completion); diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp index 4cf554b36..fef2e489c 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp @@ -550,6 +550,7 @@ namespace ps2_stubs int64_t dts90k = -1; }; + struct MpegPlaybackState { uint32_t picturesServed = 0u; @@ -2491,6 +2492,7 @@ namespace ps2_stubs { std::unique_lock lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); + { // The consumer drives decode: demux only buffers. decodePendingElementaryStream(playback); // sawSequenceEnd is the only end-of-video signal a game streaming @@ -2598,6 +2600,7 @@ namespace ps2_stubs movieEnded = !haveFrame && playback.decodedFrames.empty() && playback.pendingEs.empty() && (playback.sawSequenceEnd || playback.streamEnded || playback.decoderFailed || g_mpeg_stub_state.currentCdStreamEofSeen); + } } mpegGuestWrite32(rdram, mpegAddr + 0x00u, width); diff --git a/ps2xRuntime/src/lib/gs/gs_frontend.cpp b/ps2xRuntime/src/lib/gs/gs_frontend.cpp index 6d99b04c9..aa622f741 100644 --- a/ps2xRuntime/src/lib/gs/gs_frontend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_frontend.cpp @@ -2,15 +2,22 @@ #include "runtime/gs/gs_cpu_backend.h" #include "ps2_log.h" #include "runtime/ps2_memory.h" + +// Weak on purpose: the offline gs-dump harness links this file without the pad +// backend, and there a screenshot just falls back to counting latches. +extern "C++" __attribute__((weak)) uint64_t ps2PadCurrentGuestFrame(); #include #include #include #include #include +#include #include +#include #include #include #include +#include namespace { @@ -525,6 +532,8 @@ GSPresentationRequest GS::buildPresentationRequestUnlocked() const return request; } +static void maybeWriteScreenshot(GS &gs); + void GS::latchHostPresentationFrame() { GSPresentationRequest request{}; @@ -578,6 +587,11 @@ void GS::latchHostPresentationFrame() m_hasHostPresentationFrame = hasHostFrame; } + if (hasHostFrame) + { + maybeWriteScreenshot(*this); + } + if (presented) { std::lock_guard lock(m_stateMutex); @@ -585,6 +599,80 @@ void GS::latchHostPresentationFrame() } } +// DQ8_GFX_SCREENSHOT_DIR + DQ8_GFX_SCREENSHOT_EVERY=N dump the presented frame +// as a binary PPM every N frames. PPM keeps this dependency-free; convert with +// `ffmpeg -i shot.ppm shot.png` when a viewer is wanted. A backend that presents +// natively skips the CPU compose, so it disables that fast path while this is on. +// +// A free function on purpose: gs_frontend.h reaches the whole recompiled corpus +// through ps2_runtime.h, so declaring this on GS would cost a full rebuild. +static void maybeWriteScreenshot(GS &gs) +{ + struct Config + { + std::string directory; + uint64_t every = 0u; + }; + static const Config config = [] { + Config parsed{}; + const char *dir = std::getenv("DQ8_GFX_SCREENSHOT_DIR"); + if (dir == nullptr || *dir == '\0') + { + return parsed; + } + parsed.directory = dir; + const char *every = std::getenv("DQ8_GFX_SCREENSHOT_EVERY"); + const long long value = every ? std::atoll(every) : 0; + parsed.every = value > 0 ? static_cast(value) : 60ull; + return parsed; + }(); + if (config.every == 0u) + { + return; + } + + // Named by guest vsync tick, the same clock DQ8_PAD_SCRIPT uses, so a + // screenshot tells you directly which frame to script an input at. + static uint64_t s_latches = 0u; + const uint64_t frame = + ps2PadCurrentGuestFrame != nullptr ? ps2PadCurrentGuestFrame() : s_latches++; + static uint64_t s_lastWritten = std::numeric_limits::max(); + if (s_lastWritten != std::numeric_limits::max() && + frame < s_lastWritten + config.every) + { + return; + } + s_lastWritten = frame; + + std::vector pixels; + uint32_t width = 0u; + uint32_t height = 0u; + if (!gs.copyLatchedHostPresentationFrame(pixels, width, height, nullptr, nullptr, nullptr) || + width == 0u || height == 0u) + { + return; + } + + char path[512]; + std::snprintf(path, sizeof(path), "%s/frame_%06llu.ppm", config.directory.c_str(), + static_cast(frame)); + std::ofstream out(path, std::ios::binary); + if (!out) + { + return; + } + out << "P6\n" << width << " " << height << "\n255\n"; + // copyLatchedHostPresentationFrame hands back tightly packed RGBA. + std::vector rgb(static_cast(width) * height * 3u); + for (size_t i = 0u, n = static_cast(width) * height; i < n; ++i) + { + rgb[i * 3u + 0u] = pixels[i * 4u + 0u]; + rgb[i * 3u + 1u] = pixels[i * 4u + 1u]; + rgb[i * 3u + 2u] = pixels[i * 4u + 2u]; + } + out.write(reinterpret_cast(rgb.data()), static_cast(rgb.size())); +} + bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, uint32_t &outWidth, uint32_t &outHeight, diff --git a/ps2xRuntime/src/lib/ps2_pad.cpp b/ps2xRuntime/src/lib/ps2_pad.cpp index db37b887e..d9a0a4e39 100644 --- a/ps2xRuntime/src/lib/ps2_pad.cpp +++ b/ps2xRuntime/src/lib/ps2_pad.cpp @@ -1,9 +1,21 @@ #include "runtime/ps2_pad.h" #include "runtime/ps2_pad_host.h" #include "ps2_host_backend.h" +#include #include +#include #include +#include +#include #include +#include +#include +#include +#include +#include + +// Defined by the MPEG stub; advances once per picture the game asks for. +extern std::atomic g_mpegGetPictureCount; namespace { @@ -167,10 +179,201 @@ namespace namespace { + // Scripted input. DQ8_PAD_SCRIPT is either a file path or the script text + // itself, entries separated by ';' or newlines: + // + //