diff --git a/ps2xAnalyzer/CMakeLists.txt b/ps2xAnalyzer/CMakeLists.txt index 5933bca14..97d4cee26 100644 --- a/ps2xAnalyzer/CMakeLists.txt +++ b/ps2xAnalyzer/CMakeLists.txt @@ -27,8 +27,8 @@ add_library(ps2_analyzer_lib STATIC ${PS2ANALYZER_LIB_SOURCES}) target_include_directories(ps2_analyzer_lib PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include - ${CMAKE_SOURCE_DIR}/ps2xRecomp/include - ${CMAKE_SOURCE_DIR}/ps2xRuntime/include + ${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRecomp/include + ${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRuntime/include ) target_link_libraries(ps2_analyzer_lib PUBLIC @@ -50,7 +50,7 @@ install(TARGETS ps2_analyzer ps2_analyzer_lib ARCHIVE DESTINATION lib ) -include("${CMAKE_SOURCE_DIR}/ps2xRuntime/cmake/ReleaseMode.cmake") +include("${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRuntime/cmake/ReleaseMode.cmake") if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") EnableFastReleaseMode(ps2_analyzer_lib) diff --git a/ps2xIOP/CMakeLists.txt b/ps2xIOP/CMakeLists.txt index e17b4724d..d289937c3 100644 --- a/ps2xIOP/CMakeLists.txt +++ b/ps2xIOP/CMakeLists.txt @@ -16,6 +16,12 @@ add_library(ps2_iop STATIC src/modules/clfile.cpp src/modules/sound_update_stub.cpp src/modules/sdrdrv.cpp + src/native_iop.cpp + src/lle/cpu.cpp + src/lle/irx.cpp + src/lle/iop.cpp + src/lle/kernel.cpp + src/lle/spu2.cpp ) target_compile_features(ps2_iop PUBLIC cxx_std_20) diff --git a/ps2xIOP/README.md b/ps2xIOP/README.md index 4538f5091..f45377c1b 100644 --- a/ps2xIOP/README.md +++ b/ps2xIOP/README.md @@ -4,9 +4,11 @@ `ps2xRuntime`. It implements the behavior that games expect from IOP services exposed through SIF RPC and DMA. -This subsystem does not emulate the IOP's R3000A CPU and does not load or -execute IRX binaries. Its scope is the RPC/DMA behavior needed by recompiled -games. +The services below do not emulate the IOP's R3000A CPU and do not load or +execute IRX binaries. Their scope is the RPC/DMA behavior needed by recompiled +games. A runtime can additionally run a game's own modules on a small emulated +IOP, which is how sound drivers are handled; see +[Native IRX modules](#native-irx-modules). > [!IMPORTANT] > `ps2_iop`/`ps2x::iop` is a C++20 static library linked into the runtime. @@ -121,6 +123,69 @@ services, and every active service receives each SIF transfer notification. Implementations must filter the relevant SID/function or transfer kind/phase/address range themselves. +## Native IRX modules + +Some drivers are easier to run than to rewrite. A sound driver is the usual +case: its RPC protocol is the only part the EE sees, but what the game hears +comes from years of tuning in how it drives the SPU2. `NativeIop` +([`native_iop.h`](include/ps2x/iop/native_iop.h)) loads a game's own IRX files +onto an emulated IOP and lets their RPC servers answer the game directly. + +The emulator under [`src/lle/`](src/lle) is only as large as sound drivers +need: + +| File | Contents | +| --- | --- | +| `cpu.*` | R3000A interpreter, with load delay slots and no caches or MMU | +| `irx.*` | IRX parsing, relocation and import stub discovery | +| `kernel.cpp` | High-level kernel: threads, semaphores, event flags, mailboxes, alarms, hardware timers, interrupts, the IOP heap, SIF DMA and RPC, and the sysclib calls the drivers use | +| `spu2.*` | Both SPU2 cores: 48 ADPCM voices with ADSR, noise, pitch modulation, reverb, AutoDMA input and 2 MiB of sound memory | +| `iop.*` | The pieces together: memory map, DMA channels 4 and 7, interrupts and the clock | + +An import the kernel does not provide returns zero, and an instruction the CPU +does not implement stops its thread with a log message. Both mean a driver +needs more than has been written so far. + +### Timing + +IOP time is counted at 36.864 MHz and advances only as the SPU2 produces +samples, 768 cycles for each 48 kHz stereo pair. The host audio callback calls +`render()`, so the IOP runs in step with the audio device rather than with the +EE. Work that answers an RPC runs without moving the clock, until every thread +is waiting again, and so do interrupt, alarm and timer handlers. Charging that +work to the clock let heavy RPC traffic push the drivers' timers ahead of the +audio, and music played fast. + +With no audio device, or while it is paused, an RPC that waits on IOP time +steps the IOP itself and discards the samples. A call that still cannot finish +gives up after two seconds and logs a warning. + +### Hooking it up + +The runtime names the modules to run natively before the game starts. The +native path then takes over in three places: + +- `sceSifLoadModule` of a named module also loads it on the emulated IOP. The + high-level module tracker still hands out the id the game sees. +- `IopSubsystem::handleRpc` gives an RPC to a server a native module + registered before any high-level service with the same SID. +- IOP heap allocation and EE-to-IOP `sceSifSetDma` use the emulated IOP's + memory, since native modules read what the game sends them from there. + +IOP modules that bind back to an EE server, like SDRDRV's callback thread, +wait until the EE has registered it (`IopHost::hasGuestRpcServer`). Answering +the bind early left the thread spinning and starved the rest of the driver. + +On the runtime side, [`ps2_native_iop.h`](../ps2xRuntime/include/runtime/ps2_native_iop.h) +takes the module list, opens a 48 kHz stereo stream once the first module +loads, and sets the volume. Volume zero keeps the stream running: games wait +on their sound drivers, so the IOP has to keep going even when nothing is +heard. `PS2X_AUDIO_DUMP=path` keeps a raw copy of everything played +(48 kHz, stereo, signed 16-bit) for comparing against a reference offline. + +The SPU2 input path and the DMA registers have tests in +[`ps2_native_iop_tests.cpp`](../ps2xTest/src/ps2_native_iop_tests.cpp). + ## Linking the static library ```cmake diff --git a/ps2xIOP/include/ps2x/iop/iop_host.h b/ps2xIOP/include/ps2x/iop/iop_host.h index 1c872effa..dceafa3bb 100644 --- a/ps2xIOP/include/ps2x/iop/iop_host.h +++ b/ps2xIOP/include/ps2x/iop/iop_host.h @@ -82,6 +82,13 @@ namespace ps2x::iop virtual int32_t memoryCard(const MemoryCardRequest &request) = 0; virtual bool hasGuestFunction(uint32_t address) const = 0; + // Whether the EE has registered an RPC server under this id, which is + // what an IOP module binding to it waits for. + virtual bool hasGuestRpcServer(uint32_t sid) const + { + (void)sid; + return false; + } virtual bool invokeGuestFunction(uint64_t callToken, uint32_t address, uint32_t a0, diff --git a/ps2xIOP/include/ps2x/iop/iop_subsystem.h b/ps2xIOP/include/ps2x/iop/iop_subsystem.h index 0d086a0e9..c8dd03f54 100644 --- a/ps2xIOP/include/ps2x/iop/iop_subsystem.h +++ b/ps2xIOP/include/ps2x/iop/iop_subsystem.h @@ -1,6 +1,7 @@ #pragma once #include "ps2x/iop/iop_host.h" +#include "ps2x/iop/native_iop.h" #include "ps2x/iop/iop_types.h" #include @@ -33,6 +34,9 @@ namespace ps2x::iop [[nodiscard]] DebugSnapshot debugSnapshot() const; + // Original IRX modules running on an emulated IOP; see NativeIop. + [[nodiscard]] NativeIop &native(); + private: class Impl; std::unique_ptr m_impl; diff --git a/ps2xIOP/include/ps2x/iop/native_iop.h b/ps2xIOP/include/ps2x/iop/native_iop.h new file mode 100644 index 000000000..7cf6d9226 --- /dev/null +++ b/ps2xIOP/include/ps2x/iop/native_iop.h @@ -0,0 +1,53 @@ +#pragma once + +#include "ps2x/iop/iop_host.h" + +#include +#include +#include +#include +#include + +namespace ps2x::iop +{ + // Runs a game's own IRX modules, sound drivers mostly, on an emulated IOP + // with an SPU2. The audio callback calls render() while the EE thread loads + // modules and makes calls, so every entry point takes the same lock. + class NativeIop + { + public: + explicit NativeIop(IopHost &host); + ~NativeIop(); + + NativeIop(const NativeIop &) = delete; + NativeIop &operator=(const NativeIop &) = delete; + + // Module file names (for example "LIBSD.IRX") that run natively. + // Once any are set, IOP memory handed to the EE is this IOP's. + void setModules(std::vector names); + bool enabled() const; + bool claims(std::string_view guestPath) const; + // Load a claimed module; arguments are NUL-separated as sceSifLoadModule passes them. + int32_t load(std::string_view guestPath, std::string_view arguments); + + // A server a native module registered, with its record and receive buffer. + bool server(uint32_t sid, uint32_t &record, uint32_t &buffer) const; + // Blocking SIF RPC. The reply is written to receiveAddress in EE memory. + bool call(uint32_t sid, uint32_t function, uint32_t sendAddress, uint32_t sendSize, + uint32_t receiveAddress, uint32_t receiveSize); + + // IOP memory as the EE sees it through SIF DMA and the IOP heap calls. + bool write(uint32_t address, const void *data, uint32_t size); + bool read(uint32_t address, void *data, uint32_t size); + uint32_t allocate(uint32_t size); + void release(uint32_t address); + + // Fill interleaved 48 kHz stereo frames, running the IOP in step. + void render(int16_t *frames, uint32_t count); + bool active() const; + + private: + class Impl; + std::unique_ptr m_impl; + }; +} diff --git a/ps2xIOP/src/iop_subsystem.cpp b/ps2xIOP/src/iop_subsystem.cpp index d848f8de7..f07ba8356 100644 --- a/ps2xIOP/src/iop_subsystem.cpp +++ b/ps2xIOP/src/iop_subsystem.cpp @@ -68,7 +68,8 @@ namespace ps2x::iop { public: explicit Impl(IopHost &hostRef) - : host(hostRef), pluginCatalog(hostRef), coreServices(detail::createCoreServices(hostRef)), profiles(detail::createBuiltinProfiles()) + : host(hostRef), pluginCatalog(hostRef), coreServices(detail::createCoreServices(hostRef)), profiles(detail::createBuiltinProfiles()), + native(hostRef) { rebuildRoutes(); } @@ -118,6 +119,7 @@ namespace ps2x::iop std::string activeProvider; std::string lastError; bool routesValid = true; + NativeIop native; }; IopSubsystem::IopSubsystem(IopHost &host) @@ -279,6 +281,18 @@ namespace ps2x::iop RpcResult IopSubsystem::handleRpc(const RpcRequest &request) { + // A server registered by a native module replaces any high-level one. + uint32_t record = 0, buffer = 0; + if (m_impl->native.server(request.sid, record, buffer)) + { + RpcResult result; + result.handled = m_impl->native.call(request.sid, request.function, request.send.address, + request.send.size, request.receive.address, + request.receive.size); + result.resultAddress = request.receive.address; + result.serverDispatchPolicy = ServerDispatchPolicy::Suppress; + return result; + } const auto it = m_impl->routes.find(request.sid); if (it == m_impl->routes.end() || !it->second) { @@ -305,6 +319,11 @@ namespace ps2x::iop } } + NativeIop &IopSubsystem::native() + { + return m_impl->native; + } + DebugSnapshot IopSubsystem::debugSnapshot() const { DebugSnapshot snapshot; diff --git a/ps2xIOP/src/lle/bus.h b/ps2xIOP/src/lle/bus.h new file mode 100644 index 000000000..7072536b8 --- /dev/null +++ b/ps2xIOP/src/lle/bus.h @@ -0,0 +1,70 @@ +#pragma once + +#include +#include +#include + +namespace ps2x::iop::lle +{ + // Hardware registers behind 0x1F801000 and the SPU2 window. + class IoDevice + { + public: + virtual ~IoDevice() = default; + virtual uint32_t ioRead(uint32_t physical, unsigned bytes) = 0; + virtual void ioWrite(uint32_t physical, uint32_t value, unsigned bytes) = 0; + }; + + // The IOP's physical address space: 2 MiB of RAM (mirrored through the + // first 8 MiB), the 1 KiB scratchpad and memory-mapped hardware. + class Bus + { + public: + static constexpr uint32_t kRamBytes = 2u << 20; + + explicit Bus(IoDevice &io) : m_io(io), m_ram(kRamBytes, 0u), m_scratchpad(1024u, 0u) {} + + uint8_t *ram() { return m_ram.data(); } + const uint8_t *ram() const { return m_ram.data(); } + + uint32_t fetch(uint32_t address) { return read32(address); } + + uint8_t read8(uint32_t address) { return static_cast(read<1>(address)); } + uint16_t read16(uint32_t address) { return static_cast(read<2>(address)); } + uint32_t read32(uint32_t address) { return read<4>(address); } + void write8(uint32_t address, uint8_t value) { write<1>(address, value); } + void write16(uint32_t address, uint16_t value) { write<2>(address, value); } + void write32(uint32_t address, uint32_t value) { write<4>(address, value); } + + private: + template + uint32_t read(uint32_t address) + { + const uint32_t physical = address & 0x1FFFFFFFu; + uint32_t value = 0; + if (physical < 0x00800000u) + std::memcpy(&value, &m_ram[physical & (kRamBytes - Bytes)], Bytes); + else if ((physical & ~0x3FFu) == 0x1F800000u) + std::memcpy(&value, &m_scratchpad[physical & (0x400u - Bytes)], Bytes); + else + value = m_io.ioRead(physical, Bytes); + return value; + } + + template + void write(uint32_t address, uint32_t value) + { + const uint32_t physical = address & 0x1FFFFFFFu; + if (physical < 0x00800000u) + std::memcpy(&m_ram[physical & (kRamBytes - Bytes)], &value, Bytes); + else if ((physical & ~0x3FFu) == 0x1F800000u) + std::memcpy(&m_scratchpad[physical & (0x400u - Bytes)], &value, Bytes); + else + m_io.ioWrite(physical, value, Bytes); + } + + IoDevice &m_io; + std::vector m_ram; + std::vector m_scratchpad; + }; +} diff --git a/ps2xIOP/src/lle/cpu.cpp b/ps2xIOP/src/lle/cpu.cpp new file mode 100644 index 000000000..75350a939 --- /dev/null +++ b/ps2xIOP/src/lle/cpu.cpp @@ -0,0 +1,247 @@ +#include "cpu.h" +#include "bus.h" + +namespace ps2x::iop::lle +{ + namespace + { + inline int32_t signExtend16(uint32_t value) { return static_cast(value & 0xFFFFu); } + } + + StopReason Cpu::run(CpuContext &c, uint64_t &cycles, uint64_t budget) + { + uint32_t *const r = c.gpr; + while (cycles < budget) + { + const uint32_t pc = c.pc; + const uint32_t op = m_bus.fetch(pc); + ++cycles; + + // Leaving a delay slot takes the branch; otherwise fall through. + uint32_t next = pc + 4u; + if (c.branchPending) + { + next = c.branchTarget; + c.branchPending = false; + } + + // A load issued by the previous instruction lands after this one + // reads its operands. + const uint8_t delayedRegister = c.loadRegister; + const uint32_t delayedValue = c.loadValue; + c.loadRegister = 0; + + const uint32_t rs = (op >> 21) & 31u; + const uint32_t rt = (op >> 16) & 31u; + const uint32_t rd = (op >> 11) & 31u; + const uint32_t imm = op & 0xFFFFu; + const int32_t simm = signExtend16(op); + uint32_t writeRegister = 0; + uint32_t writeValue = 0; + const auto set = [&](uint32_t reg, uint32_t value) { + writeRegister = reg; + writeValue = value; + }; + const auto branch = [&](bool taken, uint32_t target) { + if (taken) + { + c.branchPending = true; + c.branchTarget = target; + } + }; + const auto load = [&](uint32_t reg, uint32_t value) { + if (reg != 0u) + { + c.loadRegister = static_cast(reg); + c.loadValue = value; + } + }; + const uint32_t address = r[rs] + static_cast(simm); + + switch (op >> 26) + { + case 0x00: + switch (op & 63u) + { + case 0x00: set(rd, r[rt] << ((op >> 6) & 31u)); break; // SLL + case 0x02: set(rd, r[rt] >> ((op >> 6) & 31u)); break; // SRL + case 0x03: set(rd, static_cast(static_cast(r[rt]) >> ((op >> 6) & 31u))); break; // SRA + case 0x04: set(rd, r[rt] << (r[rs] & 31u)); break; // SLLV + case 0x06: set(rd, r[rt] >> (r[rs] & 31u)); break; // SRLV + case 0x07: set(rd, static_cast(static_cast(r[rt]) >> (r[rs] & 31u))); break; // SRAV + case 0x08: branch(true, r[rs]); break; // JR + case 0x09: // JALR + set(rd, pc + 8u); + branch(true, r[rs]); + break; + case 0x0C: // SYSCALL + m_syscallCode = (op >> 6) & 0xFFFFFu; + c.pc = next; + if (delayedRegister != 0u) + r[delayedRegister] = delayedValue; + return StopReason::Syscall; + case 0x0D: break; // BREAK + case 0x10: set(rd, c.hi); break; // MFHI + case 0x11: c.hi = r[rs]; break; // MTHI + case 0x12: set(rd, c.lo); break; // MFLO + case 0x13: c.lo = r[rs]; break; // MTLO + case 0x18: // MULT + { + const int64_t product = static_cast(static_cast(r[rs])) * + static_cast(r[rt]); + c.lo = static_cast(product); + c.hi = static_cast(static_cast(product) >> 32); + break; + } + case 0x19: // MULTU + { + const uint64_t product = static_cast(r[rs]) * r[rt]; + c.lo = static_cast(product); + c.hi = static_cast(product >> 32); + break; + } + case 0x1A: // DIV + { + const int32_t n = static_cast(r[rs]), d = static_cast(r[rt]); + if (d == 0) + { + c.lo = n >= 0 ? 0xFFFFFFFFu : 1u; + c.hi = static_cast(n); + } + else if (n == INT32_MIN && d == -1) + { + c.lo = static_cast(INT32_MIN); + c.hi = 0; + } + else + { + c.lo = static_cast(n / d); + c.hi = static_cast(n % d); + } + break; + } + case 0x1B: // DIVU + if (r[rt] == 0u) + { + c.lo = 0xFFFFFFFFu; + c.hi = r[rs]; + } + else + { + c.lo = r[rs] / r[rt]; + c.hi = r[rs] % r[rt]; + } + break; + // Overflow traps are never relied on by IOP drivers; wrap like ADDU. + case 0x20: case 0x21: set(rd, r[rs] + r[rt]); break; // ADD, ADDU + case 0x22: case 0x23: set(rd, r[rs] - r[rt]); break; // SUB, SUBU + case 0x24: set(rd, r[rs] & r[rt]); break; // AND + case 0x25: set(rd, r[rs] | r[rt]); break; // OR + case 0x26: set(rd, r[rs] ^ r[rt]); break; // XOR + case 0x27: set(rd, ~(r[rs] | r[rt])); break; // NOR + case 0x2A: set(rd, static_cast(r[rs]) < static_cast(r[rt]) ? 1u : 0u); break; // SLT + case 0x2B: set(rd, r[rs] < r[rt] ? 1u : 0u); break; // SLTU + default: + m_faultPc = pc; + return StopReason::Fault; + } + break; + case 0x01: // REGIMM + { + const bool less = static_cast(r[rs]) < 0; + const uint32_t target = pc + 4u + (static_cast(simm) << 2); + const bool link = (rt & 0x1Eu) == 0x10u; + if (link) + set(31, pc + 8u); + branch((rt & 1u) ? !less : less, target); // BLTZ, BGEZ(AL) + break; + } + case 0x02: branch(true, (pc & 0xF0000000u) | ((op & 0x03FFFFFFu) << 2)); break; // J + case 0x03: // JAL + set(31, pc + 8u); + branch(true, (pc & 0xF0000000u) | ((op & 0x03FFFFFFu) << 2)); + break; + case 0x04: branch(r[rs] == r[rt], pc + 4u + (static_cast(simm) << 2)); break; // BEQ + case 0x05: branch(r[rs] != r[rt], pc + 4u + (static_cast(simm) << 2)); break; // BNE + case 0x06: branch(static_cast(r[rs]) <= 0, pc + 4u + (static_cast(simm) << 2)); break; // BLEZ + case 0x07: branch(static_cast(r[rs]) > 0, pc + 4u + (static_cast(simm) << 2)); break; // BGTZ + case 0x08: case 0x09: set(rt, r[rs] + static_cast(simm)); break; // ADDI, ADDIU + case 0x0A: set(rt, static_cast(r[rs]) < simm ? 1u : 0u); break; // SLTI + case 0x0B: set(rt, r[rs] < static_cast(simm) ? 1u : 0u); break; // SLTIU + case 0x0C: set(rt, r[rs] & imm); break; // ANDI + case 0x0D: set(rt, r[rs] | imm); break; // ORI + case 0x0E: set(rt, r[rs] ^ imm); break; // XORI + case 0x0F: set(rt, imm << 16); break; // LUI + case 0x10: // COP0 + if (rs == 0x00u) + load(rt, rd == 12u ? m_status : 0u); // MFC0 + else if (rs == 0x04u && rd == 12u) + m_status = r[rt]; // MTC0 SR + else if (rs == 0x10u && (op & 63u) == 0x10u) + m_status = (m_status & ~0xFu) | ((m_status >> 2) & 0xFu); // RFE + break; + case 0x20: load(rt, static_cast(static_cast(m_bus.read8(address)))); break; // LB + case 0x21: load(rt, static_cast(static_cast(m_bus.read16(address)))); break; // LH + case 0x22: // LWL + { + const uint32_t word = m_bus.read32(address & ~3u); + const uint32_t shift = (address & 3u) * 8u; + const uint32_t current = delayedRegister == rt ? delayedValue : r[rt]; + const uint32_t mask = shift == 24u ? 0u : 0x00FFFFFFu >> shift; + load(rt, (current & mask) | (word << (24u - shift))); + break; + } + case 0x23: load(rt, m_bus.read32(address)); break; // LW + case 0x24: load(rt, m_bus.read8(address)); break; // LBU + case 0x25: load(rt, m_bus.read16(address)); break; // LHU + case 0x26: // LWR + { + const uint32_t word = m_bus.read32(address & ~3u); + const uint32_t shift = (address & 3u) * 8u; + const uint32_t current = delayedRegister == rt ? delayedValue : r[rt]; + const uint32_t mask = shift == 0u ? 0u : 0xFFFFFF00u << (24u - shift); + load(rt, (current & mask) | (word >> shift)); + break; + } + case 0x28: m_bus.write8(address, static_cast(r[rt])); break; // SB + case 0x29: m_bus.write16(address, static_cast(r[rt])); break; // SH + case 0x2A: // SWL + { + const uint32_t aligned = address & ~3u; + const uint32_t shift = (address & 3u) * 8u; + const uint32_t word = m_bus.read32(aligned); + const uint32_t mask = shift == 24u ? 0u : 0xFFFFFF00u << shift; + m_bus.write32(aligned, (word & mask) | (r[rt] >> (24u - shift))); + break; + } + case 0x2B: m_bus.write32(address, r[rt]); break; // SW + case 0x2E: // SWR + { + const uint32_t aligned = address & ~3u; + const uint32_t shift = (address & 3u) * 8u; + const uint32_t word = m_bus.read32(aligned); + const uint32_t mask = shift == 0u ? 0u : 0x00FFFFFFu >> (24u - shift); + m_bus.write32(aligned, (word & mask) | (r[rt] << shift)); + break; + } + case 0x2F: case 0x33: break; // CACHE, PREF + default: + m_faultPc = pc; + return StopReason::Fault; + } + + // An ALU result written in a load delay slot wins over the load. + if (delayedRegister != 0u) + r[delayedRegister] = delayedValue; + if (writeRegister != 0u) + { + r[writeRegister] = writeValue; + if (c.loadRegister == writeRegister) + c.loadRegister = 0; + } + r[0] = 0; + c.pc = next; + } + return StopReason::Budget; + } +} diff --git a/ps2xIOP/src/lle/cpu.h b/ps2xIOP/src/lle/cpu.h new file mode 100644 index 000000000..189890cc7 --- /dev/null +++ b/ps2xIOP/src/lle/cpu.h @@ -0,0 +1,51 @@ +#pragma once + +#include + +namespace ps2x::iop::lle +{ + class Bus; + + // One R3000A register context. Every IOP thread and the interrupt + // context own one; the scheduler swaps them in and out of the core. + struct CpuContext + { + uint32_t gpr[32]{}; + uint32_t hi = 0; + uint32_t lo = 0; + uint32_t pc = 0; + uint32_t branchTarget = 0; + bool branchPending = false; + // R3000A loads land one instruction late. + uint8_t loadRegister = 0; + uint32_t loadValue = 0; + }; + + enum class StopReason : uint8_t + { + Budget, // cycle budget used up + Syscall, // a SYSCALL trapped; the code is in syscallCode() + Fault, // an instruction the IOP sound modules never use + }; + + class Cpu + { + public: + explicit Cpu(Bus &bus) noexcept : m_bus(bus) {} + + // Run until the budget is spent or a trap needs the kernel. Returns + // with pc already past the trapping instruction. + StopReason run(CpuContext &context, uint64_t &cycles, uint64_t budget); + + uint32_t syscallCode() const noexcept { return m_syscallCode; } + uint32_t faultPc() const noexcept { return m_faultPc; } + uint32_t cop0Status() const noexcept { return m_status; } + void setCop0Status(uint32_t status) noexcept { m_status = status; } + + private: + Bus &m_bus; + uint32_t m_syscallCode = 0; + uint32_t m_faultPc = 0; + uint32_t m_status = 0; + }; +} diff --git a/ps2xIOP/src/lle/iop.cpp b/ps2xIOP/src/lle/iop.cpp new file mode 100644 index 000000000..b02d849c6 --- /dev/null +++ b/ps2xIOP/src/lle/iop.cpp @@ -0,0 +1,569 @@ +#include "iop.h" + +#include +#include + +namespace ps2x::iop::lle +{ + namespace + { + constexpr uint32_t kTrapBase = 0x1000u; // return traps, one word each + constexpr uint32_t kCtypeBase = 0x1100u; // sysclib's character class table + constexpr uint32_t kInterruptStack = 0x8000u; // top of the interrupt context stack + constexpr uint32_t kHeapBase = 0x10000u; + constexpr uint32_t kSpuDmaLines[2] = {36u, 40u}; // INUM_DMA_4, INUM_DMA_7 + constexpr uint32_t kSpuLine = 9u; + + uint32_t syscallWord(uint32_t code) { return 0x0000000Cu | (code << 6); } + } + + Iop::Iop(EeLink &ee) : m_ee(ee), m_bus(*this), m_cpu(m_bus), m_spu2(*this) + { + installKernel(); + uint8_t *ram = m_bus.ram(); + // Threads and handlers return here; the trap tells the kernel they are done. + m_returnTrap = kTrapBase; + const uint32_t returnWord = syscallWord(handlerFor("", 1)); + std::memcpy(ram + kTrapBase, &returnWord, 4); + m_rpcTrap = kTrapBase + 8u; + const uint32_t rpcWord = syscallWord(handlerFor("", 2)); + std::memcpy(ram + m_rpcTrap, &rpcWord, 4); + // The ctype table sysclib hands out: bit 2 digit, 1 lower, 0 upper, 3 space, 6 hex. + m_ctypeTable = kCtypeBase; + for (uint32_t ch = 0; ch < 256u; ++ch) + { + uint8_t flags = 0; + if (ch >= '0' && ch <= '9') flags |= 0x04u; + if (ch >= 'a' && ch <= 'z') flags |= 0x02u; + if (ch >= 'A' && ch <= 'Z') flags |= 0x01u; + if (ch == ' ' || (ch >= 9u && ch <= 13u)) flags |= 0x08u; + if ((ch >= 'a' && ch <= 'f') || (ch >= 'A' && ch <= 'F')) flags |= 0x40u; + ram[kCtypeBase + 1u + ch] = flags; + } + m_allocations[0] = kHeapBase; // everything below the heap is the kernel's + } + + Iop::~Iop() = default; + + std::string Iop::readString(uint32_t address, size_t limit) + { + std::string text; + for (size_t index = 0; index < limit; ++index) + { + const uint8_t ch = m_bus.read8(address + static_cast(index)); + if (ch == 0u) + break; + text.push_back(static_cast(ch)); + } + return text; + } + + uint32_t Iop::arg(CpuContext &c, unsigned index) + { + return index < 4u ? c.gpr[4 + index] : m_bus.read32(c.gpr[29] + 16u + (index - 4u) * 4u); + } + + uint32_t Iop::allocate(uint32_t size) + { + size = (size + 255u) & ~255u; + uint32_t cursor = kHeapBase; + for (const auto &[address, length] : m_allocations) + { + if (address >= cursor && address - cursor >= size) + break; + cursor = std::max(cursor, (address + length + 255u) & ~255u); + } + if (cursor + size > Bus::kRamBytes) + return 0; + m_allocations[cursor] = size; + return cursor; + } + + void Iop::release(uint32_t address) + { + if (address >= kHeapBase) + m_allocations.erase(address); + } + + void Iop::writeMemory(uint32_t address, const void *data, uint32_t size) + { + address &= Bus::kRamBytes - 1u; + size = std::min(size, Bus::kRamBytes - address); + std::memcpy(m_bus.ram() + address, data, size); + } + + void Iop::readMemory(uint32_t address, void *data, uint32_t size) + { + address &= Bus::kRamBytes - 1u; + size = std::min(size, Bus::kRamBytes - address); + std::memcpy(data, m_bus.ram() + address, size); + } + + // ------------------------------------------------------------------ hardware + + uint32_t Iop::ioRead(uint32_t physical, unsigned bytes) + { + if (physical >= 0x1F900000u && physical < 0x1F900800u) + { + uint32_t value = m_spu2.read(physical & 0x7FFu); + if (bytes == 4u) + value |= static_cast(m_spu2.read((physical + 2u) & 0x7FFu)) << 16; + return value; + } + int channel = 0; + if (uint32_t *reg = dmaRegister(physical, channel)) + return *reg >> ((physical & 3u) * 8u); + return 0; + } + + uint32_t *Iop::dmaRegister(uint32_t physical, int &channel) + { + for (channel = 0; channel < 2; ++channel) + { + const uint32_t base = channel == 0 ? 0x1F8010C0u : 0x1F801500u; + if (physical >= base && physical < base + 12u) + { + DmaChannel &dma = m_dma[channel]; + const uint32_t offset = physical - base; + return offset < 4u ? &dma.madr : offset < 8u ? &dma.bcr : &dma.chcr; + } + } + return nullptr; + } + + void Iop::ioWrite(uint32_t physical, uint32_t value, unsigned bytes) + { + if (physical >= 0x1F900000u && physical < 0x1F900800u) + { + m_spu2.write(physical & 0x7FFu, static_cast(value)); + if (bytes == 4u) + m_spu2.write((physical + 2u) & 0x7FFu, static_cast(value >> 16)); + return; + } + // libsd writes the block count and size as separate halfwords. + int channel = 0; + if (uint32_t *reg = dmaRegister(physical, channel)) + { + const uint32_t shift = (physical & 3u) * 8u; + const uint32_t mask = (bytes >= 4u ? 0xFFFFFFFFu : (1u << (bytes * 8u)) - 1u) << shift; + *reg = (*reg & ~mask) | ((value << shift) & mask); + DmaChannel &dma = m_dma[channel]; + dma.madr &= 0x00FFFFFFu; + if (reg == &dma.chcr && (mask & value << shift & 0x01000000u) != 0u) + dmaTransfer(channel); + } + } + + void Iop::dmaTransfer(int channel) + { + DmaChannel &dma = m_dma[channel]; + const uint32_t blockWords = dma.bcr & 0xFFFFu; + const uint32_t blocks = std::max(dma.bcr >> 16, 1u); + const uint32_t bytes = blockWords * blocks * 4u; + m_spu2.dmaStarted(channel); + if ((dma.chcr & 1u) != 0u && m_spu2.autoDmaEnabled(channel)) + { + // AutoDMA: the SPU2 pulls 1 KiB at a time as it plays. + dma.remaining = bytes; + return; + } + const uint32_t address = dma.madr & (Bus::kRamBytes - 1u); + const uint32_t length = std::min(bytes, Bus::kRamBytes - address); + auto *words = reinterpret_cast(m_bus.ram() + address); + if ((dma.chcr & 1u) != 0u) + m_spu2.dmaWrite(channel, words, length / 2u); + else + m_spu2.dmaRead(channel, words, length / 2u); + dma.madr = (dma.madr + length) & 0x00FFFFFFu; + dma.chcr &= ~0x01000000u; + raiseInterrupt(kSpuDmaLines[channel]); + } + + bool Iop::fetchAutoDma(int core, uint16_t *block) + { + DmaChannel &dma = m_dma[core]; + if ((dma.chcr & 0x01000000u) == 0u || dma.remaining < 1024u) + return false; + readMemory(dma.madr, block, 1024u); + dma.madr = (dma.madr + 1024u) & 0x00FFFFFFu; + dma.remaining -= 1024u; + if (dma.remaining < 1024u) + { + dma.chcr &= ~0x01000000u; + raiseInterrupt(kSpuDmaLines[core]); + } + return true; + } + + void Iop::raiseSpuInterrupt() { raiseInterrupt(kSpuLine); } + + void Iop::raiseInterrupt(uint32_t line) + { + if (std::find(m_pendingInterrupts.begin(), m_pendingInterrupts.end(), line) == m_pendingInterrupts.end()) + m_pendingInterrupts.push_back(line); + } + + // ------------------------------------------------------------------ execution + + void Iop::call(uint32_t function, std::initializer_list arguments, uint32_t *result, uint32_t gp) + { + CpuContext context{}; + context.pc = function; + unsigned index = 4; + for (const uint32_t value : arguments) + context.gpr[index++] = value; + context.gpr[28] = gp; + context.gpr[29] = kInterruptStack - 16u; + context.gpr[31] = m_returnTrap; + const bool nested = m_inInterrupt; + m_inInterrupt = true; + // Handlers run between samples, off the clock like RPC work. + uint64_t spent = 0; + const uint64_t limit = 4'000'000u; + for (;;) + { + const StopReason reason = m_cpu.run(context, spent, limit); + if (reason != StopReason::Syscall) + { + m_ee.log("[iop] handler at " + std::to_string(function) + " did not return"); + break; + } + const uint32_t code = m_cpu.syscallCode(); + if (code < m_handlers.size() && m_handlers[code] == &Iop::functionReturn) + break; + context.pc = context.gpr[31]; + if (code < m_handlers.size()) + (this->*m_handlers[code])(context); + } + m_inInterrupt = nested; + if (result) + *result = context.gpr[2]; + } + + void Iop::serviceInterrupts() + { + if (m_intrSuspend > 0) + return; + while (!m_pendingInterrupts.empty()) + { + const uint32_t line = m_pendingInterrupts.front(); + m_pendingInterrupts.erase(m_pendingInterrupts.begin()); + const auto found = m_interrupts.find(line); + if (found == m_interrupts.end() || !found->second.enabled || found->second.handler == 0u) + continue; + uint32_t result = 0; + call(found->second.handler, {found->second.argument}, &result, found->second.gp); + // A handler returning zero leaves its line disabled. + if (result == 0u) + found->second.enabled = false; + reschedule(); + } + } + + void Iop::serviceEvents() + { + // Threads sleeping in DelayThread. + for (auto &[id, thread] : m_threads) + if (thread.wait == Wait::Delay && thread.wakeCycle <= m_cycle) + makeReady(thread, 0u); + + // Alarms: the handler returns the next interval, or zero to stop. + for (auto it = m_alarms.begin(); it != m_alarms.end();) + { + if (it->second.cycle > m_cycle) + { + ++it; + continue; + } + const Alarm alarm = it->second; + it = m_alarms.erase(it); + uint32_t next = 0; + call(alarm.handler, {alarm.argument}, &next, alarm.gp); + if (next != 0u) + m_alarms.emplace(alarm.handler, Alarm{m_cycle + next, alarm.handler, alarm.argument, alarm.gp}); + reschedule(); + } + + // Hardware timers: the handler returns the next compare value, or zero to stop. + for (auto &timer : m_timers) + { + if (!timer.running || timer.handler == 0u || timer.next > m_cycle) + continue; + uint32_t compare = 0; + call(timer.handler, {timer.argument}, &compare, timer.gp); + if (compare == 0u) + timer.running = false; + else + { + timer.compare = compare; + timer.next = m_cycle + static_cast(compare) * timer.prescale; + } + reschedule(); + } + } + + Iop::Thread *Iop::pickThread() + { + Thread *best = nullptr; + for (auto &[id, thread] : m_threads) + { + if (thread.dormant || thread.wait != Wait::None) + continue; + if (!best || thread.priority < best->priority || + (thread.priority == best->priority && thread.readySequence < best->readySequence)) + best = &thread; + } + return best; + } + + void Iop::makeReady(Thread &thread, uint32_t result) + { + thread.wait = Wait::None; + thread.context.gpr[2] = result; + thread.readySequence = ++m_readySequence; + reschedule(); + } + + void Iop::block(Wait wait, int32_t id) + { + Thread *thread = current(); + if (!thread || m_inInterrupt) + { + m_ee.log("[iop] blocking call outside a thread"); + return; + } + thread->wait = wait; + thread->waitId = id; + reschedule(); + } + + bool Iop::runThreads(uint64_t budget, bool advanceClock) + { + // Only the sample clock moves time. Work done off it, answering an + // RPC the moment it arrives, would otherwise run the timers fast. + uint64_t offClock = 0; + uint64_t &clock = advanceClock ? m_cycle : offClock; + bool ran = false; + const uint64_t end = clock + budget; + while (clock < end) + { + serviceInterrupts(); + Thread *thread = m_intrSuspend > 0 && current() && current()->wait == Wait::None ? current() : pickThread(); + if (!thread) + break; + m_current = thread->id; + m_reschedule = false; + ran = true; + while (!m_reschedule && clock < end) + { + const StopReason reason = m_cpu.run(thread->context, clock, end); + if (reason == StopReason::Budget) + break; + if (reason == StopReason::Fault) + { + m_ee.log("[iop] thread " + std::to_string(thread->id) + " stopped on an unsupported instruction"); + thread->dormant = true; + reschedule(); + break; + } + const uint32_t code = m_cpu.syscallCode(); + thread->context.pc = thread->context.gpr[31]; + if (code < m_handlers.size()) + (this->*m_handlers[code])(thread->context); + if (thread->dormant || thread->wait != Wait::None) + break; + } + deliverRequests(); + } + m_current = 0; + return ran; + } + + void Iop::settle() + { + deliverRequests(); + // Enough for any command handler; threads spinning on hardware stop here. + runThreads(2'000'000u, false); + } + + void Iop::step(int16_t &left, int16_t &right) + { + const uint64_t target = (m_cycle / kCyclesPerSample + 1u) * kCyclesPerSample; + serviceEvents(); + deliverRequests(); + runThreads(target > m_cycle ? target - m_cycle : 0u); + m_cycle = std::max(m_cycle, target); + m_spu2.tick(left, right); + serviceInterrupts(); + } + + // ------------------------------------------------------------------ RPC + + bool Iop::hasServer(uint32_t sid) const { return m_servers.count(sid) != 0u; } + + bool Iop::server(uint32_t sid, uint32_t &record, uint32_t &buffer) const + { + const auto found = m_servers.find(sid); + if (found == m_servers.end()) + return false; + record = found->second.record; + buffer = found->second.buffer; + return true; + } + + void Iop::submit(std::shared_ptr request) { m_requests.push_back(std::move(request)); } + + void Iop::deliverRequests() + { + for (auto it = m_requests.begin(); it != m_requests.end();) + { + const auto server = m_servers.find((*it)->sid); + if (server == m_servers.end()) + { + (*it)->done = true; // nobody serves it; the EE side falls back + it = m_requests.erase(it); + continue; + } + const auto owner = m_threads.find(server->second.thread); + if (owner == m_threads.end() || owner->second.wait != Wait::Rpc || m_active.count(owner->first) != 0u) + { + ++it; + continue; + } + Thread &thread = owner->second; + const Request &request = **it; + if (server->second.buffer != 0u && !request.send.empty()) + writeMemory(server->second.buffer, request.send.data(), static_cast(request.send.size())); + // Call the server function on top of the loop's own frame. + m_saved[thread.id] = thread.context; + CpuContext &c = thread.context; + c.pc = server->second.function; + c.gpr[4] = request.function; + c.gpr[5] = server->second.buffer; + c.gpr[6] = static_cast(request.send.size()); + c.gpr[29] -= 64u; + c.gpr[31] = m_rpcTrap; + c.branchPending = false; + c.loadRegister = 0; + m_active[thread.id] = *it; + makeReady(thread, 0u); + it = m_requests.erase(it); + } + } + + void Iop::finishRequest(Thread &thread, uint32_t resultAddress) + { + const auto active = m_active.find(thread.id); + if (active == m_active.end()) + return; + Request &request = *active->second; + if (request.receiveSize != 0u && request.receiveAddress != 0u && resultAddress != 0u) + { + std::vector reply(request.receiveSize); + readMemory(resultAddress, reply.data(), request.receiveSize); + m_ee.writeEe(request.receiveAddress, reply.data(), request.receiveSize); + } + request.done = true; + m_active.erase(active); + thread.context = m_saved[thread.id]; + thread.wait = Wait::Rpc; + reschedule(); + } + + // ------------------------------------------------------------------ modules + + void Iop::linkImports(const std::vector &imports) + { + for (const IrxImport &import : imports) + { + uint32_t word = 0; + const auto exported = m_exports.find(import.library); + if (exported != m_exports.end() && import.ordinal < exported->second.size() && + exported->second[import.ordinal] != 0u) + word = 0x08000000u | ((exported->second[import.ordinal] >> 2) & 0x03FFFFFFu); // j target + else + word = syscallWord(handlerFor(import.library, import.ordinal)); + m_bus.write32(import.stubAddress, word); + } + } + + int32_t Iop::loadModule(const std::vector &file, const std::string &path, + const std::vector &arguments) + { + IrxImage irx; + std::string error; + if (!parseIrx(file, irx, error)) + { + m_ee.log("[iop] " + path + ": " + error); + return -1; + } + const uint32_t size = static_cast(irx.image.size()) + irx.bssSize; + const uint32_t base = allocate(size); + std::vector imports; + if (base == 0u || !relocateIrx(irx, m_bus.ram(), Bus::kRamBytes, base, imports, error)) + { + m_ee.log("[iop] " + path + ": " + (base == 0u ? std::string("out of IOP memory") : error)); + release(base); + return -1; + } + linkImports(imports); + + // argv: the path followed by the arguments, as C strings. + std::vector argv{path}; + argv.insert(argv.end(), arguments.begin(), arguments.end()); + uint32_t bytes = static_cast(argv.size() + 1u) * 4u; + for (const auto &text : argv) + bytes += static_cast(text.size()) + 1u; + const uint32_t block = allocate(bytes); + uint32_t strings = block + static_cast(argv.size() + 1u) * 4u; + for (size_t index = 0; index < argv.size(); ++index) + { + m_bus.write32(block + static_cast(index) * 4u, strings); + writeMemory(strings, argv[index].c_str(), static_cast(argv[index].size()) + 1u); + strings += static_cast(argv[index].size()) + 1u; + } + m_bus.write32(block + static_cast(argv.size()) * 4u, 0u); + + // The start function runs in a thread of its own, like MODLOAD's. + Thread loader; + loader.id = m_nextId++; + loader.priority = loader.initialPriority = 8; + loader.stackSize = 0x2000u; + loader.stackBase = allocate(loader.stackSize); + loader.gp = base + irx.gp; + loader.dormant = false; + loader.context.pc = base + irx.entry; + loader.context.gpr[4] = static_cast(argv.size()); + loader.context.gpr[5] = block; + loader.context.gpr[28] = loader.gp; + loader.context.gpr[29] = loader.stackBase + loader.stackSize - 16u; + loader.context.gpr[31] = m_returnTrap; + loader.readySequence = ++m_readySequence; + const int32_t loaderId = loader.id; + m_threads[loaderId] = loader; + m_loaderResult = 0xFFFFFFFFu; + m_loaderThread = loaderId; + for (int round = 0; round < 64 && m_threads.count(loaderId) && !m_threads.at(loaderId).dormant; ++round) + runThreads(2'000'000u); + const uint32_t result = m_loaderResult; + m_loaderThread = 0; + if (m_threads.count(loaderId)) + { + release(m_threads.at(loaderId).stackBase); + m_threads.erase(loaderId); + } + release(block); + + Module module{m_nextId++, irx.name, base, size}; + if ((result & 3u) == 1u) + { + // NO_RESIDENT_END: the module is done and leaves nothing behind. + release(base); + m_ee.log("[iop] " + irx.name + " did not stay resident"); + return module.id; + } + m_modules.push_back(module); + m_ee.log("[iop] started " + irx.name + " at " + std::to_string(base)); + return module.id; + } +} diff --git a/ps2xIOP/src/lle/iop.h b/ps2xIOP/src/lle/iop.h new file mode 100644 index 000000000..e5806885a --- /dev/null +++ b/ps2xIOP/src/lle/iop.h @@ -0,0 +1,228 @@ +#pragma once + +#include "bus.h" +#include "cpu.h" +#include "irx.h" +#include "spu2.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ps2x::iop::lle +{ + // What the emulated IOP needs from the EE side. + class EeLink + { + public: + virtual ~EeLink() = default; + virtual bool readEe(uint32_t address, void *destination, uint32_t size) = 0; + virtual bool writeEe(uint32_t address, const void *source, uint32_t size) = 0; + // An RPC server the EE has registered, for IOP modules that bind back. + virtual bool eeServer(uint32_t sid) = 0; + virtual void log(const std::string &message) = 0; + }; + + // An IOP running original IRX modules on an R3000A interpreter with a + // small high-level kernel underneath. Time is counted in IOP cycles + // (36.864 MHz) and advances as the SPU2 produces 48 kHz samples. + class Iop final : public IoDevice, public Spu2Host + { + public: + static constexpr uint64_t kClock = 36'864'000u; + static constexpr uint32_t kCyclesPerSample = 768u; + + explicit Iop(EeLink &ee); + ~Iop() override; + + // Load and start an IRX. Returns a positive module id, or a negative error. + int32_t loadModule(const std::vector &file, const std::string &path, + const std::vector &arguments); + + // EE-side SIF RPC into a server registered by an IOP module. + bool hasServer(uint32_t sid) const; + bool server(uint32_t sid, uint32_t &record, uint32_t &buffer) const; + struct Request + { + uint32_t sid = 0; + uint32_t function = 0; + std::vector send; + uint32_t receiveAddress = 0; // EE address the reply goes to + uint32_t receiveSize = 0; + bool done = false; + }; + // Queue a request; the server thread handles it as the IOP runs. + void submit(std::shared_ptr request); + + // EE DMA into IOP memory (sceSifSetDma with an IOP destination). + void writeMemory(uint32_t address, const void *data, uint32_t size); + void readMemory(uint32_t address, void *data, uint32_t size); + uint32_t allocate(uint32_t size); + void release(uint32_t address); + + // Run until every ready thread has blocked, without advancing time. + void settle(); + // Advance time by one output sample and produce it. + void step(int16_t &left, int16_t &right); + + // IoDevice + uint32_t ioRead(uint32_t physical, unsigned bytes) override; + void ioWrite(uint32_t physical, uint32_t value, unsigned bytes) override; + // Spu2Host + bool fetchAutoDma(int core, uint16_t *block) override; + void raiseSpuInterrupt() override; + + private: + enum class Wait : uint8_t + { + None, + Sleep, + Delay, + Semaphore, + EventFlag, + Mailbox, + Rpc, + }; + + struct Thread + { + int32_t id = 0; + uint32_t entry = 0, stackBase = 0, stackSize = 0, gp = 0, attr = 0, option = 0; + int32_t priority = 0, initialPriority = 0; + CpuContext context; + bool dormant = true; + Wait wait = Wait::None; + int32_t waitId = 0; + uint32_t waitBits = 0, waitMode = 0, waitResult = 0; + uint64_t wakeCycle = 0; + int32_t wakeupCount = 0; + uint64_t readySequence = 0; + }; + + struct Semaphore + { + int32_t count = 0, maximum = 0; + uint32_t attr = 0, option = 0; + }; + + struct EventFlag + { + uint32_t bits = 0, attr = 0, option = 0; + }; + + struct Mailbox + { + uint32_t attr = 0, option = 0; + std::deque messages; + }; + + struct Alarm + { + uint64_t cycle = 0; + uint32_t handler = 0, argument = 0, gp = 0; + }; + + struct HardTimer + { + bool allocated = false, running = false; + uint32_t source = 0, prescale = 1, compare = 0, handler = 0, argument = 0, gp = 0; + uint64_t start = 0, next = 0; + }; + + struct Interrupt + { + uint32_t handler = 0, argument = 0, mode = 0, gp = 0; + bool enabled = false; + }; + + struct Server + { + uint32_t sid = 0, function = 0, buffer = 0, queue = 0, record = 0; + int32_t thread = 0; + }; + + struct DmaChannel + { + uint32_t madr = 0, bcr = 0, chcr = 0; + uint32_t remaining = 0; // bytes left for AutoDMA + }; + + struct Module + { + int32_t id = 0; + std::string name; + uint32_t base = 0, size = 0; + }; + + using Handler = void (Iop::*)(CpuContext &); + + // Kernel + void installKernel(); + uint32_t handlerFor(const std::string &library, uint16_t ordinal); + void linkImports(const std::vector &imports); + void call(uint32_t function, std::initializer_list arguments, uint32_t *result, uint32_t gp); + bool runThreads(uint64_t budget, bool advanceClock = true); + Thread *current() { return m_current ? &m_threads.at(m_current) : nullptr; } + void makeReady(Thread &thread, uint32_t result); + void block(Wait wait, int32_t id); + void reschedule() { m_reschedule = true; } + Thread *pickThread(); + void deliverRequests(); + void finishRequest(Thread &thread, uint32_t resultAddress); + void raiseInterrupt(uint32_t line); + void serviceEvents(); + void serviceInterrupts(); + void dmaTransfer(int channel); + uint32_t *dmaRegister(uint32_t physical, int &channel); + uint32_t arg(CpuContext &c, unsigned index); + void ret(CpuContext &c, uint32_t value) { c.gpr[2] = value; } + std::string readString(uint32_t address, size_t limit = 256); + +#define PS2X_IOP_HANDLER(name) void name(CpuContext &c); +#include "kernel_handlers.inc" +#undef PS2X_IOP_HANDLER + + EeLink &m_ee; + Bus m_bus; + Cpu m_cpu; + Spu2 m_spu2; + uint64_t m_cycle = 0; + + std::vector m_handlers; + std::unordered_map> m_exports; + std::map m_threads; + std::map m_semaphores; + std::map m_eventFlags; + std::map m_mailboxes; + std::multimap m_alarms; // keyed by handler + std::array m_timers{}; + std::map m_interrupts; + std::vector m_pendingInterrupts; + std::map m_servers; + std::map m_queues; // RPC queue record -> owning thread + std::deque> m_requests; + std::map> m_active; // by server thread + std::map m_saved; // server loop frames under a call + std::array m_dma{}; // channels 4 and 7 + std::vector m_modules; + std::map m_allocations; // address -> size + int32_t m_nextId = 1; + int32_t m_current = 0; + uint64_t m_readySequence = 0; + bool m_reschedule = false; + bool m_inInterrupt = false; + int32_t m_intrSuspend = 0; + uint32_t m_ctypeTable = 0; + uint32_t m_returnTrap = 0; + uint32_t m_rpcTrap = 0; + int32_t m_loaderThread = 0; + uint32_t m_loaderResult = 0; + uint32_t m_sifDmaId = 0; + }; +} diff --git a/ps2xIOP/src/lle/irx.cpp b/ps2xIOP/src/lle/irx.cpp new file mode 100644 index 000000000..c51da9afc --- /dev/null +++ b/ps2xIOP/src/lle/irx.cpp @@ -0,0 +1,193 @@ +#include "irx.h" + +#include + +namespace ps2x::iop::lle +{ + namespace + { + constexpr uint16_t kIrxType = 0xFF80u; + constexpr uint32_t kIopModSegment = 0x70000080u; + constexpr uint32_t kLoadSegment = 1u; + constexpr uint32_t kRelocationSection = 9u; + constexpr uint32_t kImportMagic = 0x41E00000u; + + template + bool readAt(const std::vector &data, size_t offset, T &value) + { + if (offset > data.size() || data.size() - offset < sizeof(T)) + return false; + std::memcpy(&value, data.data() + offset, sizeof(T)); + return true; + } + + uint32_t load32(const uint8_t *ram, uint32_t offset) + { + uint32_t value; + std::memcpy(&value, ram + offset, sizeof(value)); + return value; + } + + void store32(uint8_t *ram, uint32_t offset, uint32_t value) + { + std::memcpy(ram + offset, &value, sizeof(value)); + } + } + + bool parseIrx(const std::vector &file, IrxImage &out, std::string &error) + { + uint16_t type = 0, phentsize = 0, phnum = 0, shentsize = 0, shnum = 0; + uint32_t phoff = 0, shoff = 0; + if (file.size() < 0x34u || std::memcmp(file.data(), "\x7F" "ELF", 4) != 0 || + !readAt(file, 0x10, type) || type != kIrxType || !readAt(file, 0x1C, phoff) || + !readAt(file, 0x20, shoff) || !readAt(file, 0x2A, phentsize) || !readAt(file, 0x2C, phnum) || + !readAt(file, 0x2E, shentsize) || !readAt(file, 0x30, shnum)) + { + error = "not an IRX module"; + return false; + } + + uint32_t loadOffset = 0, loadFileSize = 0, loadMemorySize = 0; + bool haveModule = false, haveLoad = false; + for (uint32_t index = 0; index < phnum; ++index) + { + const size_t header = phoff + static_cast(index) * phentsize; + uint32_t segmentType = 0, offset = 0, fileSize = 0, memorySize = 0; + if (!readAt(file, header, segmentType) || !readAt(file, header + 4, offset) || + !readAt(file, header + 16, fileSize) || !readAt(file, header + 20, memorySize)) + break; + if (segmentType == kIopModSegment) + { + uint32_t fields[6]; + for (uint32_t field = 0; field < 6u; ++field) + if (!readAt(file, offset + field * 4u, fields[field])) + break; + readAt(file, offset + 24u, out.version); + out.entry = fields[1]; + out.gp = fields[2]; + out.textSize = fields[3]; + out.dataSize = fields[4]; + out.bssSize = fields[5]; + for (size_t at = offset + 26u; at < file.size() && at < offset + fileSize && file[at] != 0u; ++at) + out.name.push_back(static_cast(file[at])); + haveModule = true; + } + else if (segmentType == kLoadSegment) + { + loadOffset = offset; + loadFileSize = fileSize; + loadMemorySize = memorySize; + haveLoad = true; + } + } + if (!haveModule || !haveLoad || loadOffset > file.size() || file.size() - loadOffset < loadFileSize || + loadMemorySize < loadFileSize) + { + error = "IRX is missing its .iopmod or load segment"; + return false; + } + out.image.assign(file.begin() + loadOffset, file.begin() + loadOffset + loadFileSize); + out.bssSize = loadMemorySize - loadFileSize; + + for (uint32_t index = 0; index < shnum; ++index) + { + const size_t header = shoff + static_cast(index) * shentsize; + uint32_t sectionType = 0, offset = 0, size = 0; + if (!readAt(file, header + 4, sectionType) || !readAt(file, header + 16, offset) || + !readAt(file, header + 20, size)) + break; + if (sectionType != kRelocationSection) + continue; + for (uint32_t entry = 0; entry + 8u <= size; entry += 8u) + { + uint32_t target = 0, info = 0; + if (!readAt(file, offset + entry, target) || !readAt(file, offset + entry + 4u, info)) + { + error = "truncated relocation table"; + return false; + } + out.relocations.push_back({target, static_cast(info & 0xFFu)}); + } + } + return true; + } + + bool relocateIrx(const IrxImage &irx, uint8_t *ram, uint32_t ramSize, uint32_t base, + std::vector &imports, std::string &error) + { + const uint32_t total = static_cast(irx.image.size()) + irx.bssSize; + if (base >= ramSize || ramSize - base < total) + { + error = "IRX does not fit in IOP memory"; + return false; + } + uint8_t *image = ram + base; + std::memcpy(image, irx.image.data(), irx.image.size()); + std::memset(image + irx.image.size(), 0, irx.bssSize); + + // HI16 entries wait for the LO16 that completes their address. + std::vector pendingHigh; + for (const auto &relocation : irx.relocations) + { + if (relocation.offset + 4u > irx.image.size()) + { + error = "relocation outside the module"; + return false; + } + const uint32_t word = load32(image, relocation.offset); + switch (relocation.type) + { + case 2: // R_MIPS_32 + store32(image, relocation.offset, word + base); + break; + case 4: // R_MIPS_26 + { + const uint32_t target = ((word & 0x03FFFFFFu) << 2) + base; + store32(image, relocation.offset, (word & 0xFC000000u) | ((target >> 2) & 0x03FFFFFFu)); + break; + } + case 5: // R_MIPS_HI16 + pendingHigh.push_back(relocation.offset); + break; + case 6: // R_MIPS_LO16 + { + const int32_t low = static_cast(word & 0xFFFFu); + for (const uint32_t highOffset : pendingHigh) + { + const uint32_t high = load32(image, highOffset); + const uint32_t value = ((high & 0xFFFFu) << 16) + static_cast(low) + base; + const uint32_t adjusted = ((value >> 16) + ((value & 0x8000u) ? 1u : 0u)) & 0xFFFFu; + store32(image, highOffset, (high & 0xFFFF0000u) | adjusted); + } + pendingHigh.clear(); + store32(image, relocation.offset, (word & 0xFFFF0000u) | ((static_cast(low) + base) & 0xFFFFu)); + break; + } + case 7: // R_MIPS_GPREL16: gp moves with the module + break; + default: + error = "unsupported relocation type " + std::to_string(relocation.type); + return false; + } + } + + for (uint32_t offset = 0; offset + 20u <= irx.textSize && offset + 20u <= irx.image.size(); offset += 4u) + { + if (load32(image, offset) != kImportMagic) + continue; + std::string library; + for (uint32_t index = 0; index < 8u && image[offset + 12u + index] != 0u; ++index) + library.push_back(static_cast(image[offset + 12u + index])); + uint32_t stub = offset + 20u; + for (; stub + 8u <= irx.textSize; stub += 8u) + { + const uint32_t jump = load32(image, stub), slot = load32(image, stub + 4u); + if (jump != 0x03E00008u || (slot & 0xFFFF0000u) != 0x24000000u) + break; + imports.push_back({library, static_cast(slot & 0xFFFFu), base + stub}); + } + offset = stub - 4u; + } + return true; + } +} diff --git a/ps2xIOP/src/lle/irx.h b/ps2xIOP/src/lle/irx.h new file mode 100644 index 000000000..050b7e6d4 --- /dev/null +++ b/ps2xIOP/src/lle/irx.h @@ -0,0 +1,42 @@ +#pragma once + +#include +#include +#include + +namespace ps2x::iop::lle +{ + // One stub from an IRX import table: `jr $ra; addiu $zero, $zero, ordinal`. + struct IrxImport + { + std::string library; + uint16_t ordinal = 0; + uint32_t stubAddress = 0; // after relocation + }; + + struct IrxImage + { + std::string name; + uint16_t version = 0; + uint32_t entry = 0; // relative to the load address + uint32_t gp = 0; // relative to the load address + uint32_t textSize = 0; + uint32_t dataSize = 0; + uint32_t bssSize = 0; + std::vector image; // text and data, not yet relocated + struct Relocation + { + uint32_t offset; + uint8_t type; + }; + std::vector relocations; + }; + + // Parse an IRX (an ELF of type 0xFF80 with an .iopmod section). + bool parseIrx(const std::vector &file, IrxImage &out, std::string &error); + + // Copy the image to memory at base, apply relocations and zero BSS. + // Returns the import stubs found in the relocated text. + bool relocateIrx(const IrxImage &irx, uint8_t *ram, uint32_t ramSize, uint32_t base, + std::vector &imports, std::string &error); +} diff --git a/ps2xIOP/src/lle/kernel.cpp b/ps2xIOP/src/lle/kernel.cpp new file mode 100644 index 000000000..4bc649568 --- /dev/null +++ b/ps2xIOP/src/lle/kernel.cpp @@ -0,0 +1,1013 @@ +#include "iop.h" + +#include +#include +#include + +namespace ps2x::iop::lle +{ + namespace + { + // About 100 us, a SIF command to the EE and its answer. + constexpr uint64_t kSifRoundTrip = Iop::kClock / 10'000u; + + struct Binding + { + const char *library; + uint16_t ordinal; + const char *handler; + }; + + // Library ordinals as the IOP kernel exports them. + constexpr Binding kBindings[] = { + {"sysmem", 4, "allocSysMemory"}, {"sysmem", 5, "freeSysMemory"}, + {"sysmem", 7, "queryMaxFreeMemSize"}, {"sysmem", 8, "queryTotalFreeMemSize"}, + {"sysmem", 14, "kprintf"}, + {"loadcore", 4, "noop"}, {"loadcore", 5, "noop"}, + {"loadcore", 6, "registerLibraryEntries"}, {"loadcore", 7, "releaseLibraryEntries"}, + {"intrman", 4, "registerIntrHandler"}, {"intrman", 5, "releaseIntrHandler"}, + {"intrman", 6, "enableIntr"}, {"intrman", 7, "disableIntr"}, + {"intrman", 17, "cpuSuspendIntr"}, {"intrman", 18, "cpuResumeIntr"}, + {"intrman", 23, "queryIntrContext"}, + {"stdio", 4, "kprintf"}, + {"sysclib", 8, "lookCtypeTable"}, {"sysclib", 11, "memcmpHandler"}, + {"sysclib", 12, "memcpyHandler"}, {"sysclib", 13, "memmoveHandler"}, + {"sysclib", 14, "memsetHandler"}, {"sysclib", 17, "bzeroHandler"}, + {"sysclib", 22, "strcmpHandler"}, {"sysclib", 23, "strcpyHandler"}, + {"sysclib", 27, "strlenHandler"}, {"sysclib", 29, "strncmpHandler"}, + {"sysclib", 30, "strncpyHandler"}, {"sysclib", 36, "strtolHandler"}, + {"thbase", 4, "createThread"}, {"thbase", 5, "deleteThread"}, {"thbase", 6, "startThread"}, + {"thbase", 8, "exitThread"}, {"thbase", 10, "terminateThread"}, {"thbase", 11, "terminateThread"}, + {"thbase", 14, "changeThreadPriority"}, {"thbase", 15, "changeThreadPriority"}, + {"thbase", 16, "rotateThreadReadyQueue"}, {"thbase", 17, "rotateThreadReadyQueue"}, + {"thbase", 20, "getThreadId"}, {"thbase", 22, "referThreadStatus"}, + {"thbase", 23, "referThreadStatus"}, {"thbase", 24, "sleepThread"}, + {"thbase", 25, "wakeupThread"}, {"thbase", 26, "wakeupThread"}, + {"thbase", 27, "cancelWakeupThread"}, {"thbase", 28, "cancelWakeupThread"}, + {"thbase", 33, "delayThread"}, {"thbase", 34, "getSystemTime"}, + {"thbase", 35, "setAlarm"}, {"thbase", 36, "setAlarm"}, + {"thbase", 37, "cancelAlarm"}, {"thbase", 38, "cancelAlarm"}, + {"thbase", 39, "usecToSysClock"}, {"thbase", 40, "sysClockToUsec"}, + {"thsemap", 4, "createSema"}, {"thsemap", 5, "deleteSema"}, {"thsemap", 6, "signalSema"}, + {"thsemap", 7, "signalSema"}, {"thsemap", 8, "waitSema"}, {"thsemap", 9, "pollSema"}, + {"thsemap", 10, "pollSema"}, {"thsemap", 11, "referSemaStatus"}, {"thsemap", 12, "referSemaStatus"}, + {"thevent", 4, "createEventFlag"}, {"thevent", 5, "deleteEventFlag"}, + {"thevent", 6, "setEventFlag"}, {"thevent", 7, "setEventFlag"}, + {"thevent", 8, "clearEventFlag"}, {"thevent", 9, "clearEventFlag"}, + {"thevent", 10, "waitEventFlag"}, {"thevent", 11, "pollEventFlag"}, + {"thmsgbx", 4, "createMbx"}, {"thmsgbx", 5, "deleteMbx"}, {"thmsgbx", 6, "sendMbx"}, + {"thmsgbx", 7, "sendMbx"}, {"thmsgbx", 8, "receiveMbx"}, {"thmsgbx", 9, "pollMbx"}, + {"sifman", 5, "noop"}, {"sifman", 7, "sifSetDma"}, {"sifman", 8, "sifDmaStat"}, + {"sifman", 29, "sifCheckInit"}, + {"sifcmd", 4, "noop"}, {"sifcmd", 14, "noop"}, {"sifcmd", 15, "sifBindRpc"}, + {"sifcmd", 16, "sifCallRpc"}, {"sifcmd", 17, "sifRegisterRpc"}, {"sifcmd", 19, "sifSetRpcQueue"}, + {"sifcmd", 22, "sifRpcLoop"}, {"sifcmd", 23, "sifGetOtherData"}, {"sifcmd", 24, "sifRemoveRpc"}, + {"sifcmd", 25, "noop"}, + {"timrman", 4, "allocHardTimer"}, {"timrman", 6, "freeHardTimer"}, + {"timrman", 10, "getTimerCounter"}, {"timrman", 20, "setTimerHandler"}, + {"timrman", 22, "setupHardTimer"}, {"timrman", 23, "startHardTimer"}, + {"timrman", 24, "stopHardTimer"}, + }; + + struct Named + { + const char *name; + void (Iop::*handler)(CpuContext &); + }; + } + + void Iop::installKernel() + { +#define PS2X_IOP_HANDLER(name) m_handlers.push_back(&Iop::name); +#include "kernel_handlers.inc" +#undef PS2X_IOP_HANDLER + } + + uint32_t Iop::handlerFor(const std::string &library, uint16_t ordinal) + { + static const std::vector names = { +#define PS2X_IOP_HANDLER(name) #name, +#include "kernel_handlers.inc" +#undef PS2X_IOP_HANDLER + }; + const auto index = [&](const char *name) { + return static_cast(std::find(names.begin(), names.end(), name) - names.begin()); + }; + if (library.empty()) + return index(ordinal == 1u ? "functionReturn" : "rpcReturn"); + for (const Binding &binding : kBindings) + if (library == binding.library && ordinal == binding.ordinal) + return index(binding.handler); + m_ee.log("[iop] no kernel service for " + library + " #" + std::to_string(ordinal)); + return index("unknownImport"); + } + + void Iop::unknownImport(CpuContext &c) { ret(c, 0); } + void Iop::noop(CpuContext &c) { ret(c, 0); } + + void Iop::functionReturn(CpuContext &c) + { + // A thread's entry function returned: that is ExitThread. + Thread *thread = current(); + if (!thread) + return; + if (thread->id == m_loaderThread) + m_loaderResult = c.gpr[2]; + thread->dormant = true; + reschedule(); + } + + void Iop::rpcReturn(CpuContext &c) + { + if (Thread *thread = current()) + finishRequest(*thread, c.gpr[2]); + } + + // ---------------------------------------------------------------- sysmem, loadcore, stdio + + void Iop::allocSysMemory(CpuContext &c) { ret(c, allocate(arg(c, 1))); } + + void Iop::freeSysMemory(CpuContext &c) + { + release(arg(c, 0)); + ret(c, 0); + } + + void Iop::queryMaxFreeMemSize(CpuContext &c) + { + uint32_t largest = 0, cursor = 0; + for (const auto &[address, length] : m_allocations) + { + largest = std::max(largest, address > cursor ? address - cursor : 0u); + cursor = std::max(cursor, address + length); + } + ret(c, std::max(largest, Bus::kRamBytes - cursor)); + } + + void Iop::queryTotalFreeMemSize(CpuContext &c) + { + uint32_t used = 0; + for (const auto &[address, length] : m_allocations) + used += length; + ret(c, Bus::kRamBytes - used); + } + + void Iop::kprintf(CpuContext &c) + { + // Only the format string; enough to follow what the drivers report. + std::string text = readString(arg(c, 0)); + while (!text.empty() && (text.back() == '\n' || text.back() == '\r')) + text.pop_back(); + if (!text.empty()) + m_ee.log("[iop] " + text); + ret(c, 0); + } + + void Iop::registerLibraryEntries(CpuContext &c) + { + const uint32_t table = arg(c, 0); + std::string name; + for (uint32_t index = 0; index < 8u; ++index) + { + const uint8_t ch = m_bus.read8(table + 12u + index); + if (ch == 0u) + break; + name.push_back(static_cast(ch)); + } + std::vector functions; + for (uint32_t entry = table + 20u; entry < table + 20u + 1024u; entry += 4u) + { + const uint32_t address = m_bus.read32(entry); + if (address == 0u) + break; + functions.push_back(address); + } + m_exports[name] = std::move(functions); + ret(c, 0); + } + + void Iop::releaseLibraryEntries(CpuContext &c) { ret(c, 0); } + + // ---------------------------------------------------------------- intrman + + void Iop::registerIntrHandler(CpuContext &c) + { + Interrupt &line = m_interrupts[arg(c, 0)]; + line.mode = arg(c, 1); + line.handler = arg(c, 2); + line.argument = arg(c, 3); + line.gp = c.gpr[28]; + ret(c, 0); + } + + void Iop::releaseIntrHandler(CpuContext &c) + { + m_interrupts.erase(arg(c, 0)); + ret(c, 0); + } + + void Iop::enableIntr(CpuContext &c) + { + m_interrupts[arg(c, 0) & 0xFFu].enabled = true; + ret(c, 0); + } + + void Iop::disableIntr(CpuContext &c) + { + auto &line = m_interrupts[arg(c, 0) & 0xFFu]; + if (arg(c, 1) != 0u) + m_bus.write32(arg(c, 1), line.enabled ? arg(c, 0) : 0u); + line.enabled = false; + ret(c, 0); + } + + void Iop::cpuSuspendIntr(CpuContext &c) + { + if (arg(c, 0) != 0u) + m_bus.write32(arg(c, 0), m_intrSuspend > 0 ? 0u : 1u); + ++m_intrSuspend; + ret(c, m_intrSuspend > 1 ? static_cast(-102) : 0u); // KE_CPUDI when already off + } + + void Iop::cpuResumeIntr(CpuContext &c) + { + if (arg(c, 0) != 0u && m_intrSuspend > 0) + m_intrSuspend = 0; + else if (m_intrSuspend > 0) + --m_intrSuspend; + ret(c, 0); + } + + void Iop::queryIntrContext(CpuContext &c) { ret(c, m_inInterrupt ? 1u : 0u); } + + // ---------------------------------------------------------------- sysclib + + void Iop::lookCtypeTable(CpuContext &c) + { + // Returns the class of one character, as the table lookup would. + ret(c, m_bus.read8(m_ctypeTable + 1u + (arg(c, 0) & 0xFFu))); + } + + void Iop::memcpyHandler(CpuContext &c) + { + const uint32_t destination = arg(c, 0), source = arg(c, 1), size = arg(c, 2); + for (uint32_t index = 0; index < size; ++index) + m_bus.write8(destination + index, m_bus.read8(source + index)); + ret(c, destination); + } + + void Iop::memmoveHandler(CpuContext &c) + { + const uint32_t destination = arg(c, 0), source = arg(c, 1), size = arg(c, 2); + std::vector copy(size); + for (uint32_t index = 0; index < size; ++index) + copy[index] = m_bus.read8(source + index); + for (uint32_t index = 0; index < size; ++index) + m_bus.write8(destination + index, copy[index]); + ret(c, destination); + } + + void Iop::memsetHandler(CpuContext &c) + { + const uint32_t destination = arg(c, 0), size = arg(c, 2); + for (uint32_t index = 0; index < size; ++index) + m_bus.write8(destination + index, static_cast(arg(c, 1))); + ret(c, destination); + } + + void Iop::memcmpHandler(CpuContext &c) + { + int32_t result = 0; + for (uint32_t index = 0; index < arg(c, 2) && result == 0; ++index) + result = static_cast(m_bus.read8(arg(c, 0) + index)) - m_bus.read8(arg(c, 1) + index); + ret(c, static_cast(result)); + } + + void Iop::bzeroHandler(CpuContext &c) + { + for (uint32_t index = 0; index < arg(c, 1); ++index) + m_bus.write8(arg(c, 0) + index, 0u); + ret(c, 0); + } + + void Iop::strlenHandler(CpuContext &c) { ret(c, static_cast(readString(arg(c, 0), 65536).size())); } + + void Iop::strcmpHandler(CpuContext &c) + { + ret(c, static_cast(readString(arg(c, 0), 65536).compare(readString(arg(c, 1), 65536)))); + } + + void Iop::strncmpHandler(CpuContext &c) + { + const uint32_t count = arg(c, 2); + const std::string a = readString(arg(c, 0), count), b = readString(arg(c, 1), count); + ret(c, static_cast(a.compare(b))); + } + + void Iop::strcpyHandler(CpuContext &c) + { + const std::string text = readString(arg(c, 1), 65536); + writeMemory(arg(c, 0), text.c_str(), static_cast(text.size()) + 1u); + ret(c, arg(c, 0)); + } + + void Iop::strncpyHandler(CpuContext &c) + { + const uint32_t count = arg(c, 2); + std::string text = readString(arg(c, 1), count); + text.resize(count, '\0'); + writeMemory(arg(c, 0), text.data(), count); + ret(c, arg(c, 0)); + } + + void Iop::strtolHandler(CpuContext &c) + { + const uint32_t start = arg(c, 0); + const std::string text = readString(start, 64); + char *end = nullptr; + const long value = std::strtol(text.c_str(), &end, static_cast(arg(c, 2))); + if (arg(c, 1) != 0u) + m_bus.write32(arg(c, 1), start + static_cast(end - text.c_str())); + ret(c, static_cast(value)); + } + + // ---------------------------------------------------------------- threads + + void Iop::createThread(CpuContext &c) + { + const uint32_t param = arg(c, 0); + Thread thread; + thread.id = m_nextId++; + thread.attr = m_bus.read32(param); + thread.option = m_bus.read32(param + 4u); + thread.entry = m_bus.read32(param + 8u); + thread.stackSize = (m_bus.read32(param + 12u) + 15u) & ~15u; + thread.priority = thread.initialPriority = static_cast(m_bus.read32(param + 16u)); + thread.stackBase = allocate(thread.stackSize); + thread.gp = c.gpr[28]; + if (thread.stackBase == 0u) + { + ret(c, static_cast(-400)); // KE_NO_MEMORY + return; + } + m_threads[thread.id] = thread; + ret(c, static_cast(thread.id)); + } + + void Iop::deleteThread(CpuContext &c) + { + const auto found = m_threads.find(static_cast(arg(c, 0))); + if (found == m_threads.end() || !found->second.dormant) + { + ret(c, static_cast(-407)); + return; + } + release(found->second.stackBase); + m_threads.erase(found); + ret(c, 0); + } + + void Iop::startThread(CpuContext &c) + { + const auto found = m_threads.find(static_cast(arg(c, 0))); + if (found == m_threads.end()) + { + ret(c, static_cast(-407)); + return; + } + Thread &thread = found->second; + thread.context = {}; + thread.context.pc = thread.entry; + thread.context.gpr[4] = arg(c, 1); + thread.context.gpr[28] = thread.gp; + thread.context.gpr[29] = thread.stackBase + thread.stackSize - 16u; + thread.context.gpr[31] = m_returnTrap; + thread.priority = thread.initialPriority; + thread.dormant = false; + thread.wakeupCount = 0; + makeReady(thread, 0u); + ret(c, 0); + } + + void Iop::exitThread(CpuContext &c) + { + (void)c; + if (Thread *thread = current()) + { + thread->dormant = true; + reschedule(); + } + } + + void Iop::terminateThread(CpuContext &c) + { + const auto found = m_threads.find(static_cast(arg(c, 0))); + if (found != m_threads.end()) + { + found->second.dormant = true; + found->second.wait = Wait::None; + m_active.erase(found->first); + } + ret(c, 0); + } + + void Iop::changeThreadPriority(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_threads.find(id == 0 ? m_current : id); + if (found != m_threads.end()) + { + const int32_t priority = static_cast(arg(c, 1)); + found->second.priority = priority == 0 ? found->second.initialPriority : priority; + reschedule(); + } + ret(c, 0); + } + + void Iop::rotateThreadReadyQueue(CpuContext &c) + { + if (Thread *thread = current()) + thread->readySequence = ++m_readySequence; + reschedule(); + ret(c, 0); + } + + void Iop::getThreadId(CpuContext &c) { ret(c, static_cast(m_current)); } + + void Iop::referThreadStatus(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_threads.find(id == 0 ? m_current : id); + const uint32_t info = arg(c, 1); + if (found == m_threads.end() || info == 0u) + { + ret(c, static_cast(-407)); + return; + } + const Thread &thread = found->second; + // struct _iop_thread_status: attr, option, status, entry, stack, stackSize, + // gpReg, initPriority, currentPriority, waitType, waitId, wakeupCount, ... + const uint32_t status = thread.dormant ? 0x10u : thread.wait != Wait::None ? 0x04u + : thread.id == m_current ? 0x01u : 0x02u; + const uint32_t fields[12] = {thread.attr, thread.option, status, thread.entry, thread.stackBase, + thread.stackSize, thread.gp, static_cast(thread.initialPriority), + static_cast(thread.priority), 0u, static_cast(thread.waitId), + static_cast(thread.wakeupCount)}; + for (uint32_t index = 0; index < 12u; ++index) + m_bus.write32(info + index * 4u, fields[index]); + ret(c, 0); + } + + void Iop::sleepThread(CpuContext &c) + { + Thread *thread = current(); + if (thread && thread->wakeupCount > 0) + { + --thread->wakeupCount; + ret(c, 0); + return; + } + ret(c, 0); + block(Wait::Sleep, 0); + } + + void Iop::wakeupThread(CpuContext &c) + { + const auto found = m_threads.find(static_cast(arg(c, 0))); + if (found == m_threads.end()) + { + ret(c, static_cast(-407)); + return; + } + if (found->second.wait == Wait::Sleep) + makeReady(found->second, 0u); + else + ++found->second.wakeupCount; + ret(c, 0); + } + + void Iop::cancelWakeupThread(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_threads.find(id == 0 ? m_current : id); + int32_t count = 0; + if (found != m_threads.end()) + { + count = found->second.wakeupCount; + found->second.wakeupCount = 0; + } + ret(c, static_cast(count)); + } + + void Iop::delayThread(CpuContext &c) + { + Thread *thread = current(); + if (!thread) + return; + thread->wakeCycle = m_cycle + static_cast(arg(c, 0)) * kClock / 1'000'000u; + ret(c, 0); + block(Wait::Delay, 0); + } + + void Iop::getSystemTime(CpuContext &c) + { + m_bus.write32(arg(c, 0), static_cast(m_cycle)); + m_bus.write32(arg(c, 0) + 4u, static_cast(m_cycle >> 32)); + ret(c, 0); + } + + void Iop::setAlarm(CpuContext &c) + { + // The clock argument points at an iop_sys_clock_t. + const uint64_t clock = m_bus.read32(arg(c, 0)) | (static_cast(m_bus.read32(arg(c, 0) + 4u)) << 32); + m_alarms.emplace(arg(c, 1), Alarm{m_cycle + clock, arg(c, 1), arg(c, 2), c.gpr[28]}); + ret(c, 0); + } + + void Iop::cancelAlarm(CpuContext &c) + { + const auto range = m_alarms.equal_range(arg(c, 0)); + for (auto it = range.first; it != range.second;) + it = it->second.argument == arg(c, 1) ? m_alarms.erase(it) : std::next(it); + ret(c, 0); + } + + void Iop::usecToSysClock(CpuContext &c) + { + const uint64_t clock = static_cast(arg(c, 0)) * kClock / 1'000'000u; + m_bus.write32(arg(c, 1), static_cast(clock)); + m_bus.write32(arg(c, 1) + 4u, static_cast(clock >> 32)); + ret(c, 0); + } + + void Iop::sysClockToUsec(CpuContext &c) + { + const uint64_t clock = m_bus.read32(arg(c, 0)) | (static_cast(m_bus.read32(arg(c, 0) + 4u)) << 32); + const uint64_t usec = clock * 1'000'000u / kClock; + m_bus.write32(arg(c, 1), static_cast(usec / 1'000'000u)); + m_bus.write32(arg(c, 2), static_cast(usec % 1'000'000u)); + ret(c, 0); + } + + // ---------------------------------------------------------------- semaphores + + void Iop::createSema(CpuContext &c) + { + const uint32_t param = arg(c, 0); + Semaphore semaphore; + semaphore.attr = m_bus.read32(param); + semaphore.option = m_bus.read32(param + 4u); + semaphore.count = static_cast(m_bus.read32(param + 8u)); + semaphore.maximum = static_cast(m_bus.read32(param + 12u)); + const int32_t id = m_nextId++; + m_semaphores[id] = semaphore; + ret(c, static_cast(id)); + } + + void Iop::deleteSema(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + for (auto &[threadId, thread] : m_threads) + if (thread.wait == Wait::Semaphore && thread.waitId == id) + makeReady(thread, static_cast(-425)); // KE_WAIT_DELETE + m_semaphores.erase(id); + ret(c, 0); + } + + void Iop::signalSema(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_semaphores.find(id); + if (found == m_semaphores.end()) + { + ret(c, static_cast(-408)); + return; + } + // Wake the longest-waiting thread, or count the signal. + Thread *waiter = nullptr; + for (auto &[threadId, thread] : m_threads) + if (thread.wait == Wait::Semaphore && thread.waitId == id && + (!waiter || thread.readySequence < waiter->readySequence)) + waiter = &thread; + if (waiter) + makeReady(*waiter, 0u); + else if (found->second.count < found->second.maximum || found->second.maximum == 0) + ++found->second.count; + ret(c, 0); + } + + void Iop::waitSema(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_semaphores.find(id); + if (found == m_semaphores.end()) + { + ret(c, static_cast(-408)); + return; + } + if (found->second.count > 0) + { + --found->second.count; + ret(c, 0); + return; + } + if (Thread *thread = current()) + thread->readySequence = ++m_readySequence; + ret(c, 0); + block(Wait::Semaphore, id); + } + + void Iop::pollSema(CpuContext &c) + { + const auto found = m_semaphores.find(static_cast(arg(c, 0))); + if (found == m_semaphores.end() || found->second.count <= 0) + { + ret(c, static_cast(-419)); // KE_SEMA_ZERO + return; + } + --found->second.count; + ret(c, 0); + } + + void Iop::referSemaStatus(CpuContext &c) + { + const auto found = m_semaphores.find(static_cast(arg(c, 0))); + if (found == m_semaphores.end()) + { + ret(c, static_cast(-408)); + return; + } + int32_t waiting = 0; + for (const auto &[threadId, thread] : m_threads) + waiting += thread.wait == Wait::Semaphore && thread.waitId == found->first ? 1 : 0; + const uint32_t info = arg(c, 1); + const uint32_t fields[6] = {found->second.attr, found->second.option, 0u, + static_cast(found->second.maximum), + static_cast(found->second.count), static_cast(waiting)}; + for (uint32_t index = 0; index < 6u; ++index) + m_bus.write32(info + index * 4u, fields[index]); + ret(c, 0); + } + + // ---------------------------------------------------------------- event flags + + void Iop::createEventFlag(CpuContext &c) + { + const uint32_t param = arg(c, 0); + EventFlag flag; + flag.attr = m_bus.read32(param); + flag.option = m_bus.read32(param + 4u); + flag.bits = m_bus.read32(param + 8u); + const int32_t id = m_nextId++; + m_eventFlags[id] = flag; + ret(c, static_cast(id)); + } + + void Iop::deleteEventFlag(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + for (auto &[threadId, thread] : m_threads) + if (thread.wait == Wait::EventFlag && thread.waitId == id) + makeReady(thread, static_cast(-425)); + m_eventFlags.erase(id); + ret(c, 0); + } + + void Iop::setEventFlag(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_eventFlags.find(id); + if (found == m_eventFlags.end()) + { + ret(c, static_cast(-409)); + return; + } + EventFlag &flag = found->second; + flag.bits |= arg(c, 1); + for (auto &[threadId, thread] : m_threads) + { + if (thread.wait != Wait::EventFlag || thread.waitId != id) + continue; + // Mode bit 0: OR (any bit) instead of AND; 0x10 clear all, 0x20 clear pattern. + const bool satisfied = (thread.waitMode & 1u) ? (flag.bits & thread.waitBits) != 0u + : (flag.bits & thread.waitBits) == thread.waitBits; + if (!satisfied) + continue; + if (thread.waitResult != 0u) + m_bus.write32(thread.waitResult, flag.bits); + if (thread.waitMode & 0x10u) + flag.bits = 0; + else if (thread.waitMode & 0x20u) + flag.bits &= ~thread.waitBits; + makeReady(thread, 0u); + } + ret(c, 0); + } + + void Iop::clearEventFlag(CpuContext &c) + { + const auto found = m_eventFlags.find(static_cast(arg(c, 0))); + if (found != m_eventFlags.end()) + found->second.bits &= arg(c, 1); + ret(c, 0); + } + + void Iop::waitEventFlag(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_eventFlags.find(id); + if (found == m_eventFlags.end()) + { + ret(c, static_cast(-409)); + return; + } + EventFlag &flag = found->second; + const uint32_t bits = arg(c, 1), mode = arg(c, 2), result = arg(c, 3); + const bool satisfied = (mode & 1u) ? (flag.bits & bits) != 0u : (flag.bits & bits) == bits; + if (satisfied) + { + if (result != 0u) + m_bus.write32(result, flag.bits); + if (mode & 0x10u) + flag.bits = 0; + else if (mode & 0x20u) + flag.bits &= ~bits; + ret(c, 0); + return; + } + if (Thread *thread = current()) + { + thread->waitBits = bits; + thread->waitMode = mode; + thread->waitResult = result; + } + ret(c, 0); + block(Wait::EventFlag, id); + } + + void Iop::pollEventFlag(CpuContext &c) + { + const auto found = m_eventFlags.find(static_cast(arg(c, 0))); + if (found == m_eventFlags.end()) + { + ret(c, static_cast(-409)); + return; + } + EventFlag &flag = found->second; + const uint32_t bits = arg(c, 1), mode = arg(c, 2), result = arg(c, 3); + const bool satisfied = (mode & 1u) ? (flag.bits & bits) != 0u : (flag.bits & bits) == bits; + if (!satisfied) + { + ret(c, static_cast(-421)); // KE_EVF_COND + return; + } + if (result != 0u) + m_bus.write32(result, flag.bits); + if (mode & 0x10u) + flag.bits = 0; + else if (mode & 0x20u) + flag.bits &= ~bits; + ret(c, 0); + } + + // ---------------------------------------------------------------- message boxes + + void Iop::createMbx(CpuContext &c) + { + Mailbox mailbox; + mailbox.attr = m_bus.read32(arg(c, 0)); + mailbox.option = m_bus.read32(arg(c, 0) + 4u); + const int32_t id = m_nextId++; + m_mailboxes[id] = mailbox; + ret(c, static_cast(id)); + } + + void Iop::deleteMbx(CpuContext &c) + { + m_mailboxes.erase(static_cast(arg(c, 0))); + ret(c, 0); + } + + void Iop::sendMbx(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 0)); + const auto found = m_mailboxes.find(id); + if (found == m_mailboxes.end()) + { + ret(c, static_cast(-410)); + return; + } + for (auto &[threadId, thread] : m_threads) + if (thread.wait == Wait::Mailbox && thread.waitId == id) + { + m_bus.write32(thread.waitResult, arg(c, 1)); + makeReady(thread, 0u); + ret(c, 0); + return; + } + found->second.messages.push_back(arg(c, 1)); + ret(c, 0); + } + + void Iop::receiveMbx(CpuContext &c) + { + const int32_t id = static_cast(arg(c, 1)); + const auto found = m_mailboxes.find(id); + if (found == m_mailboxes.end()) + { + ret(c, static_cast(-410)); + return; + } + if (!found->second.messages.empty()) + { + m_bus.write32(arg(c, 0), found->second.messages.front()); + found->second.messages.pop_front(); + ret(c, 0); + return; + } + if (Thread *thread = current()) + thread->waitResult = arg(c, 0); + ret(c, 0); + block(Wait::Mailbox, id); + } + + void Iop::pollMbx(CpuContext &c) + { + const auto found = m_mailboxes.find(static_cast(arg(c, 1))); + if (found == m_mailboxes.end() || found->second.messages.empty()) + { + ret(c, static_cast(-424)); // KE_MBOX_NOMSG + return; + } + m_bus.write32(arg(c, 0), found->second.messages.front()); + found->second.messages.pop_front(); + ret(c, 0); + } + + // ---------------------------------------------------------------- SIF + + void Iop::sifSetDma(CpuContext &c) + { + // Each descriptor: IOP source, EE destination, size, attributes. + const uint32_t list = arg(c, 0), count = arg(c, 1); + for (uint32_t index = 0; index < count; ++index) + { + const uint32_t entry = list + index * 16u; + const uint32_t source = m_bus.read32(entry), destination = m_bus.read32(entry + 4u); + const uint32_t size = m_bus.read32(entry + 8u); + std::vector data(size); + readMemory(source, data.data(), size); + m_ee.writeEe(destination, data.data(), size); + } + ret(c, ++m_sifDmaId); + } + + void Iop::sifDmaStat(CpuContext &c) { ret(c, static_cast(-1)); } // always complete + + void Iop::sifCheckInit(CpuContext &c) { ret(c, 1u); } + + void Iop::sifRegisterRpc(CpuContext &c) + { + Server server; + server.record = arg(c, 0); + server.sid = arg(c, 1); + server.function = arg(c, 2); + server.buffer = arg(c, 3); + server.queue = arg(c, 6); + for (const auto &[queue, owner] : m_queues) + if (queue == server.queue) + server.thread = owner; + m_servers[server.sid] = server; + m_ee.log("[iop] RPC server " + std::to_string(server.sid) + " registered"); + ret(c, 0); + } + + void Iop::sifSetRpcQueue(CpuContext &c) + { + m_queues[arg(c, 0)] = static_cast(arg(c, 1)); + ret(c, 0); + } + + void Iop::sifRpcLoop(CpuContext &c) + { + // The loop never returns; requests are delivered to the waiting thread. + (void)c; + block(Wait::Rpc, static_cast(arg(c, 0))); + } + + void Iop::sifRemoveRpc(CpuContext &c) + { + for (auto it = m_servers.begin(); it != m_servers.end();) + it = it->second.record == arg(c, 0) ? m_servers.erase(it) : std::next(it); + ret(c, 0); + } + + void Iop::sifGetOtherData(CpuContext &c) + { + // (record, EE source, IOP destination, size, mode) + const uint32_t source = arg(c, 1), destination = arg(c, 2), size = arg(c, 3); + std::vector data(size); + m_ee.readEe(source, data.data(), size); + writeMemory(destination, data.data(), size); + ret(c, 0); + } + + void Iop::sifBindRpc(CpuContext &c) + { + // Callers poll the server field until the EE has registered the id, + // some in a busy loop, so the call has to wait like the real round + // trip does or it starves every lower priority thread. + const uint32_t client = arg(c, 0); + for (uint32_t offset = 0; offset < 0x28u; offset += 4u) + m_bus.write32(client + offset, 0u); + // Only tested against zero here; it names EE memory the IOP never reads. + if (m_ee.eeServer(arg(c, 1))) + m_bus.write32(client + 0x24u, 1u); + ret(c, 0); + if (Thread *thread = current()) + { + thread->wakeCycle = m_cycle + kSifRoundTrip; + block(Wait::Delay, 0); + } + } + + void Iop::sifCallRpc(CpuContext &c) { ret(c, static_cast(-1)); } + + // ---------------------------------------------------------------- hardware timers + + void Iop::allocHardTimer(CpuContext &c) + { + // Timers 4 and 5 are the 32-bit ones drivers ask for; hand out any free one. + for (uint32_t index = 0; index < m_timers.size(); ++index) + if (!m_timers[index].allocated) + { + m_timers[index] = {}; + m_timers[index].allocated = true; + ret(c, index + 1u); + return; + } + ret(c, static_cast(-150)); + } + + void Iop::freeHardTimer(CpuContext &c) + { + const uint32_t index = arg(c, 0) - 1u; + if (index < m_timers.size()) + m_timers[index] = {}; + ret(c, 0); + } + + void Iop::setTimerHandler(CpuContext &c) + { + const uint32_t index = arg(c, 0) - 1u; + if (index < m_timers.size()) + { + m_timers[index].compare = arg(c, 1); + m_timers[index].handler = arg(c, 2); + m_timers[index].argument = arg(c, 3); + m_timers[index].gp = c.gpr[28]; + } + ret(c, 0); + } + + void Iop::setupHardTimer(CpuContext &c) + { + // (timer, source, mode, prescale): only the system clock source is used by drivers. + const uint32_t index = arg(c, 0) - 1u; + if (index < m_timers.size()) + { + m_timers[index].source = arg(c, 1); + m_timers[index].prescale = std::max(arg(c, 3), 1u); + } + ret(c, 0); + } + + void Iop::startHardTimer(CpuContext &c) + { + const uint32_t index = arg(c, 0) - 1u; + if (index < m_timers.size()) + { + HardTimer &timer = m_timers[index]; + timer.running = true; + timer.start = m_cycle; + timer.next = m_cycle + static_cast(std::max(timer.compare, 1u)) * timer.prescale; + } + ret(c, 0); + } + + void Iop::stopHardTimer(CpuContext &c) + { + const uint32_t index = arg(c, 0) - 1u; + if (index < m_timers.size()) + m_timers[index].running = false; + ret(c, 0); + } + + void Iop::getTimerCounter(CpuContext &c) + { + const uint32_t index = arg(c, 0) - 1u; + ret(c, index < m_timers.size() ? static_cast((m_cycle - m_timers[index].start) / + std::max(m_timers[index].prescale, 1u)) + : 0u); + } +} diff --git a/ps2xIOP/src/lle/kernel_handlers.inc b/ps2xIOP/src/lle/kernel_handlers.inc new file mode 100644 index 000000000..ac90ed265 --- /dev/null +++ b/ps2xIOP/src/lle/kernel_handlers.inc @@ -0,0 +1,84 @@ +// Kernel services the emulated IOP provides at a high level. Each entry is a +// member of Iop; kernel.cpp maps library ordinals onto them. +PS2X_IOP_HANDLER(unknownImport) +PS2X_IOP_HANDLER(functionReturn) +PS2X_IOP_HANDLER(rpcReturn) +PS2X_IOP_HANDLER(noop) +PS2X_IOP_HANDLER(allocSysMemory) +PS2X_IOP_HANDLER(freeSysMemory) +PS2X_IOP_HANDLER(queryMaxFreeMemSize) +PS2X_IOP_HANDLER(queryTotalFreeMemSize) +PS2X_IOP_HANDLER(kprintf) +PS2X_IOP_HANDLER(registerLibraryEntries) +PS2X_IOP_HANDLER(releaseLibraryEntries) +PS2X_IOP_HANDLER(registerIntrHandler) +PS2X_IOP_HANDLER(releaseIntrHandler) +PS2X_IOP_HANDLER(enableIntr) +PS2X_IOP_HANDLER(disableIntr) +PS2X_IOP_HANDLER(cpuSuspendIntr) +PS2X_IOP_HANDLER(cpuResumeIntr) +PS2X_IOP_HANDLER(queryIntrContext) +PS2X_IOP_HANDLER(lookCtypeTable) +PS2X_IOP_HANDLER(memcpyHandler) +PS2X_IOP_HANDLER(memmoveHandler) +PS2X_IOP_HANDLER(memsetHandler) +PS2X_IOP_HANDLER(memcmpHandler) +PS2X_IOP_HANDLER(bzeroHandler) +PS2X_IOP_HANDLER(strlenHandler) +PS2X_IOP_HANDLER(strcmpHandler) +PS2X_IOP_HANDLER(strncmpHandler) +PS2X_IOP_HANDLER(strcpyHandler) +PS2X_IOP_HANDLER(strncpyHandler) +PS2X_IOP_HANDLER(strtolHandler) +PS2X_IOP_HANDLER(createThread) +PS2X_IOP_HANDLER(deleteThread) +PS2X_IOP_HANDLER(startThread) +PS2X_IOP_HANDLER(exitThread) +PS2X_IOP_HANDLER(terminateThread) +PS2X_IOP_HANDLER(changeThreadPriority) +PS2X_IOP_HANDLER(rotateThreadReadyQueue) +PS2X_IOP_HANDLER(getThreadId) +PS2X_IOP_HANDLER(referThreadStatus) +PS2X_IOP_HANDLER(sleepThread) +PS2X_IOP_HANDLER(wakeupThread) +PS2X_IOP_HANDLER(cancelWakeupThread) +PS2X_IOP_HANDLER(delayThread) +PS2X_IOP_HANDLER(getSystemTime) +PS2X_IOP_HANDLER(setAlarm) +PS2X_IOP_HANDLER(cancelAlarm) +PS2X_IOP_HANDLER(usecToSysClock) +PS2X_IOP_HANDLER(sysClockToUsec) +PS2X_IOP_HANDLER(createSema) +PS2X_IOP_HANDLER(deleteSema) +PS2X_IOP_HANDLER(signalSema) +PS2X_IOP_HANDLER(waitSema) +PS2X_IOP_HANDLER(pollSema) +PS2X_IOP_HANDLER(referSemaStatus) +PS2X_IOP_HANDLER(createEventFlag) +PS2X_IOP_HANDLER(deleteEventFlag) +PS2X_IOP_HANDLER(setEventFlag) +PS2X_IOP_HANDLER(clearEventFlag) +PS2X_IOP_HANDLER(waitEventFlag) +PS2X_IOP_HANDLER(pollEventFlag) +PS2X_IOP_HANDLER(createMbx) +PS2X_IOP_HANDLER(deleteMbx) +PS2X_IOP_HANDLER(sendMbx) +PS2X_IOP_HANDLER(receiveMbx) +PS2X_IOP_HANDLER(pollMbx) +PS2X_IOP_HANDLER(sifSetDma) +PS2X_IOP_HANDLER(sifDmaStat) +PS2X_IOP_HANDLER(sifCheckInit) +PS2X_IOP_HANDLER(sifRegisterRpc) +PS2X_IOP_HANDLER(sifSetRpcQueue) +PS2X_IOP_HANDLER(sifRpcLoop) +PS2X_IOP_HANDLER(sifRemoveRpc) +PS2X_IOP_HANDLER(sifGetOtherData) +PS2X_IOP_HANDLER(sifBindRpc) +PS2X_IOP_HANDLER(sifCallRpc) +PS2X_IOP_HANDLER(allocHardTimer) +PS2X_IOP_HANDLER(freeHardTimer) +PS2X_IOP_HANDLER(setTimerHandler) +PS2X_IOP_HANDLER(setupHardTimer) +PS2X_IOP_HANDLER(startHardTimer) +PS2X_IOP_HANDLER(stopHardTimer) +PS2X_IOP_HANDLER(getTimerCounter) diff --git a/ps2xIOP/src/lle/spu2.cpp b/ps2xIOP/src/lle/spu2.cpp new file mode 100644 index 000000000..d2853cb1b --- /dev/null +++ b/ps2xIOP/src/lle/spu2.cpp @@ -0,0 +1,692 @@ +#include "spu2.h" + +#include +#include + +namespace ps2x::iop::lle +{ + namespace + { + constexpr uint32_t kAddressMask = Spu2::kRamWords - 1u; + constexpr int32_t kFilters[16][2] = { + {0, 0}, {60, 0}, {115, -52}, {98, -55}, {122, -60}, + }; + + inline int32_t clamp16(int32_t value) { return std::clamp(value, -0x8000, 0x7FFF); } + inline int32_t scale(int32_t value, int32_t volume) { return (value * volume) >> 15; } + + // Core register offsets, relative to the core's 0x400 byte window. + enum : uint32_t + { + kPitchModL = 0x180, kPitchModH = 0x182, kNoiseL = 0x184, kNoiseH = 0x186, + kDryLL = 0x188, kDryLH = 0x18A, kWetLL = 0x18C, kWetLH = 0x18E, + kDryRL = 0x190, kDryRH = 0x192, kWetRL = 0x194, kWetRH = 0x196, + kMmix = 0x198, kAttr = 0x19A, kIrqAH = 0x19C, kIrqAL = 0x19E, + kKeyOnL = 0x1A0, kKeyOnH = 0x1A2, kKeyOffL = 0x1A4, kKeyOffH = 0x1A6, + kTsaH = 0x1A8, kTsaL = 0x1AA, kData = 0x1AC, kAdmas = 0x1B0, + kVoiceAddresses = 0x1C0, kEsaH = 0x2E0, kEsaL = 0x2E2, + kReverbAddresses = 0x2E4, kEea = 0x33C, kEndxL = 0x340, kEndxH = 0x342, kStatx = 0x344, + }; + // Per-core volume block: 0x760 for core 0, 0x788 for core 1. AVOL is + // the external input (core 0's output, on core 1), BVOL the core's + // own sound data input fed by AutoDMA. + enum : uint32_t + { + kMvolL = 0x00, kMvolR = 0x02, kEvolL = 0x04, kEvolR = 0x06, kAvolL = 0x08, kAvolR = 0x0A, + kBvolL = 0x0C, kBvolR = 0x0E, kMvolXL = 0x10, kMvolXR = 0x12, kReverbCoefficients = 0x14, + }; + constexpr uint32_t kIrqInfo = 0x7C2; + + // Reverb register indices into Core::reverb: addresses then coefficients. + enum + { + kApf1Size, kApf2Size, kLSameDst, kRSameDst, kLComb1, kRComb1, kLComb2, kRComb2, + kLSameSrc, kRSameSrc, kLDiffDst, kRDiffDst, kLComb3, kRComb3, kLComb4, kRComb4, + kLDiffSrc, kRDiffSrc, kLApf1, kRApf1, kLApf2, kRApf2, + kIir, kComb1, kComb2, kComb3, kComb4, kWall, kApf1, kApf2, kInL, kInR, + }; + } + + void Spu2::Volume::set(uint16_t value) + { + reg = value; + if ((value & 0x8000u) == 0u) + { + // Fixed volume: a signed 15-bit value, doubled. + level = static_cast(static_cast(value << 1)); + counter = 0; + } + } + + void Spu2::Volume::update() + { + if ((reg & 0x8000u) == 0u) + return; + // Sweep: bit 14 exponential, bit 13 decrease, bits 6-0 rate. + const bool exponential = (reg & 0x4000u) != 0u; + const bool decrease = (reg & 0x2000u) != 0u; + const int32_t rate = reg & 0x7F; + const int32_t shift = rate >> 2; + int32_t step = decrease ? -8 + (rate & 3) : 7 - (rate & 3); + const int32_t cycles = 1 << std::max(0, shift - 11); + step <<= std::max(0, 11 - shift); + int32_t magnitude = std::abs(level); + if (exponential && decrease) + step = (step * magnitude) >> 15; + if (++counter < cycles * (exponential && !decrease && magnitude > 0x6000 ? 4 : 1)) + return; + counter = 0; + magnitude = std::clamp(magnitude + step, 0, 0x7FFF); + level = (reg & 0x1000u) != 0u ? -magnitude : magnitude; + } + + Spu2::Spu2(Spu2Host &host) : m_host(host), m_ram(kRamWords, 0u) + { + reset(); + } + + void Spu2::reset() + { + std::fill(m_ram.begin(), m_ram.end(), 0u); + m_regs.fill(0u); + m_cores = {}; + m_irqInfo = 0; + m_outputPos = 0; + } + + void Spu2::checkIrq(uint32_t address) + { + for (int core = 0; core < 2; ++core) + { + Core &c = m_cores[core]; + if ((c.attr & 0x40u) != 0u && c.irqAddress == (address & kAddressMask) && + (m_irqInfo & (4u << core)) == 0u) + { + m_irqInfo |= static_cast(4u << core); + m_host.raiseSpuInterrupt(); + } + } + } + + void Spu2::keyOn(int core, uint32_t mask) + { + Core &c = m_cores[core]; + c.endx &= ~mask; + for (int index = 0; index < 24; ++index) + { + if ((mask & (1u << index)) == 0u) + continue; + Voice &v = c.voices[index]; + v.nextAddress = v.startAddress; + v.loopAddressSet = false; + v.counter = 0; + v.sampleIndex = 28; + v.history[0] = v.history[1] = 0; + v.window = {}; + v.blockFlags = 0; + v.envelope = {0, 0, 1}; + } + } + + void Spu2::keyOff(int core, uint32_t mask) + { + for (int index = 0; index < 24; ++index) + { + Voice &v = m_cores[core].voices[index]; + if ((mask & (1u << index)) != 0u && v.envelope.phase != 0u) + { + v.envelope.phase = 4; + v.envelope.counter = 0; + } + } + } + + void Spu2::decodeBlock(int core, Voice &v) + { + const uint32_t base = v.nextAddress & kAddressMask & ~7u; + for (uint32_t offset = 0; offset < 8u; ++offset) + checkIrq(base + offset); + const uint16_t header = m_ram[base]; + v.blockFlags = static_cast(header >> 8); + if ((v.blockFlags & 4u) != 0u && !v.loopAddressSet) + v.loopAddress = base; + int32_t shift = header & 0xF; + if (shift > 12) + shift = 9; + const int32_t *filter = kFilters[(header >> 4) & 0xF]; + for (uint32_t index = 0; index < 28u; ++index) + { + const uint16_t word = m_ram[(base + 1u + index / 4u) & kAddressMask]; + const int32_t nibble = static_cast(static_cast(((word >> ((index & 3u) * 4u)) & 0xFu) << 12)); + const int32_t sample = clamp16((nibble >> shift) + + ((v.history[0] * filter[0] + v.history[1] * filter[1] + 32) >> 6)); + v.history[1] = v.history[0]; + v.history[0] = sample; + v.decoded[index] = static_cast(sample); + } + (void)core; + } + + int32_t Spu2::voiceSample(int core, int index) + { + Core &c = m_cores[core]; + Voice &v = c.voices[index]; + + int32_t pitch = v.pitch; + if (index > 0 && (c.pitchMod & (1u << index)) != 0u) + pitch = std::clamp((pitch * (0x8000 + c.voices[index - 1].output)) >> 15, 0, 0x3FFF); + v.counter += static_cast(std::min(pitch, 0x3FFF)); + while (v.counter >= 0x1000u) + { + v.counter -= 0x1000u; + if (v.sampleIndex >= 28u) + { + if (v.sampleIndex != 29u) + decodeBlock(core, v); + v.sampleIndex = 0; + } + v.window[0] = v.window[1]; + v.window[1] = v.window[2]; + v.window[2] = v.window[3]; + v.window[3] = v.decoded[v.sampleIndex++]; + if (v.sampleIndex == 28u) + { + // The block's flags apply once its last sample has been read. + if ((v.blockFlags & 1u) != 0u) + { + c.endx |= 1u << index; + v.nextAddress = v.loopAddress; + if ((v.blockFlags & 2u) == 0u) + { + v.envelope.phase = 0; + v.envelope.level = 0; + } + } + else + v.nextAddress = (v.nextAddress + 8u) & kAddressMask; + } + } + + // Four-point Catmull-Rom between the last two samples. + const int32_t t = static_cast(v.counter); + const int32_t p0 = v.window[0], p1 = v.window[1], p2 = v.window[2], p3 = v.window[3]; + const int64_t a = -p0 + 3 * p1 - 3 * p2 + p3; + const int64_t b = 2 * p0 - 5 * p1 + 4 * p2 - p3; + const int64_t d = -p0 + p2; + const int64_t value = ((((a * t >> 12) + b) * t >> 12) + d) * t >> 13; + return clamp16(static_cast(value) + p1); + } + + void Spu2::stepEnvelope(Voice &v) + { + Envelope &e = v.envelope; + int32_t rate = 0; + bool decrease = false, exponential = false; + switch (e.phase) + { + case 1: + rate = (v.adsr1 >> 8) & 0x7F; + exponential = (v.adsr1 & 0x8000u) != 0u; + break; + case 2: + rate = ((v.adsr1 >> 4) & 0xF) << 2; + decrease = exponential = true; + break; + case 3: + rate = (v.adsr2 >> 6) & 0x7F; + decrease = (v.adsr2 & 0x4000u) != 0u; + exponential = (v.adsr2 & 0x8000u) != 0u; + break; + case 4: + rate = (v.adsr2 & 0x1F) << 2; + decrease = true; + exponential = (v.adsr2 & 0x20u) != 0u; + break; + default: + return; + } + const int32_t shift = rate >> 2; + int32_t step = decrease ? -8 + (rate & 3) : 7 - (rate & 3); + int32_t cycles = 1 << std::max(0, shift - 11); + step <<= std::max(0, 11 - shift); + if (exponential && !decrease && e.level > 0x6000) + cycles *= 4; + if (exponential && decrease) + step = (step * e.level) >> 15; + if (++e.counter >= cycles) + { + e.counter = 0; + e.level = std::clamp(e.level + step, 0, 0x7FFF); + } + switch (e.phase) + { + case 1: + if (e.level >= 0x7FFF) + e.phase = 2; + break; + case 2: + if (e.level <= ((v.adsr1 & 0xF) + 1) * 0x800) + e.phase = 3; + break; + case 4: + if (e.level == 0) + e.phase = 0; + break; + default: + break; + } + } + + void Spu2::stepNoise(Core &c) + { + // The PS1-style noise generator, clocked by ATTR bits 8-13. + const uint32_t clock = (c.attr >> 8) & 0x3Fu; + const uint32_t shift = clock >> 2; + const uint32_t step = 4u + (clock & 3u); + c.noiseCounter += step << (15u - std::min(shift, 15u)); + while (c.noiseCounter >= 0x20000u) + { + c.noiseCounter -= 0x20000u; + const uint32_t bits = static_cast(c.noiseLevel); + const uint32_t feedback = ((bits >> 15) ^ (bits >> 12) ^ (bits >> 11) ^ (bits >> 10) ^ 1u) & 1u; + c.noiseLevel = static_cast(static_cast((bits << 1) | feedback)); + } + } + + void Spu2::readInput(int core) + { + Core &c = m_cores[core]; + const uint32_t base = 0x2000u + (static_cast(core) << 10); + checkIrq(base + c.inputPos); + checkIrq(base + 0x200u + c.inputPos); + if ((c.admas & (1u << core)) == 0u) + { + c.inputL = c.inputR = 0; + return; + } + if (!c.inputPrimed) + { + // Once AutoDMA is on, the first transfer fills both halves before + // anything plays; until it starts the input stays silent. + uint16_t block[512]; + for (uint32_t half = 0; half < 2u; ++half) + { + if (!m_host.fetchAutoDma(core, block)) + break; + std::memcpy(&m_ram[base + half * 0x100u], block, 0x200u); + std::memcpy(&m_ram[base + 0x200u + half * 0x100u], block + 256, 0x200u); + c.inputPrimed = true; + } + if (!c.inputPrimed) + { + c.inputL = c.inputR = 0; + return; + } + c.inputPos = 0; + } + c.inputL = static_cast(m_ram[base + c.inputPos]); + c.inputR = static_cast(m_ram[base + 0x200u + c.inputPos]); + c.inputPos = (c.inputPos + 1u) & 0x1FFu; + if ((c.inputPos & 0xFFu) == 0u) + { + // The half just finished playing takes the next block. + const uint32_t half = c.inputPos == 0u ? 0x100u : 0u; + uint16_t block[512]; + if (m_host.fetchAutoDma(core, block)) + { + std::memcpy(&m_ram[base + half], block, 0x200u); + std::memcpy(&m_ram[base + 0x200u + half], block + 256, 0x200u); + } + } + } + + void Spu2::mixReverb(int core, int32_t inL, int32_t inR, int32_t &outL, int32_t &outR) + { + Core &c = m_cores[core]; + // The reverb runs at half rate on the average of each pair of inputs. + c.reverbPhase = !c.reverbPhase; + if (c.reverbPhase || (c.attr & 0x80u) == 0u || c.effectEnd <= c.effectStart) + { + outL = c.reverbOutL; + outR = c.reverbOutR; + if ((c.attr & 0x80u) == 0u) + outL = outR = 0; + return; + } + const uint32_t size = c.effectEnd - c.effectStart + 1u; + const auto address = [&](uint32_t offset) { + return c.effectStart + (c.reverbPos + offset) % size; + }; + const auto readAt = [&](uint32_t offset) { + return static_cast(static_cast(m_ram[address(offset) & kAddressMask])); + }; + const auto writeAt = [&](uint32_t offset, int32_t value) { + m_ram[address(offset) & kAddressMask] = static_cast(clamp16(value)); + }; + const auto reg = [&](int index) { return c.reverbAddress[index]; }; + const auto coef = [&](int index) { return static_cast(c.reverbCoefficient[index - kIir]); }; + + const int32_t lin = scale(inL, coef(kInL)); + const int32_t rin = scale(inR, coef(kInR)); + const int32_t iir = coef(kIir), wall = coef(kWall); + const auto reflect = [&](int dst, int src, int32_t input) { + const int32_t previous = readAt(reg(dst) - 1u); + writeAt(reg(dst), scale(input + scale(readAt(reg(src)), wall) - previous, iir) + previous); + }; + reflect(kLSameDst, kLSameSrc, lin); + reflect(kRSameDst, kRSameSrc, rin); + reflect(kLDiffDst, kRDiffSrc, lin); + reflect(kRDiffDst, kLDiffSrc, rin); + + int32_t l = scale(readAt(reg(kLComb1)), coef(kComb1)) + scale(readAt(reg(kLComb2)), coef(kComb2)) + + scale(readAt(reg(kLComb3)), coef(kComb3)) + scale(readAt(reg(kLComb4)), coef(kComb4)); + int32_t r = scale(readAt(reg(kRComb1)), coef(kComb1)) + scale(readAt(reg(kRComb2)), coef(kComb2)) + + scale(readAt(reg(kRComb3)), coef(kComb3)) + scale(readAt(reg(kRComb4)), coef(kComb4)); + const auto allPass = [&](int32_t value, int dst, int size, int coefficient) { + const int32_t delayed = readAt(reg(dst) - reg(size)); + value = clamp16(value - scale(delayed, coef(coefficient))); + writeAt(reg(dst), value); + return clamp16(scale(value, coef(coefficient)) + delayed); + }; + l = allPass(clamp16(l), kLApf1, kApf1Size, kApf1); + r = allPass(clamp16(r), kRApf1, kApf1Size, kApf1); + l = allPass(l, kLApf2, kApf2Size, kApf2); + r = allPass(r, kRApf2, kApf2Size, kApf2); + c.reverbOutL = outL = l; + c.reverbOutR = outR = r; + c.reverbPos = (c.reverbPos + 1u) % size; + } + + void Spu2::tick(int16_t &left, int16_t &right) + { + int32_t external[2] = {0, 0}; + int32_t out[2] = {0, 0}; + for (int core = 0; core < 2; ++core) + { + Core &c = m_cores[core]; + stepNoise(c); + int32_t dry[2] = {0, 0}, wet[2] = {0, 0}; + for (int index = 0; index < 24; ++index) + { + Voice &v = c.voices[index]; + v.left.update(); + v.right.update(); + if (v.envelope.phase == 0u) + { + v.output = 0; + continue; + } + int32_t sample = (c.noise & (1u << index)) != 0u ? c.noiseLevel : voiceSample(core, index); + stepEnvelope(v); + sample = scale(sample, v.envelope.level); + v.output = sample; + const int32_t l = scale(sample, v.left.level); + const int32_t r = scale(sample, v.right.level); + const uint32_t bit = 1u << index; + if ((c.dryL & bit) != 0u) dry[0] += l; + if ((c.dryR & bit) != 0u) dry[1] += r; + if ((c.wetL & bit) != 0u) wet[0] += l; + if ((c.wetR & bit) != 0u) wet[1] += r; + } + dry[0] = clamp16(dry[0]); + dry[1] = clamp16(dry[1]); + wet[0] = clamp16(wet[0]); + wet[1] = clamp16(wet[1]); + + readInput(core); + const uint32_t volumes = 0x760u + static_cast(core) * 0x28u; + const auto vol = [&](uint32_t offset) { return static_cast(static_cast(raw(volumes + offset))); }; + const int32_t input[2] = {scale(c.inputL, vol(kBvolL)), scale(c.inputR, vol(kBvolR))}; + const int32_t ext[2] = {core == 1 ? scale(external[0], vol(kAvolL)) : 0, + core == 1 ? scale(external[1], vol(kAvolR)) : 0}; + + // MMIX gates, from bit 11 down: voice dry L/R, voice wet L/R, + // input dry L/R, input wet L/R, external dry L/R, external wet L/R. + const uint16_t gates = core == 0 ? c.mmix & 0xFF0u : c.mmix; + const auto gate = [&](uint32_t bit, int32_t value) { return (gates & bit) != 0u ? value : 0; }; + int32_t mixL = gate(0x800u, dry[0]) + gate(0x080u, input[0]) + gate(0x008u, ext[0]); + int32_t mixR = gate(0x400u, dry[1]) + gate(0x040u, input[1]) + gate(0x004u, ext[1]); + const int32_t sendL = clamp16(gate(0x200u, wet[0]) + gate(0x020u, input[0]) + gate(0x002u, ext[0])); + const int32_t sendR = clamp16(gate(0x100u, wet[1]) + gate(0x010u, input[1]) + gate(0x001u, ext[1])); + int32_t reverbL = 0, reverbR = 0; + mixReverb(core, sendL, sendR, reverbL, reverbR); + mixL += scale(reverbL, vol(kEvolL)); + mixR += scale(reverbR, vol(kEvolR)); + + c.masterL.update(); + c.masterR.update(); + out[0] = clamp16(scale(clamp16(mixL), c.masterL.level)); + out[1] = clamp16(scale(clamp16(mixR), c.masterR.level)); + if (core == 0) + { + external[0] = out[0]; + external[1] = out[1]; + // Core 0's output area, which core 1 and captures read back. + m_ram[0x800u + m_outputPos] = static_cast(out[0]); + m_ram[0xA00u + m_outputPos] = static_cast(out[1]); + } + } + m_outputPos = (m_outputPos + 1u) & 0x1FFu; + left = static_cast(out[0]); + right = static_cast(out[1]); + } + + void Spu2::dmaWrite(int core, const uint16_t *data, uint32_t words) + { + Core &c = m_cores[core]; + for (uint32_t index = 0; index < words; ++index) + { + checkIrq(c.transferAddress); + m_ram[c.transferAddress] = data[index]; + c.transferAddress = (c.transferAddress + 1u) & kAddressMask; + } + // Bit 7 reports a finished transfer; LIBSD's DMA handler waits for it. + c.statx |= 0x80u; + } + + void Spu2::dmaRead(int core, uint16_t *data, uint32_t words) + { + Core &c = m_cores[core]; + for (uint32_t index = 0; index < words; ++index) + { + checkIrq(c.transferAddress); + data[index] = m_ram[c.transferAddress]; + c.transferAddress = (c.transferAddress + 1u) & kAddressMask; + } + c.statx |= 0x80u; + } + + bool Spu2::autoDmaEnabled(int core) const + { + return (m_cores[core].admas & (1u << core)) != 0u; + } + + void Spu2::dmaStarted(int core) + { + // Drivers restart the DMA for each half of their buffer while AutoDMA + // keeps playing, so a restart must not touch the input position. + Core &c = m_cores[core]; + c.statx &= ~0x80u; + if (autoDmaEnabled(core)) + c.statx |= 0x80u; + } + + uint16_t Spu2::readCore(int core, uint32_t offset) + { + Core &c = m_cores[core]; + if (offset < 0x180u) + { + const Voice &v = c.voices[offset >> 4]; + switch (offset & 0xFu) + { + case 0xA: return static_cast(v.envelope.level); + case 0xC: return static_cast(v.left.level); + case 0xE: return static_cast(v.right.level); + default: break; + } + } + else if (offset >= kVoiceAddresses && offset < kVoiceAddresses + 24u * 0xCu) + { + const uint32_t relative = offset - kVoiceAddresses; + const Voice &v = c.voices[relative / 0xCu]; + if (relative % 0xCu == 8u) + return static_cast(v.nextAddress >> 16); + if (relative % 0xCu == 10u) + return static_cast(v.nextAddress); + } + switch (offset) + { + case kEndxL: return static_cast(c.endx); + case kEndxH: return static_cast(c.endx >> 16); + case kStatx: return c.statx; + case kAdmas: return c.admas; + case kTsaH: return static_cast(c.transferAddress >> 16); + case kTsaL: return static_cast(c.transferAddress); + case kAttr: return c.attr; + default: break; + } + return raw(static_cast(core) * 0x400u + offset); + } + + uint16_t Spu2::read(uint32_t offset) + { + offset &= 0x7FEu; + if (offset == kIrqInfo) + return m_irqInfo; + if (offset >= 0x760u && offset < 0x7B0u) + { + const int core = offset >= 0x788u ? 1 : 0; + const uint32_t relative = offset - 0x760u - static_cast(core) * 0x28u; + if (relative == kMvolXL) + return static_cast(m_cores[core].masterL.level); + if (relative == kMvolXR) + return static_cast(m_cores[core].masterR.level); + return raw(offset); + } + if (offset < 0x760u) + return readCore(offset >= 0x400u ? 1 : 0, offset & 0x3FFu); + return raw(offset); + } + + void Spu2::writeCore(int core, uint32_t offset, uint16_t value) + { + Core &c = m_cores[core]; + const auto high = [](uint32_t &target, uint16_t v) { target = (target & 0xFFFFu) | ((v & 0xFu) << 16); }; + const auto low = [](uint32_t &target, uint16_t v) { target = (target & 0xF0000u) | v; }; + const auto maskLow = [](uint32_t &target, uint16_t v) { target = (target & 0xFF0000u) | v; }; + const auto maskHigh = [](uint32_t &target, uint16_t v) { target = (target & 0xFFFFu) | ((v & 0xFFu) << 16); }; + if (offset < 0x180u) + { + Voice &v = c.voices[offset >> 4]; + switch (offset & 0xFu) + { + case 0x0: v.left.set(value); break; + case 0x2: v.right.set(value); break; + case 0x4: v.pitch = value; break; + case 0x6: v.adsr1 = value; break; + case 0x8: v.adsr2 = value; break; + default: break; + } + return; + } + if (offset >= kVoiceAddresses && offset < kVoiceAddresses + 24u * 0xCu) + { + const uint32_t relative = offset - kVoiceAddresses; + Voice &v = c.voices[relative / 0xCu]; + switch (relative % 0xCu) + { + case 0: high(v.startAddress, value); break; + case 2: low(v.startAddress, value); break; + case 4: high(v.loopAddress, value); v.loopAddressSet = true; break; + case 6: low(v.loopAddress, value); v.loopAddressSet = true; break; + case 8: high(v.nextAddress, value); break; + case 10: low(v.nextAddress, value); break; + default: break; + } + return; + } + if (offset >= kReverbAddresses && offset < kEea) + { + // Twenty-bit offsets into the effect area, high word first. + uint32_t &address = c.reverbAddress[(offset - kReverbAddresses) >> 2]; + if ((offset & 2u) == 0u) + high(address, value); + else + low(address, value); + return; + } + switch (offset) + { + case kPitchModL: maskLow(c.pitchMod, value); break; + case kPitchModH: maskHigh(c.pitchMod, value); break; + case kNoiseL: maskLow(c.noise, value); break; + case kNoiseH: maskHigh(c.noise, value); break; + case kDryLL: maskLow(c.dryL, value); break; + case kDryLH: maskHigh(c.dryL, value); break; + case kWetLL: maskLow(c.wetL, value); break; + case kWetLH: maskHigh(c.wetL, value); break; + case kDryRL: maskLow(c.dryR, value); break; + case kDryRH: maskHigh(c.dryR, value); break; + case kWetRL: maskLow(c.wetR, value); break; + case kWetRH: maskHigh(c.wetR, value); break; + case kMmix: c.mmix = value; break; + case kAttr: + { + const bool irqWasEnabled = (c.attr & 0x40u) != 0u; + c.attr = value; + // Clearing IRQ enable acknowledges the core's pending interrupt. + if (irqWasEnabled && (value & 0x40u) == 0u) + m_irqInfo &= static_cast(~(4u << core)); + if ((value & 0x30u) == 0u) + c.statx &= ~0x80u; + break; + } + case kIrqAH: high(c.irqAddress, value); break; + case kIrqAL: low(c.irqAddress, value); break; + case kKeyOnL: keyOn(core, value); break; + case kKeyOnH: keyOn(core, static_cast(value & 0xFFu) << 16); break; + case kKeyOffL: keyOff(core, value); break; + case kKeyOffH: keyOff(core, static_cast(value & 0xFFu) << 16); break; + case kTsaH: high(c.transferAddress, value); break; + case kTsaL: low(c.transferAddress, value); break; + case kData: + checkIrq(c.transferAddress); + m_ram[c.transferAddress] = value; + c.transferAddress = (c.transferAddress + 1u) & kAddressMask; + break; + case kAdmas: + c.admas = value; + if ((value & 3u) == 0u) + c.inputPrimed = false; + break; + case kEsaH: high(c.effectStart, value); c.reverbPos = 0; break; + case kEsaL: low(c.effectStart, value); c.reverbPos = 0; break; + case kEea: c.effectEnd = (static_cast(value & 0xFu) << 16) | 0xFFFFu; break; + case kEndxL: case kEndxH: break; + default: break; + } + } + + void Spu2::write(uint32_t offset, uint16_t value) + { + offset &= 0x7FEu; + raw(offset) = value; + if (offset == kIrqInfo) + return; + if (offset >= 0x760u && offset < 0x7B0u) + { + const int core = offset >= 0x788u ? 1 : 0; + const uint32_t relative = offset - 0x760u - static_cast(core) * 0x28u; + Core &c = m_cores[core]; + if (relative == kMvolL) + c.masterL.set(value); + else if (relative == kMvolR) + c.masterR.set(value); + else if (relative >= kReverbCoefficients && relative < kReverbCoefficients + 20u) + c.reverbCoefficient[(relative - kReverbCoefficients) / 2u] = static_cast(value); + return; + } + if (offset < 0x760u) + writeCore(offset >= 0x400u ? 1 : 0, offset & 0x3FFu, value); + } +} diff --git a/ps2xIOP/src/lle/spu2.h b/ps2xIOP/src/lle/spu2.h new file mode 100644 index 000000000..075673d48 --- /dev/null +++ b/ps2xIOP/src/lle/spu2.h @@ -0,0 +1,122 @@ +#pragma once + +#include +#include +#include + +namespace ps2x::iop::lle +{ + // What the SPU2 needs from the IOP around it. + class Spu2Host + { + public: + virtual ~Spu2Host() = default; + // AutoDMA wants its next 1 KiB block (512 bytes left, 512 bytes right). + // Returns false while the core's DMA channel has nothing queued. + virtual bool fetchAutoDma(int core, uint16_t *block) = 0; + // The SPU2 interrupt line went up. + virtual void raiseSpuInterrupt() = 0; + }; + + // The two SPU2 cores, their 48 voices and 2 MiB of sound memory, mixed + // one 48 kHz sample at a time. + class Spu2 + { + public: + static constexpr uint32_t kRamWords = 1u << 20; + + explicit Spu2(Spu2Host &host); + + void reset(); + uint16_t read(uint32_t offset); + void write(uint32_t offset, uint16_t value); + + // Produce one output sample pair. + void tick(int16_t &left, int16_t &right); + + // DMA transfers between IOP memory and sound memory at the core's TSA. + void dmaWrite(int core, const uint16_t *data, uint32_t words); + void dmaRead(int core, uint16_t *data, uint32_t words); + bool autoDmaEnabled(int core) const; + // A DMA channel was started or stopped for this core. + void dmaStarted(int core); + + uint16_t *ram() { return m_ram.data(); } + + private: + struct Envelope + { + int32_t level = 0; + int32_t counter = 0; + uint8_t phase = 0; // 0 off, 1 attack, 2 decay, 3 sustain, 4 release + }; + + struct Volume + { + uint16_t reg = 0; + int32_t level = 0; // current value, -0x8000..0x7FFF + int32_t counter = 0; + void set(uint16_t value); + void update(); + }; + + struct Voice + { + Volume left, right; + uint16_t pitch = 0; + uint16_t adsr1 = 0, adsr2 = 0; + Envelope envelope; + uint32_t startAddress = 0; + uint32_t loopAddress = 0; + uint32_t nextAddress = 0; // address of the block being played + bool loopAddressSet = false; + uint32_t counter = 0; // 12-bit fraction plus sample index + std::array decoded{}; + int32_t history[2]{}; + std::array window{}; // last four samples for interpolation + uint8_t sampleIndex = 28; // 28 forces a block decode + uint8_t blockFlags = 0; + int32_t output = 0; // post-envelope sample, for modulation + }; + + struct Core + { + std::array voices; + uint32_t pitchMod = 0, noise = 0, dryL = 0, dryR = 0, wetL = 0, wetR = 0; + uint16_t mmix = 0, attr = 0, admas = 0, statx = 0x80; + uint32_t irqAddress = 0, transferAddress = 0, endx = 0; + uint32_t effectStart = 0, effectEnd = 0; + Volume masterL, masterR; + int16_t effectL = 0, effectR = 0, inputL = 0, inputR = 0, externalL = 0, externalR = 0; + uint32_t inputPos = 0; + bool inputPrimed = false; + int32_t noiseLevel = 0; + uint32_t noiseCounter = 0; + std::array reverbAddress{}; + std::array reverbCoefficient{}; + uint32_t reverbPos = 0; + int32_t reverbOutL = 0, reverbOutR = 0; + bool reverbPhase = false; + }; + + uint16_t &raw(uint32_t offset) { return m_regs[(offset >> 1) & 0x3FFu]; } + void writeCore(int core, uint32_t offset, uint16_t value); + uint16_t readCore(int core, uint32_t offset); + void keyOn(int core, uint32_t mask); + void keyOff(int core, uint32_t mask); + void checkIrq(uint32_t address); + void decodeBlock(int core, Voice &voice); + int32_t voiceSample(int core, int index); + void stepEnvelope(Voice &voice); + void stepNoise(Core &core); + void readInput(int core); + void mixReverb(int core, int32_t inL, int32_t inR, int32_t &outL, int32_t &outR); + + Spu2Host &m_host; + std::vector m_ram; + std::array m_regs{}; + std::array m_cores{}; + uint16_t m_irqInfo = 0; + uint32_t m_outputPos = 0; + }; +} diff --git a/ps2xIOP/src/native_iop.cpp b/ps2xIOP/src/native_iop.cpp new file mode 100644 index 000000000..6325ee020 --- /dev/null +++ b/ps2xIOP/src/native_iop.cpp @@ -0,0 +1,207 @@ +#include "ps2x/iop/native_iop.h" + +#include "lle/iop.h" + +#include +#include +#include +#include +#include + +namespace ps2x::iop +{ + namespace + { + std::string moduleName(std::string_view guestPath) + { + const size_t slash = guestPath.find_last_of("\\/:"); + std::string_view name = slash == std::string_view::npos ? guestPath : guestPath.substr(slash + 1); + name = name.substr(0, name.find(';')); + std::string upper(name); + for (char &ch : upper) + ch = static_cast(std::toupper(static_cast(ch))); + return upper; + } + } + + class NativeIop::Impl final : public lle::EeLink + { + public: + explicit Impl(IopHost &hostRef) : host(hostRef), iop(*this) {} + + bool readEe(uint32_t address, void *destination, uint32_t size) override + { + return host.readGuest(address, destination, size); + } + bool writeEe(uint32_t address, const void *source, uint32_t size) override + { + return host.writeGuest(address, source, size); + } + bool eeServer(uint32_t sid) override { return host.hasGuestRpcServer(sid); } + void log(const std::string &message) override { host.log(LogLevel::Info, message); } + + // Advance the IOP without an audio device, discarding the samples. + void stepSilently(uint32_t frames) + { + int16_t left, right; + for (uint32_t index = 0; index < frames; ++index) + iop.step(left, right); + } + + bool audioRunning() const + { + return rendered && std::chrono::steady_clock::now() - lastRender < std::chrono::milliseconds(250); + } + + IopHost &host; + lle::Iop iop; + mutable std::mutex mutex; + std::condition_variable progressed; + std::vector modules; + std::chrono::steady_clock::time_point lastRender{}; + bool rendered = false; + bool loaded = false; + }; + + NativeIop::NativeIop(IopHost &host) : m_impl(std::make_unique(host)) {} + NativeIop::~NativeIop() = default; + + void NativeIop::setModules(std::vector names) + { + std::lock_guard lock(m_impl->mutex); + m_impl->modules.clear(); + for (const auto &name : names) + m_impl->modules.push_back(moduleName(name)); + } + + bool NativeIop::enabled() const + { + std::lock_guard lock(m_impl->mutex); + return !m_impl->modules.empty(); + } + + bool NativeIop::claims(std::string_view guestPath) const + { + std::lock_guard lock(m_impl->mutex); + const std::string name = moduleName(guestPath); + return std::find(m_impl->modules.begin(), m_impl->modules.end(), name) != m_impl->modules.end(); + } + + int32_t NativeIop::load(std::string_view guestPath, std::string_view arguments) + { + IopHost &host = m_impl->host; + const uint64_t handle = host.openHostFile(host.translateGuestPath(guestPath)); + uint64_t size = 0; + if (handle == 0u || !host.hostFileSize(handle, size) || size == 0u || size > (4u << 20)) + { + if (handle != 0u) + host.closeHostFile(handle); + host.log(LogLevel::Warning, "[iop] cannot read " + std::string(guestPath)); + return -1; + } + std::vector file(static_cast(size)); + size_t got = 0; + const bool read = host.readHostFile(handle, 0u, file.data(), file.size(), got); + host.closeHostFile(handle); + if (!read || got != file.size()) + return -1; + + std::vector argv; + for (size_t start = 0; start < arguments.size();) + { + const size_t end = std::min(arguments.find('\0', start), arguments.size()); + if (end > start) + argv.emplace_back(arguments.substr(start, end - start)); + start = end + 1; + } + + std::lock_guard lock(m_impl->mutex); + const int32_t id = m_impl->iop.loadModule(file, std::string(guestPath), argv); + if (id > 0) + m_impl->loaded = true; + m_impl->iop.settle(); + return id; + } + + bool NativeIop::server(uint32_t sid, uint32_t &record, uint32_t &buffer) const + { + std::lock_guard lock(m_impl->mutex); + return m_impl->iop.server(sid, record, buffer); + } + + bool NativeIop::call(uint32_t sid, uint32_t function, uint32_t sendAddress, uint32_t sendSize, + uint32_t receiveAddress, uint32_t receiveSize) + { + auto request = std::make_shared(); + request->sid = sid; + request->function = function; + request->send.resize(sendSize); + if (sendSize != 0u && !m_impl->host.readGuest(sendAddress, request->send.data(), sendSize)) + return false; + request->receiveAddress = receiveAddress; + request->receiveSize = receiveSize; + + std::unique_lock lock(m_impl->mutex); + if (!m_impl->iop.hasServer(sid)) + return false; + m_impl->iop.submit(request); + m_impl->iop.settle(); + // Most commands finish at once; the rest wait for IOP time to pass. + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(2); + while (!request->done && std::chrono::steady_clock::now() < deadline) + { + if (m_impl->audioRunning()) + m_impl->progressed.wait_for(lock, std::chrono::milliseconds(20)); + else + m_impl->stepSilently(480u); + m_impl->iop.settle(); + } + if (!request->done) + m_impl->host.log(LogLevel::Warning, "[iop] RPC to " + std::to_string(sid) + " timed out"); + return request->done; + } + + bool NativeIop::write(uint32_t address, const void *data, uint32_t size) + { + std::lock_guard lock(m_impl->mutex); + m_impl->iop.writeMemory(address, data, size); + return true; + } + + bool NativeIop::read(uint32_t address, void *data, uint32_t size) + { + std::lock_guard lock(m_impl->mutex); + m_impl->iop.readMemory(address, data, size); + return true; + } + + uint32_t NativeIop::allocate(uint32_t size) + { + std::lock_guard lock(m_impl->mutex); + return m_impl->iop.allocate(size); + } + + void NativeIop::release(uint32_t address) + { + std::lock_guard lock(m_impl->mutex); + m_impl->iop.release(address); + } + + void NativeIop::render(int16_t *frames, uint32_t count) + { + { + std::lock_guard lock(m_impl->mutex); + for (uint32_t index = 0; index < count; ++index) + m_impl->iop.step(frames[index * 2u], frames[index * 2u + 1u]); + m_impl->rendered = true; + m_impl->lastRender = std::chrono::steady_clock::now(); + } + m_impl->progressed.notify_all(); + } + + bool NativeIop::active() const + { + std::lock_guard lock(m_impl->mutex); + return m_impl->loaded; + } +} diff --git a/ps2xRecomp/CMakeLists.txt b/ps2xRecomp/CMakeLists.txt index 857102a22..5ed0d0799 100644 --- a/ps2xRecomp/CMakeLists.txt +++ b/ps2xRecomp/CMakeLists.txt @@ -130,7 +130,7 @@ install(DIRECTORY include/ DESTINATION include ) -include("${CMAKE_SOURCE_DIR}/ps2xRuntime/cmake/ReleaseMode.cmake") +include("${CMAKE_CURRENT_SOURCE_DIR}/../ps2xRuntime/cmake/ReleaseMode.cmake") if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") EnableFastReleaseMode(ps2_recomp_lib) diff --git a/ps2xRecomp/src/lib/control_flow_analyzer.cpp b/ps2xRecomp/src/lib/control_flow_analyzer.cpp index a099f5cf9..bf105e079 100644 --- a/ps2xRecomp/src/lib/control_flow_analyzer.cpp +++ b/ps2xRecomp/src/lib/control_flow_analyzer.cpp @@ -135,6 +135,23 @@ namespace ps2recomp for (const auto &inst : instructions) { + // A syscall can hand control back to the scheduler before the + // instruction after it runs: SetSyscall lets the guest install its + // own handler, and dispatchSyscallOverride then suspends the + // calling thread and queues that handler as a GuestInvocation. When + // the invocation completes, the scheduler resumes the parent thread + // at the address the generated code stored before calling + // handleSyscall -- i.e. right here. Without an entry point there, + // EeScheduler's hasFunction() check fails and the thread is made + // dormant instead of resumed. + // + // Note the offset is +4, not the +8 used for JAL/JALR: syscall has + // no delay slot. + if (inst.opcode == OPCODE_SPECIAL && inst.function == SPECIAL_SYSCALL) + { + queueResumeEntryTarget(inst.address + 4u); + } + bool isStaticJump = (inst.opcode == OPCODE_J || inst.opcode == OPCODE_JAL); if (inst.isBranch && inst.opcode != OPCODE_J && inst.opcode != OPCODE_JAL) { @@ -385,7 +402,14 @@ namespace ps2recomp } } } - if (!foundTable) + // Only an unresolved computed *jump* can land on an arbitrary + // instruction of this function and therefore force every address to + // become an entry point. JALR is a call: it transfers control to + // another function and comes back to the instruction after the delay + // slot, which is already queued as a resume target above. Treating a + // call like a jump here promotes the whole function for what is + // usually just a function pointer or virtual dispatch. + if (!foundTable && jrInst->function != SPECIAL_JALR) { needsIndirectFallback = true; } diff --git a/ps2xRecomp/src/lib/control_flow_emitter.cpp b/ps2xRecomp/src/lib/control_flow_emitter.cpp index 4bf8dc550..ce3020675 100644 --- a/ps2xRecomp/src/lib/control_flow_emitter.cpp +++ b/ps2xRecomp/src/lib/control_flow_emitter.cpp @@ -404,12 +404,16 @@ namespace ps2recomp case OPCODE_BNE: case OPCODE_BNEL: return fmt::format("GPR_U64(ctx, {}) != GPR_U64(ctx, {})", rsReg, rtReg); + // The R5900 compares the full 64-bit GPR for the relational branches, as it + // already does for BEQ/BNE above. Comparing only the low word takes the wrong + // branch whenever the upper half is significant, which happens with the + // dsll32/dsra32 sign-extension idiom compilers emit ahead of these opcodes. case OPCODE_BLEZ: case OPCODE_BLEZL: - return fmt::format("GPR_S32(ctx, {}) <= 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) <= 0", rsReg); case OPCODE_BGTZ: case OPCODE_BGTZL: - return fmt::format("GPR_S32(ctx, {}) > 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) > 0", rsReg); case OPCODE_REGIMM: switch (m_branchInst.rt) { @@ -417,12 +421,12 @@ namespace ps2recomp case REGIMM_BLTZL: case REGIMM_BLTZAL: case REGIMM_BLTZALL: - return fmt::format("GPR_S32(ctx, {}) < 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) < 0", rsReg); case REGIMM_BGEZ: case REGIMM_BGEZL: case REGIMM_BGEZAL: case REGIMM_BGEZALL: - return fmt::format("GPR_S32(ctx, {}) >= 0", rsReg); + return fmt::format("GPR_S64(ctx, {}) >= 0", rsReg); default: return "false"; } diff --git a/ps2xRecomp/src/lib/elf_parser.cpp b/ps2xRecomp/src/lib/elf_parser.cpp index 91e43ee32..865dd6aad 100644 --- a/ps2xRecomp/src/lib/elf_parser.cpp +++ b/ps2xRecomp/src/lib/elf_parser.cpp @@ -986,6 +986,7 @@ namespace ps2recomp int skippedNonExecutable = 0; int skippedInvalidRange = 0; std::unordered_set mapStarts; + std::vector mapFunctions; while (std::getline(file, line)) { if (line.empty()) @@ -1029,7 +1030,7 @@ namespace ps2recomp func.isStub = false; func.isSkipped = false; - m_extraFunctions.push_back(std::move(func)); + mapFunctions.push_back(std::move(func)); mapStarts.insert(start); count++; } @@ -1062,14 +1063,27 @@ namespace ps2recomp } } + // An explicit function map is authoritative over the internal JAL-target + // scan. Those carvings end at the next JAL target or, for the last one in + // a region, at the end of the code section - which on single-PROGBITS + // executables runs straight through interleaved rodata. Because both the + // carvings ("sub_") and typical map names ("FUN_") count as + // auto-generated, a carving sharing a start with a map row used to + // survive this purge and then win the "larger end" tie-break below, + // replacing precise bounds with runaway ones. Drop every auto-named + // carving instead, then append the map rows. m_extraFunctions.erase( std::remove_if(m_extraFunctions.begin(), m_extraFunctions.end(), [&](const Function &func) { - return IsAutoGeneratedName(func.name) && !mapStarts.contains(func.start); + return IsAutoGeneratedName(func.name); }), m_extraFunctions.end()); + m_extraFunctions.insert(m_extraFunctions.end(), + std::make_move_iterator(mapFunctions.begin()), + std::make_move_iterator(mapFunctions.end())); + std::sort(m_extraFunctions.begin(), m_extraFunctions.end(), [](const Function &a, const Function &b) { diff --git a/ps2xRecomp/src/lib/fpu_translator.cpp b/ps2xRecomp/src/lib/fpu_translator.cpp index 944cee796..9bf49d49f 100644 --- a/ps2xRecomp/src/lib/fpu_translator.cpp +++ b/ps2xRecomp/src/lib/fpu_translator.cpp @@ -58,7 +58,8 @@ namespace ps2recomp "else ctx->f[{}] = ctx->f[{}] / ctx->f[{}];", ft, fd, fs, fd, fs, ft); case COP1_S_SQRT: - return fmt::format("ctx->f[{}] = FPU_SQRT_S(ctx->f[{}]);", fd, fs); + // R5900 SQRT.S uses ft, unlike ABS.S/MOV.S/NEG.S. + return fmt::format("ctx->f[{}] = FPU_SQRT_S(ctx->f[{}]);", fd, ft); case COP1_S_ABS: return fmt::format("ctx->f[{}] = FPU_ABS_S(ctx->f[{}]);", fd, fs); case COP1_S_MOV: diff --git a/ps2xRecomp/src/lib/function_table_emitter.cpp b/ps2xRecomp/src/lib/function_table_emitter.cpp index 27dc9b74c..84d3850ea 100644 --- a/ps2xRecomp/src/lib/function_table_emitter.cpp +++ b/ps2xRecomp/src/lib/function_table_emitter.cpp @@ -95,9 +95,25 @@ namespace ps2recomp } if (entryTarget.empty()) { - throw std::runtime_error("No entry function name available for registration."); + // Not every toolchain points e_entry at code. Metrowerks CodeWarrior + // for PS2 emits a crt0 data table there, so no function covers the + // address and no name resolves. Registering a synthesized name would + // reference a definition that was never emitted, so skip the entry + // instead - the real start address comes from configuration - and + // keep emitting the rest of the table rather than aborting after the + // per-function sources were already written. + if (cg.m_reporter) + { + std::ostringstream oss; + oss << "No function covers the ELF entry point; skipping its table registration. " + << "Set the entry explicitly if the runtime should start here."; + cg.m_reporter->warning("function-table", oss.str()); + } + } + else + { + addEntry(cg.m_bootstrapInfo.entry, entryTarget); } - addEntry(cg.m_bootstrapInfo.entry, entryTarget); } for (const auto &[address, name] : normalFunctions) diff --git a/ps2xRecomp/src/lib/ps2_recompiler.cpp b/ps2xRecomp/src/lib/ps2_recompiler.cpp index ab779a4f5..54fdfcf2c 100644 --- a/ps2xRecomp/src/lib/ps2_recompiler.cpp +++ b/ps2xRecomp/src/lib/ps2_recompiler.cpp @@ -1920,6 +1920,9 @@ namespace ps2recomp uint32_t start = function.start; uint32_t end = function.end; + // A branch at the mapped end still owns the next word as its delay + // slot; without this the emitter substitutes a NOP and loses it. + bool delaySlotExtended = false; for (uint32_t address = start; address < end; address += 4) { @@ -1977,6 +1980,12 @@ namespace ps2recomp } instructions.push_back(inst); + + if (!delaySlotExtended && inst.hasDelaySlot && address + 4u == end) + { + delaySlotExtended = true; + end += 4u; + } } catch (const std::exception &e) { diff --git a/ps2xRuntime/CMakeLists.txt b/ps2xRuntime/CMakeLists.txt index e4dc1956d..11d7dbd87 100644 --- a/ps2xRuntime/CMakeLists.txt +++ b/ps2xRuntime/CMakeLists.txt @@ -383,17 +383,125 @@ add_library(ps2_runtime STATIC src/lib/gs/ps2_gs_memory.cpp src/lib/gs/gs_frontend.cpp src/lib/gs/gs_cpu_backend.cpp + src/lib/gs/gs_threaded_backend.cpp src/lib/ps2_iop_host.cpp src/lib/ps2_memory.cpp + src/lib/ps2_native_iop.cpp src/lib/ps2_pad.cpp src/lib/ps2_runtime.cpp src/lib/ps2_vif1_interpreter.cpp src/lib/vu/ps2_vu1_core.cpp src/lib/vu/ps2_vu1_upper.cpp src/lib/vu/ps2_vu1_lower.cpp + src/lib/vu/ps2_vu1_pipeline.cpp + src/lib/vu/ps2_vu1_capture.cpp + src/lib/vu/ps2_vu1_cache.cpp + src/lib/vu/ps2_vu1_program.cpp src/lib/games_database.cpp ) +# Keep the VU execution loop and its arithmetic helpers visible to the optimizer. +set_target_properties(ps2_runtime PROPERTIES UNITY_BUILD ON UNITY_BUILD_MODE GROUP) +set_source_files_properties( + src/lib/vu/ps2_vu1_core.cpp + src/lib/vu/ps2_vu1_upper.cpp + src/lib/vu/ps2_vu1_lower.cpp + PROPERTIES UNITY_GROUP vu1) + +set(PS2X_VU_AOT_IMAGES "" CACHE STRING "Local VU code images or block profiles to compile; never distributed") +set(PS2X_VU_AOT_PAIR_IMAGES "" CACHE STRING "Additional local VU inputs to compile as reusable individual pairs") +set(PS2X_VU_AOT_LOOP_IMAGES "" CACHE STRING "Local VU code images to scan for bounded conditional loops") +option(PS2X_VU_AOT_DEBUG_INFO "Include expanded VU AOT debug scopes (slow to compile)" OFF) +set(PS2X_VU_AOT_SHARDS "32" CACHE STRING "Number of independent VU AOT compilation units") +set(PS2X_VU_AOT_COMPILE_JOBS "2" CACHE STRING "Maximum concurrent runtime compilations with VU AOT (Ninja)") +if(PS2X_VU_AOT_IMAGES OR PS2X_VU_AOT_PAIR_IMAGES OR PS2X_VU_AOT_LOOP_IMAGES) + find_package(Python3 COMPONENTS Interpreter REQUIRED) + if(NOT PS2X_VU_AOT_SHARDS MATCHES "^[1-9][0-9]*$") + message(FATAL_ERROR "PS2X_VU_AOT_SHARDS must be a positive integer") + endif() + if(NOT PS2X_VU_AOT_COMPILE_JOBS MATCHES "^[1-9][0-9]*$") + message(FATAL_ERROR "PS2X_VU_AOT_COMPILE_JOBS must be a positive integer") + endif() + # Static data-flow specializations otherwise exhaust memory in parallel builds. + if(CMAKE_GENERATOR MATCHES "Ninja") + set_property(GLOBAL APPEND PROPERTY JOB_POOLS ps2_vu_aot_compile=${PS2X_VU_AOT_COMPILE_JOBS}) + set_property(TARGET ps2_runtime PROPERTY JOB_POOL_COMPILE ps2_vu_aot_compile) + endif() + set(PS2X_VU_AOT_INCLUDE "${CMAKE_CURRENT_BINARY_DIR}/generated/vu_aot_blocks.inc") + set(PS2X_VU_AOT_EXTERN_INCLUDE "${CMAKE_CURRENT_BINARY_DIR}/generated/vu_aot_blocks_extern.inc") + set(PS2X_VU_AOT_SOURCES) + math(EXPR PS2X_VU_AOT_LAST_SHARD "${PS2X_VU_AOT_SHARDS} - 1") + foreach(SHARD RANGE ${PS2X_VU_AOT_LAST_SHARD}) + list(APPEND PS2X_VU_AOT_SOURCES "${CMAKE_CURRENT_BINARY_DIR}/generated/vu_aot_blocks_${SHARD}.cpp") + endforeach() + add_custom_command( + OUTPUT "${PS2X_VU_AOT_INCLUDE}" "${PS2X_VU_AOT_EXTERN_INCLUDE}" ${PS2X_VU_AOT_SOURCES} + COMMAND "${Python3_EXECUTABLE}" "${CMAKE_CURRENT_SOURCE_DIR}/../tools/compile_vu_blocks.py" + --output "${PS2X_VU_AOT_INCLUDE}" --shards ${PS2X_VU_AOT_SHARDS} + --pair-images ${PS2X_VU_AOT_PAIR_IMAGES} + --loop-images ${PS2X_VU_AOT_LOOP_IMAGES} + -- ${PS2X_VU_AOT_IMAGES} + DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../tools/compile_vu_blocks.py" + ${PS2X_VU_AOT_IMAGES} ${PS2X_VU_AOT_PAIR_IMAGES} ${PS2X_VU_AOT_LOOP_IMAGES} + VERBATIM) + add_custom_target(ps2_vu_aot DEPENDS "${PS2X_VU_AOT_INCLUDE}" "${PS2X_VU_AOT_EXTERN_INCLUDE}" ${PS2X_VU_AOT_SOURCES}) + add_dependencies(ps2_runtime ps2_vu_aot) + target_sources(ps2_runtime PRIVATE ${PS2X_VU_AOT_SOURCES}) + target_compile_definitions(ps2_runtime PRIVATE + PS2X_VU_AOT_INCLUDE="${PS2X_VU_AOT_INCLUDE}" + PS2X_VU_AOT_EXTERN_INCLUDE="${PS2X_VU_AOT_EXTERN_INCLUDE}") + set_source_files_properties(${PS2X_VU_AOT_SOURCES} PROPERTIES + SKIP_UNITY_BUILD_INCLUSION ON + INCLUDE_DIRECTORIES "${CMAKE_CURRENT_SOURCE_DIR}/src/lib/vu") + # Thousands of specialized inline scopes otherwise dominate compile time. + if(CMAKE_CXX_COMPILER_ID MATCHES "Clang|GNU" AND NOT PS2X_VU_AOT_DEBUG_INFO) + set_source_files_properties( + "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/ps2_runtime.dir/Unity/unity_vu1_cxx.cxx" + ${PS2X_VU_AOT_SOURCES} + PROPERTIES COMPILE_OPTIONS "-g0") + endif() +endif() + +# Whole VU1 routines, compiled from PS2_VU_PROGRAM_PROFILE recordings. +set(PS2X_VU_PROGRAM_PROFILES "" CACHE STRING "Local VU1 program profile directories to compile; never distributed") +set(PS2X_VU_PROGRAM_SHARDS "8" CACHE STRING "Number of compilation units for compiled VU1 routines") +if(PS2X_VU_PROGRAM_PROFILES) + find_package(Python3 COMPONENTS Interpreter REQUIRED) + if(NOT PS2X_VU_PROGRAM_SHARDS MATCHES "^[1-9][0-9]*$") + message(FATAL_ERROR "PS2X_VU_PROGRAM_SHARDS must be a positive integer") + endif() + set(PS2X_VU_PROGRAM_INCLUDE "${CMAKE_CURRENT_BINARY_DIR}/generated/vu_programs.inc") + set(PS2X_VU_PROGRAM_EXTERN_INCLUDE "${CMAKE_CURRENT_BINARY_DIR}/generated/vu_programs_extern.inc") + set(PS2X_VU_PROGRAM_SOURCES) + math(EXPR PS2X_VU_PROGRAM_LAST_SHARD "${PS2X_VU_PROGRAM_SHARDS} - 1") + foreach(SHARD RANGE ${PS2X_VU_PROGRAM_LAST_SHARD}) + list(APPEND PS2X_VU_PROGRAM_SOURCES "${CMAKE_CURRENT_BINARY_DIR}/generated/vu_programs_${SHARD}.cpp") + endforeach() + set(PS2X_VU_PROGRAM_LISTS) + foreach(PROFILE ${PS2X_VU_PROGRAM_PROFILES}) + list(APPEND PS2X_VU_PROGRAM_LISTS "${PROFILE}/entries.txt") + endforeach() + add_custom_command( + OUTPUT "${PS2X_VU_PROGRAM_INCLUDE}" "${PS2X_VU_PROGRAM_EXTERN_INCLUDE}" ${PS2X_VU_PROGRAM_SOURCES} + COMMAND "${Python3_EXECUTABLE}" "${CMAKE_CURRENT_SOURCE_DIR}/../tools/compile_vu_programs.py" + --output "${PS2X_VU_PROGRAM_INCLUDE}" --shards ${PS2X_VU_PROGRAM_SHARDS} + ${PS2X_VU_PROGRAM_PROFILES} + DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/../tools/compile_vu_programs.py" ${PS2X_VU_PROGRAM_LISTS} + VERBATIM) + add_custom_target(ps2_vu_programs DEPENDS "${PS2X_VU_PROGRAM_INCLUDE}" + "${PS2X_VU_PROGRAM_EXTERN_INCLUDE}" ${PS2X_VU_PROGRAM_SOURCES}) + add_dependencies(ps2_runtime ps2_vu_programs) + target_sources(ps2_runtime PRIVATE ${PS2X_VU_PROGRAM_SOURCES}) + set_source_files_properties(src/lib/vu/ps2_vu1_program.cpp PROPERTIES COMPILE_DEFINITIONS + "PS2X_VU_PROGRAM_INCLUDE=\"${PS2X_VU_PROGRAM_INCLUDE}\";PS2X_VU_PROGRAM_EXTERN_INCLUDE=\"${PS2X_VU_PROGRAM_EXTERN_INCLUDE}\"") + set_source_files_properties(${PS2X_VU_PROGRAM_SOURCES} PROPERTIES + SKIP_UNITY_BUILD_INCLUSION ON + INCLUDE_DIRECTORIES "${CMAKE_CURRENT_SOURCE_DIR}/src/lib/vu") + if(CMAKE_CXX_COMPILER_ID MATCHES "Clang|GNU") + set_source_files_properties(${PS2X_VU_PROGRAM_SOURCES} PROPERTIES COMPILE_OPTIONS "-g0") + endif() +endif() + if(PS2X_ENABLE_RUNTIME_LOGS) target_compile_definitions(ps2_runtime PUBLIC PS2_RUNTIME_LOGS=1 @@ -552,6 +660,11 @@ if(PS2X_IS_VITA) endif() endif() +# Every configuration, not just Release: SSE4.1 is required to compile, not an +# optimisation. PUBLIC so recompiled game code linking against ps2_runtime +# inherits it -- that code is where most of the SSE4.1 intrinsics actually are. +EnableX86SimdBaseline(ps2_runtime) + if(CMAKE_BUILD_TYPE STREQUAL "Release" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") EnableFastReleaseMode(ps2_runtime) EnableFastReleaseMode(ps2EntryRunner) diff --git a/ps2xRuntime/Readme.md b/ps2xRuntime/Readme.md index 4ae035fef..1e9a365e7 100644 --- a/ps2xRuntime/Readme.md +++ b/ps2xRuntime/Readme.md @@ -47,6 +47,35 @@ The runtime handles PS2's memory addressing, including: ## Vector Unit Support PS2-specific 128-bit MMI instructions and VU0 macro mode instructions are supported via SSE/AVX intrinsics. +### Compiled VU1 programs + +VU1 microprograms normally run on the interpreter. A build can instead compile +the programs a game actually runs into C++ ahead of time, whole, from the entry +point to the E bit: every pair still issues in order with the interpreter's +stall rules, but the per-pair bookkeeping is decided at compile time and the +hot state stays in registers. Anything the compiled code cannot express, or +microcode that changed since it was recorded, falls back to the interpreter. + +The input is recorded, not guessed, because the microcode is game data: + +1. Run the game with `PS2_VU_PROGRAM_PROFILE=some/directory`. Each entry point + reached without a compiled routine is saved there as a 16 KiB code image + plus a line in `entries.txt`, and so is every place a register jump lands, + so one run covers the code it went through. Recording keeps adding to the + directory. +2. Configure with `-DPS2X_VU_PROGRAM_PROFILES=dir1;dir2` and rebuild. The + build runs `tools/compile_vu_programs.py` over the recordings and compiles + the result in `PS2X_VU_PROGRAM_SHARDS` units (8 by default). + +Recordings and the generated sources contain game code; keep them local. + +The FMAC arithmetic has fast paths with exact fallbacks, NEON on AArch64 and +SSE2 on x86. Other hosts get a plain C++ path. All three can be tested on an ARM +machine: `PS2X_VU_PROGRAM_SSE` builds the SSE2 one through sse2neon, and +`PS2X_VU_PROGRAM_PORTABLE` the plain one. `ps2xTest/data/vu_programs.py` writes +synthetic programs for the differential test in +`ps2xTest/src/ps2_vu1_program_tests.cpp`. + ## Instruction Patching You can patch specific instructions in the recompiled code to fix game issues or implement custom behavior. diff --git a/ps2xRuntime/cmake/ReleaseMode.cmake b/ps2xRuntime/cmake/ReleaseMode.cmake index 92ef0d54d..0b97e9016 100644 --- a/ps2xRuntime/cmake/ReleaseMode.cmake +++ b/ps2xRuntime/cmake/ReleaseMode.cmake @@ -2,6 +2,26 @@ include(CheckIPOSupported) check_ipo_supported(RESULT IPO_SUPPORTED OUTPUT IPO_ERROR) +# ps2_runtime.h unconditionally includes and the recompiler emits +# SSE4.1-only intrinsics (_mm_blendv_ps and friends) for the COP2/FPU select +# idioms, so SSE4.1 is a hard requirement of the codebase rather than a tuning +# knob. MSVC enables it implicitly (its intrinsics are not gated by a target +# feature), but GCC and Clang default to the plain x86-64 baseline, which is +# SSE2 -- so on any stock Linux or macOS toolchain those translation units fail +# to compile with "always_inline function ... requires target feature 'sse4.1'". +# +# This must not live in EnableFastReleaseMode: that is only applied for Release +# and RelWithDebInfo, whereas the requirement applies to every configuration. +function(EnableX86SimdBaseline TargetName) + if(MSVC) + return() + endif() + if(NOT CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64|i[3-6]86|x86)$") + return() + endif() + target_compile_options(${TargetName} PUBLIC -msse4.1) +endfunction() + function(EnableFastReleaseMode TargetName) message("> Enabling optimization for: ${TargetName}") if(MSVC) diff --git a/ps2xRuntime/include/ps2_runtime.h b/ps2xRuntime/include/ps2_runtime.h index a899408aa..44c10b4df 100644 --- a/ps2xRuntime/include/ps2_runtime.h +++ b/ps2xRuntime/include/ps2_runtime.h @@ -150,10 +150,33 @@ struct alignas(16) R5900Context // Reset COP0 registers cop0_random = 47; // Start at maximum value - // cop0_status = 0x400000; // BEV set, ERL clear, kernel mode - // 0x00400000 = BEV (Boot Exception Vectors). - // 0x00000000 = Normal mode (after BIOS handoff). - cop0_status = 0x00000000; + // Status as the EE kernel leaves it when it hands control to the game, + // which is the state recompiled code starts in -- we never execute the + // boot ROM that would otherwise set this up. + // + // Both interrupt-enable bits matter, and they are not the same bit: + // + // IE (bit 0) the architectural MIPS interrupt enable. The kernel + // sets it once during boot and it normally stays set. + // EIE (bit 16) the EE-specific enable that the `ei` and `di` + // instructions toggle. + // + // Interrupts are only really on when both are set, and guest code reads + // them separately. libkernel's StartThread, for instance, opens with + // `mfc0 Status; xori 1; andi 1` and refuses to run when IE is clear -- + // that is its "you must call iStartThread from an interrupt handler" + // guard. Leaving Status at zero made that guard fire forever, so every + // StartThread returned -1 and any game that creates a thread stalled + // with no diagnostic. + // + // EIE matters for the matching reason: DIntr reports whether it was set + // so the caller knows whether to pair it with an EIntr. Starting at zero + // makes DIntr always answer "already disabled" and the re-enable never + // happens. + // + // BEV (0x00400000) is deliberately not set: that selects the boot + // exception vectors, which is the pre-handoff state, not this one. + cop0_status = 0x00010001; // EIE | IE cop0_prid = 0x00002e20; // CPU ID for R5900 in_delay_slot = false; @@ -296,6 +319,17 @@ class PS2Runtime DebugUiCallback shutdownCallback, void *userData); + // A host that owns its own window and shows frames itself. When one is + // installed the runtime opens no window of its own and does no drawing: + // the raster backend has already presented by the time Present() returns, + // and this is called once per frame to service the window. Returning false + // asks the runtime to stop, the way a close button would. + // + // Set before initialize(), which is where the built-in window is created. + using FramePumpCallback = bool (*)(void *userData); + void setExternalPresenter(FramePumpCallback pump, void *userData); + [[nodiscard]] bool hasExternalPresenter() const { return m_framePump != nullptr; } + using RecompiledFunction = void (*)(uint8_t *, R5900Context *, PS2Runtime *); enum class GuestBranchKind @@ -322,6 +356,25 @@ class PS2Runtime SkipCallDebug = 3, }; + // Overlaid guest code: one dense table per streamed-in image, so the same + // arena address can mean different code depending on which is resident. + // The host owns the images and answers "which table covers this address?". + struct FunctionRegion + { + uint32_t base = 0u; // first guest address covered + uint32_t end = 0u; // one past the last covered address + uint32_t slotCount = 0u; // entries in `slots` + RecompiledFunction *slots = nullptr; // dense, indexed (addr - base) >> 2 + }; + + // Return the region covering `address`, or nullptr. Called on the guest + // thread from dispatch, so it must be cheap and must not block. + using FunctionRegionResolver = FunctionRegion *(*)(uint32_t address, void *userData); + + // Global rather than per-instance to match the generated table it extends. + // Passing nullptr removes the resolver. + static void setFunctionRegionResolver(FunctionRegionResolver resolver, void *userData); + bool replaceFunction(uint32_t address, RecompiledFunction func); // TODO remove this later need to update all tests bool registerFunction(uint32_t address, RecompiledFunction func); @@ -370,8 +423,18 @@ class PS2Runtime uint32_t guestRealloc(uint32_t guestAddr, uint32_t newSize, uint32_t alignment = 16u); void guestFree(uint32_t guestAddr); uint32_t guestHeapBase() const; + + // Ceiling for a guest that grows its own heap through sbrk/EndOfHeap. + // Zero (default) makes EndOfHeap report guestHeapLimit(), as before. + void setGuestHeapCeiling(uint32_t ceiling); + uint32_t guestHeapCeiling() const; + uint32_t guestHeapEnd() const; uint32_t guestHeapLimit() const; + + // Highest address any heap may reach. Unlike guestHeapLimit() this is + // valid before the heap is configured, so a caller can split the range. + uint32_t guestHeapHardLimit() const; uint32_t reserveAsyncCallbackStack(uint32_t size, uint32_t alignment = 16u); void drainCompletedDmacHandlers(uint8_t *rdram); @@ -492,6 +555,7 @@ class PS2Runtime uint32_t m_guestHeapBase = 0x00100000u; uint32_t m_guestHeapEnd = 0x00100000u; uint32_t m_guestHeapLimit = PS2_RAM_SIZE; + uint32_t m_guestHeapCeiling = 0u; uint32_t m_guestHeapSuggestedBase = 0x00100000u; bool m_guestHeapConfigured = false; uint32_t m_asyncCallbackStackFloor = 0x01F00000u; @@ -500,6 +564,8 @@ class PS2Runtime std::atomic m_missingFunctionPolicy{static_cast(MissingFunctionPolicy::ContinueToTarget)}; std::atomic m_missingFunctionReported{false}; std::atomic m_stopRequested{false}; + FramePumpCallback m_framePump = nullptr; + void *m_framePumpUserData = nullptr; DebugUiCallback m_debugUiInitCallback = nullptr; DebugUiCallback m_debugUiDrawCallback = nullptr; DebugUiCallback m_debugUiShutdownCallback = nullptr; diff --git a/ps2xRuntime/include/runtime/ee_fiber.h b/ps2xRuntime/include/runtime/ee_fiber.h new file mode 100644 index 000000000..c802f84a4 --- /dev/null +++ b/ps2xRuntime/include/runtime/ee_fiber.h @@ -0,0 +1,48 @@ +#pragma once + +#include +#include + +// A stack a guest thread can be suspended on and resumed into. +// +// The EE scheduler used to transfer control by throwing EeDispatcherTransfer, +// which unwinds the guest's whole C++ call chain and then rebuilds it on the +// way back. That cost ~10us per switch against the ~540ns a 60fps movie frame +// can afford, because DQ8's movie thread yields ~31,000 times per frame. +// Suspending the stack instead makes a switch a register save and a stack +// pointer swap. +// +// x86-64 and ARM64 get native switches; other targets fall back to ucontext, +// which is correct but pays a sigprocmask syscall per switch. setjmp/longjmp +// across stacks is deliberately not used: glibc's _FORTIFY_SOURCE turns it +// into an abort ("longjmp causes uninitialized stack frame"). +class EeFiber +{ +public: + using EntryFn = void (*)(void *user); + + EeFiber() = default; + ~EeFiber(); + + EeFiber(const EeFiber &) = delete; + EeFiber &operator=(const EeFiber &) = delete; + + // Allocates the stack and arms `entry`; entry must never return. + bool create(EntryFn entry, void *user, size_t stackBytes); + [[nodiscard]] bool valid() const noexcept { return m_stack != nullptr; } + void destroy(); + + // Called from the scheduler: run this fiber until it switches back. + void resume(); + // Called from inside the fiber: hand control back to whoever resumed it. + void suspend(); + + [[nodiscard]] static bool usingFastSwitch() noexcept; + +private: + void *m_stack = nullptr; // lowest address of the allocation + size_t m_stackBytes = 0u; + void *m_fiberSp = nullptr; // suspended fiber stack pointer + void *m_returnSp = nullptr; // stack pointer of whoever called resume() + void *m_platform = nullptr; // ucontext pair on the fallback path +}; diff --git a/ps2xRuntime/include/runtime/ee_scheduler.h b/ps2xRuntime/include/runtime/ee_scheduler.h index fae8b8540..9bb137a49 100644 --- a/ps2xRuntime/include/runtime/ee_scheduler.h +++ b/ps2xRuntime/include/runtime/ee_scheduler.h @@ -1,6 +1,9 @@ #pragma once #include "ps2_runtime.h" +#include "runtime/ee_fiber.h" + +#include #include #include @@ -122,6 +125,11 @@ struct GuestThread EeWaitState wait{}; std::function resumeCompletion; std::vector invocations; + // Host stack this thread's guest code runs on, created on first dispatch. + // While inGuestCall is set the stack is suspended part way through a guest + // call and must be resumed, not re-entered from the saved pc. + std::unique_ptr fiber; + bool inGuestCall = false; [[nodiscard]] R5900Context &activeContext() { @@ -181,6 +189,8 @@ struct EeThreadSnapshot { int id = 0; uint32_t pc = 0; + uint32_t ra = 0; + uint32_t sp = 0; uint32_t entry = 0; uint32_t stack = 0; uint32_t stackSize = 0; @@ -315,6 +325,7 @@ class EeScheduler [[noreturn]] void invokeCurrentSequence(std::vector invocations); [[nodiscard]] bool hasInvocation(GuestInvocationKind kind, uint64_t tag) const; [[nodiscard]] uint32_t invocationStackTop(); + void releaseInvocationStacks(int threadId); int addIrqHandler(bool dmac, uint32_t cause, @@ -353,9 +364,14 @@ class EeScheduler void bindMainContextForSyscall(R5900Context &ctx, uint8_t *rdram); [[nodiscard]] EeKernelSnapshot snapshot() const; + // Cheap unless snapshot() has asked for a refresh since the last build. void publishSnapshot(); + // Builds unconditionally; for the idle path, where no later scheduler + // operation is guaranteed to service a pending request. + void publishSnapshotNow(); private: + friend struct EeSchedulerPacingTestAccess; struct ScheduledEvent { uint64_t deadlineCycle = 0; @@ -367,13 +383,28 @@ class EeScheduler void assertExecutor() const; [[nodiscard]] int allocateThreadId(); GuestThread &acquireInvocationThread(); + void reserveGuestStackFromAsyncPool(uint32_t guestStackBase); void enqueueReady(GuestThread &thread, bool front = false); void removeReady(GuestThread &thread); + // Runs one guest dispatch on `thread`'s fiber, creating it if needed, and + // returns when the fiber suspends or the call completes. + void enterGuest(GuestThread &thread); + // The fiber's body: run the pending dispatch, hand control back, repeat. + static void fiberEntry(void *user); + void runPendingGuestCall(); + [[nodiscard]] GuestThread *selectReady(); + // Call after any change to m_readyQueues[priority]. + void refreshReadyMask(int priority) noexcept; + // Highest-priority (lowest index) non-empty queue, or -1 when none. + [[nodiscard]] int firstReadyPriority() const noexcept; void makeRunning(GuestThread &thread); void makeDormant(GuestThread &thread); void removeFromWaitObject(GuestThread &thread); [[noreturn]] void blockCurrent(EeWaitState wait); + // Suspends the guest stack instead of unwinding it; see the definition for + // when that is allowed. + void blockCurrentResumable(EeWaitState wait); void makeReady(GuestThread &thread, int result, bool interruptSafe); void requestPreemptionIfHigher(const GuestThread &readyThread, bool interruptSafe); void applyPendingPreemption(); @@ -394,7 +425,14 @@ class EeScheduler PS2Runtime &m_runtime; uint8_t *m_rdram = nullptr; std::array, kPriorityCount> m_readyQueues{}; + // One bit per priority, set while that queue is non-empty. selectReady() + // used to scan all 128 deques on every dispatch, which was ~7% of EE thread + // time; with this it is a count-trailing-zeros. Kept in sync by + // refreshReadyMask(), which every mutation of m_readyQueues calls. + std::array m_readyMask{}; std::unordered_map m_threads; + // EE thread IDs are bounded; map nodes remain stable across rehashes. + std::array m_threadIndex{}; std::unordered_map m_semaphores; std::unordered_map m_eventFlags; std::unordered_map m_alarms; @@ -413,6 +451,8 @@ class EeScheduler int m_dmacTailOrder = 1000; uint32_t m_enabledIntcMask = 0xFFFFFFFFu; uint32_t m_enabledDmacMask = 0xFFFFFFFFu; + uint32_t m_pendingIntcMask = 0u; + uint32_t m_pendingDmacMask = 0u; int m_currentThreadId = 0; bool m_rescheduleRequested = false; bool m_timeSliceExpired = false; @@ -426,10 +466,30 @@ class EeScheduler std::atomic m_stopRequested{false}; std::atomic m_checkpointPending{false}; uint32_t m_debugPublishCountdown = 0u; + // The thread whose C++ stack the executor is standing in, or 0 outside a + // guest dispatch. A yield back onto this thread can skip the unwind. + int m_dispatchedThreadId = 0; + + // Guest code runs on a per-thread fiber so a yield can suspend the stack + // instead of unwinding it. Null while the executor is on its own stack. + EeFiber *m_activeFiber = nullptr; + // Handed to the fiber for the dispatch it is about to run. + PS2Runtime::RecompiledFunction m_pendingFunction = nullptr; + R5900Context *m_pendingContext = nullptr; + bool m_pendingInsideInterrupt = false; + // A non-transfer exception escaping guest code, rethrown by the executor + // once it is back on its own stack. + std::exception_ptr m_fiberException; mutable std::mutex m_eventMutex; std::condition_variable m_eventCv; std::deque m_events; + // Nonzero while m_events has anything in it. The post-dispatch pump's + // tail used to take m_eventMutex to ask m_events.empty(); at the movie + // thread's ~4M pumps a second that mutex pair was measurable, and the + // count answers the same question without it. postEvent increments under + // m_eventMutex, the pump zeroes it under the same mutex when it drains. + std::atomic m_pendingEventCount{0}; std::vector m_deadlines; std::deque m_pendingInvocations; uint64_t m_eventSequence = 0; @@ -441,9 +501,15 @@ class EeScheduler uint32_t m_gsVSyncCallbackGp = 0; uint32_t m_gsVSyncCallbackSp = 0; std::unordered_map m_invocationStackTops; + // Freed by releaseInvocationStacks() when a thread record goes away; the + // underlying pool only bumps down and cannot hand memory back itself. + std::vector m_freeInvocationStacks; std::atomic m_nextDeadlineCycle{0}; mutable std::mutex m_snapshotMutex; EeKernelSnapshot m_snapshot; uint64_t m_snapshotSequence = 0; + // Set by snapshot(), cleared by publishSnapshot(). The snapshot is debug + // state, so it is only worth building when someone has asked for it. + mutable std::atomic m_snapshotWanted{false}; }; diff --git a/ps2xRuntime/include/runtime/gs/gs_backend.h b/ps2xRuntime/include/runtime/gs/gs_backend.h index 9419cc23a..2b41fd51b 100644 --- a/ps2xRuntime/include/runtime/gs/gs_backend.h +++ b/ps2xRuntime/include/runtime/gs/gs_backend.h @@ -3,8 +3,16 @@ #include "runtime/gs/gs_types.h" #include +#include #include +struct GSPreparedPresentation +{ + virtual ~GSPreparedPresentation() = default; + uint64_t sourceVsyncTick = 0u; +}; +using GSPresentationTicket = std::shared_ptr; + class GSRasterBackend { public: @@ -30,4 +38,14 @@ class GSRasterBackend virtual void WriteVram(uint32_t psm, uint32_t base, uint32_t bw, uint32_t x, uint32_t y, uint32_t value) = 0; virtual void SnapshotVram(std::vector &out) const = 0; virtual GSTransferSnapshot GetTransferSnapshot() const = 0; + + // Display has one consumer on the window thread; concurrent repeats are + // unsupported. The owner outlives calls, but tickets may outlive the owner. + virtual bool SupportsPreparedPresentation() const { return false; } + // Only queued preparation may run under the frontend's state lock. + virtual bool QueuesPreparedPresentation() const { return false; } + virtual GSPresentationTicket PreparePresentation(const GSPresentationRequest &) { return {}; } + virtual PresentationFrame DisplayPreparedPresentation(const GSPresentationTicket &) { return {}; } + // Shutdown must wake a producer blocked waiting for a free snapshot slot. + virtual void CancelPreparedPresentations() noexcept {} }; diff --git a/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h b/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h index 71e80324f..c428fed9a 100644 --- a/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h +++ b/ps2xRuntime/include/runtime/gs/gs_cpu_backend.h @@ -56,7 +56,8 @@ class GSCpuBackend final : public GSRasterBackend bool useLocalMemoryLayout, bool frameBaseIsPages, uint32_t sourceOriginX, - uint32_t sourceOriginY) const; + uint32_t sourceOriginY, + bool doubleSourceRows = false) const; using WriteVramFunc = std::function; using ReadVramFunc = std::function; diff --git a/ps2xRuntime/include/runtime/gs/gs_frontend.h b/ps2xRuntime/include/runtime/gs/gs_frontend.h index 4c3f6410a..b754133d5 100644 --- a/ps2xRuntime/include/runtime/gs/gs_frontend.h +++ b/ps2xRuntime/include/runtime/gs/gs_frontend.h @@ -225,6 +225,7 @@ class GS std::vector m_hostPresentationFrame; uint32_t m_hostPresentationWidth = 0; uint32_t m_hostPresentationHeight = 0; + uint32_t m_hostPresentationRowPitchBytes = 0; uint32_t m_hostPresentationDisplayFbp = 0; uint32_t m_hostPresentationSourceFbp = 0; bool m_hostPresentationUsedPreferred = false; diff --git a/ps2xRuntime/include/runtime/gs/gs_threaded_backend.h b/ps2xRuntime/include/runtime/gs/gs_threaded_backend.h new file mode 100644 index 000000000..ccfadf68d --- /dev/null +++ b/ps2xRuntime/include/runtime/gs/gs_threaded_backend.h @@ -0,0 +1,41 @@ +#pragma once + +#include "runtime/gs/gs_backend.h" + +#include +#include + +// Owns submitted data until the worker consumes it. Observations and GS +// completion barriers drain the ordered queue; presentation stays on its caller. +class GSThreadedBackend final : public GSRasterBackend +{ +public: + explicit GSThreadedBackend(std::unique_ptr backend, + size_t queueBytes = 4u * 1024u * 1024u); + ~GSThreadedBackend() override; + + void Initialize(uint8_t *vram, uint32_t size) override; + void Reset() override; + void Submit(const GSPrimitiveBatch &batch) override; + void BeginTransfer(const GSTransferCommand &command) override; + void UploadImage(const uint8_t *data, uint32_t size) override; + void Flush() override; + void TextureFlush() override; + void Sync(GSSyncReason reason) override; + PresentationFrame Present(const GSPresentationRequest &request) override; + bool SupportsPreparedPresentation() const override; + bool QueuesPreparedPresentation() const override { return SupportsPreparedPresentation(); } + GSPresentationTicket PreparePresentation(const GSPresentationRequest &request) override; + PresentationFrame DisplayPreparedPresentation(const GSPresentationTicket &ticket) override; + void CancelPreparedPresentations() noexcept override; + bool ClearFramebuffer(const GSContext &context, uint32_t rgba) override; + uint32_t ConsumeLocalToHostBytes(uint8_t *dst, uint32_t size) override; + uint32_t ReadVram(uint32_t psm, uint32_t base, uint32_t bw, uint32_t x, uint32_t y) const override; + void WriteVram(uint32_t psm, uint32_t base, uint32_t bw, uint32_t x, uint32_t y, uint32_t value) override; + void SnapshotVram(std::vector &out) const override; + GSTransferSnapshot GetTransferSnapshot() const override; + +private: + struct Impl; + std::unique_ptr m_impl; +}; diff --git a/ps2xRuntime/include/runtime/gs/gs_types.h b/ps2xRuntime/include/runtime/gs/gs_types.h index 836a3ff49..0e0b79b1d 100644 --- a/ps2xRuntime/include/runtime/gs/gs_types.h +++ b/ps2xRuntime/include/runtime/gs/gs_types.h @@ -3,6 +3,7 @@ #include #include #include +#include #include enum GSPrimType : uint8_t @@ -290,18 +291,52 @@ struct GSPresentationRequest bool hasPreferredSource = false; }; +enum class GSPresentationMode : uint8_t +{ + // The backend returned host-readable RGBA8 pixels. rowPitchBytes describes + // their layout and may be wider than width * 4. + HostPixels, + // The backend presented through its own native swapchain. There is no host + // pixel payload to copy through the legacy presentation path. + BackendNative, +}; + struct PresentationFrame { std::vector pixels; uint32_t width = 0; uint32_t height = 0; + uint32_t rowPitchBytes = 0; uint32_t displayFbp = 0; uint32_t sourceFbp = 0; bool usedPreferred = false; + GSPresentationMode mode = GSPresentationMode::HostPixels; + + bool HasHostPixels() const + { + if (mode != GSPresentationMode::HostPixels || width == 0u || height == 0u) + return false; + + if (static_cast(width) > std::numeric_limits::max() / 4u) + return false; + const size_t packedRowBytes = static_cast(width) * 4u; + const size_t sourceRowBytes = rowPitchBytes != 0u ? rowPitchBytes : packedRowBytes; + if (sourceRowBytes < packedRowBytes) + return false; + + if (height > 1u && + sourceRowBytes > (std::numeric_limits::max() - packedRowBytes) / + static_cast(height - 1u)) + return false; + const size_t requiredBytes = sourceRowBytes * static_cast(height - 1u) + packedRowBytes; + return pixels.size() >= requiredBytes; + } explicit operator bool() const { - return !pixels.empty() && width != 0u && height != 0u; + if (width == 0u || height == 0u) + return false; + return mode == GSPresentationMode::BackendNative || HasHostPixels(); } }; diff --git a/ps2xRuntime/include/runtime/gs/ps2_gs_memory.h b/ps2xRuntime/include/runtime/gs/ps2_gs_memory.h index 095e4e5cf..e4c5873b0 100644 --- a/ps2xRuntime/include/runtime/gs/ps2_gs_memory.h +++ b/ps2xRuntime/include/runtime/gs/ps2_gs_memory.h @@ -541,6 +541,18 @@ namespace GSMem void WriteCT32(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 value); void WriteZ32(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 value); + // Writes `count` little-endian 32-bit pixels from `src`, starting at (x, y) + // and wrapping back to `dsax` when x reaches `rowEnd`. The per-pixel entry + // points above are out-of-line wrappers around a template whose lookup table + // is private to ps2_gs_memory.cpp, so a caller writing a run of pixels pays a + // cross-TU call each time and nothing inlines. A host-to-local transfer moves + // 256 pixels per 16x16 tile and DQ8's movies send ~900 tiles a frame, which + // made that the hottest thing in the runtime. + void WriteRunCT32(u8* data, u32 bp, u32 bw, u32 dsax, u32 rowEnd, u32 x, u32 y, + const u8* src, u32 count); + void WriteRunZ32(u8* data, u32 bp, u32 bw, u32 dsax, u32 rowEnd, u32 x, u32 y, + const u8* src, u32 count); + void WriteCT24(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 value); void WriteZ24(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 value); @@ -580,4 +592,13 @@ namespace GSMem u32 ReadP4HH(u8* data, u32 bp, u32 bw, u32 x, u32 y); u32 ReadNull(u8* data, u32 bp, u32 bw, u32 x, u32 y); + + // Reads `count` consecutive-x pixels starting at (x, y) into dst, packed + // at the PSM's storage width. Texture expansion reads whole rows; the + // per-pixel entry points recompute page, block row and column from (x, y) + // with three integer divisions each, and a 512x448 expand paid that 229k + // times. Same values, same order, increments instead of divisions. + void ReadRowCT32(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 count, u8* dst); + void ReadRowZ32(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 count, u8* dst); + void ReadRowP8(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 count, u8* dst); } \ No newline at end of file diff --git a/ps2xRuntime/include/runtime/ps2_memory.h b/ps2xRuntime/include/runtime/ps2_memory.h index cea5b98a8..de2f8ed76 100644 --- a/ps2xRuntime/include/runtime/ps2_memory.h +++ b/ps2xRuntime/include/runtime/ps2_memory.h @@ -313,6 +313,10 @@ class PS2Memory [[nodiscard]] uint64_t cyclesUntilNextEeTimerInterrupt() const noexcept; void resetEeTimers() noexcept; + // A VIFcode carrying the i bit raises INTC VIF0/VIF1. Bit 0 is VIF0, + // bit 1 is VIF1; the scheduler drains this and dispatches the handlers. + uint32_t takePendingVifInterrupts() noexcept; + using GifPacketCallback = std::function; void setGifPacketCallback(GifPacketCallback cb) { m_gifPacketCallback = std::move(cb); } void setGifArbiter(GifArbiter *arbiter) { m_gifArbiter = arbiter; } @@ -374,6 +378,7 @@ class PS2Memory std::atomic m_gifCopyCount{0}; std::atomic m_gsWriteCount{0}; std::atomic m_vifWriteCount{0}; + std::atomic m_pendingVifInterrupts{0}; std::atomic m_vu0CodeGeneration{0}; std::atomic m_vu1CodeGeneration{0}; // I/O registers @@ -449,6 +454,11 @@ class PS2Memory }; std::array m_eeTimers{}; + // True while any timer has CUE set. advanceEeTimers runs on every guest + // safe point; without this it walks all four timers each time even when + // the game never armed one. + bool m_anyEeTimerCued = false; + void refreshEeTimersCued() noexcept; void queueCompletedDmacCause(uint32_t cause); }; diff --git a/ps2xRuntime/include/runtime/ps2_native_iop.h b/ps2xRuntime/include/runtime/ps2_native_iop.h new file mode 100644 index 000000000..4fcc207ce --- /dev/null +++ b/ps2xRuntime/include/runtime/ps2_native_iop.h @@ -0,0 +1,23 @@ +#pragma once + +#include +#include + +class PS2Runtime; + +// Modules named here load onto an emulated IOP with an SPU2 instead of being +// stubbed: the game's RPCs run their real code, and what the SPU2 mixes plays +// on the host audio device. +namespace ps2_native_iop +{ + // File names as the game loads them, for example "LIBSD.IRX". Call this + // before the game starts. + void setModules(PS2Runtime &runtime, std::vector names); + // 0 mutes the output. The IOP keeps running in step with the device either + // way, since games wait on their sound driver. + void setVolume(float volume); + + // Called by the runtime; the stream starts with the first native module. + void startAudio(PS2Runtime &runtime); + void stopAudio(); +} diff --git a/ps2xRuntime/include/runtime/ps2_pad_host.h b/ps2xRuntime/include/runtime/ps2_pad_host.h new file mode 100644 index 000000000..3cbd06994 --- /dev/null +++ b/ps2xRuntime/include/runtime/ps2_pad_host.h @@ -0,0 +1,25 @@ +#ifndef PS2_PAD_HOST_H +#define PS2_PAD_HOST_H + +#include + +// Render-thread half of the pad backend. Samples raylib input once per presented +// frame and latches press edges so a tap between two guest polls still registers. +// Declared here rather than on PSPadBackend so the recompiled corpus, which +// includes ps2_pad.h through ps2_runtime.h, does not rebuild for input changes. +void ps2PadPollHost(); + +// The same latching, driven by a host that is not raylib. A backend owning its +// own window samples input itself and publishes it here, in the pad report's +// button encoding, with `pressed` carrying edges seen since the last call. +void ps2PadPublishHostState(uint32_t held, uint32_t pressed, uint32_t sticks); + +// Clock for DQ8_PAD_SCRIPT, in guest vsync ticks rather than host frames: the +// game runs well under 60 fps, so a script timed in host frames would fire at +// the wrong point in the game and replay differently on every run. +void ps2PadSetGuestFrame(uint64_t frame); + +// The same clock, for anything that wants to name its output in script time. +uint64_t ps2PadCurrentGuestFrame(); + +#endif diff --git a/ps2xRuntime/include/runtime/ps2_vu1.h b/ps2xRuntime/include/runtime/ps2_vu1.h index 67c8184fd..b32db25ea 100644 --- a/ps2xRuntime/include/runtime/ps2_vu1.h +++ b/ps2xRuntime/include/runtime/ps2_vu1.h @@ -7,6 +7,11 @@ class GS; class PS2Memory; +namespace ps2_vu_program +{ + struct Access; +} + struct VU1State { float vf[32][4]; @@ -62,7 +67,14 @@ class VU1Interpreter VU1State &state() { return m_state; } const VU1State &state() const { return m_state; } + void setCompiledExecutionEnabled(bool enabled) { m_useCompiledExecution = enabled; } + uint64_t compiledPairsExecuted() const { return m_compiledPairsExecuted; } + uint64_t interpretedPairsExecuted() const { return m_interpretedPairsExecuted; } + private: + // Whole-program compiled code reads and publishes interpreter state. + friend struct ps2_vu_program::Access; + enum Pipeline : uint8_t { PipelineNone = 0, @@ -200,6 +212,12 @@ class VU1Interpreter static constexpr uint32_t kMaxPendingAccWrites = 8u; static constexpr uint32_t kMaxDecodedPairs = 0x4000u / 8u; + using CompiledBlock = bool (*)(VU1Interpreter &, uint64_t); + std::array m_compiledCodeCache{}; + bool m_useCompiledExecution = true; + uint64_t m_compiledPairsExecuted = 0; + uint64_t m_interpretedPairsExecuted = 0; + Unit m_unit; VU1State m_state; std::array m_decodedCodeCache{}; @@ -216,6 +234,11 @@ class VU1Interpreter std::array m_vfWritePipeline{}; std::array m_viWritePipeline{}; std::array m_accWritePipeline{}; + uint32_t m_activeFlags = 0; + uint32_t m_activeStores = 0; + uint32_t m_activeVfWrites = 0; + uint32_t m_activeViWrites = 0; + uint32_t m_activeAccWrites = 0; XgkickPipeline m_xgkick{}; std::array, 32> m_vfReady{}; std::array m_viReady{}; @@ -244,17 +267,25 @@ class VU1Interpreter uint8_t *vuData, uint32_t dataSize, GS &gs, PS2Memory *memory, uint32_t maxCycles); - InstructionUsage decodeUpperUsage(uint32_t upper) const; - InstructionUsage decodeLowerUsage(uint32_t lower) const; - static void addVfRead(InstructionUsage &usage, uint8_t reg, uint8_t lanes); - static void addVfWrite(InstructionUsage &usage, uint8_t reg, uint8_t lanes); - static uint8_t vfReadLanes(const InstructionUsage &usage, uint8_t reg); + static constexpr InstructionUsage decodeUpperUsage(uint32_t upper); + static constexpr InstructionUsage decodeLowerUsage(uint32_t lower, Unit unit); + static constexpr void addVfRead(InstructionUsage &usage, uint8_t reg, uint8_t lanes); + static constexpr void addVfWrite(InstructionUsage &usage, uint8_t reg, uint8_t lanes); + static constexpr uint8_t vfReadLanes(const InstructionUsage &usage, uint8_t reg); + static constexpr DecodedInstructionPair decodeInstructionWords(uint32_t lower, uint32_t upper, Unit unit); DecodedInstructionPair decodeInstructionPair(const uint8_t *vuCode, uint32_t pc) const; DecodedInstructionPair getDecodedInstructionPairForPc(const uint8_t *vuCode, uint32_t codeSize, PS2Memory *memory, uint32_t pc); void rebuildDecodedCodeCache(const uint8_t *vuCode, uint32_t codeSize, const PS2Memory *memory, uint64_t generation); + static CompiledBlock findCompiledBlock(const uint8_t *code, uint32_t size, Unit unit); + template + static bool runCompiledBlock(VU1Interpreter &vu, uint64_t budgetEnd); + template + bool runDecodedPair(const DecodedInstructionPair &decoded, uint64_t budgetEnd, uint32_t codeSize); void execUpper(uint32_t instr); void execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSize, GS &gs, PS2Memory *memory, uint32_t upperInstr); + void execUpperInline(uint32_t instr); + void execLowerInline(uint32_t instr, uint8_t *vuData, uint32_t dataSize, GS &gs, PS2Memory *memory, uint32_t upperInstr); void applyDest(float *dst, const float *result, uint8_t dest); void applyDestAcc(const float *result, uint8_t dest); @@ -284,7 +315,9 @@ class VU1Interpreter void progressXgkick(); void finishXgkick(); uint64_t calculatePairReadyCycle(const DecodedInstructionPair &decoded) const; + uint64_t calculatePairReadyCycleInline(const DecodedInstructionPair &decoded) const; void markPairWrites(const DecodedInstructionPair &decoded); + void markPairWritesInline(const DecodedInstructionPair &decoded); bool pipelinesPending() const; float normalizeOperand(float value) const; diff --git a/ps2xRuntime/src/lib/Kernel/EeFiber.cpp b/ps2xRuntime/src/lib/Kernel/EeFiber.cpp new file mode 100644 index 000000000..a70d51cf8 --- /dev/null +++ b/ps2xRuntime/src/lib/Kernel/EeFiber.cpp @@ -0,0 +1,380 @@ +#include "runtime/ee_fiber.h" + +#include +#include +#include + +#if defined(__unix__) || defined(__APPLE__) +#include +#include +#define EE_FIBER_HAVE_MMAP 1 +#else +#define EE_FIBER_HAVE_MMAP 0 +#endif + +namespace +{ + size_t stackPageSize() + { +#if EE_FIBER_HAVE_MMAP + static const size_t size = static_cast(sysconf(_SC_PAGESIZE)); + return size; +#else + return 4096u; +#endif + } + // Whole mapping including one guard page at the low end. + void *mmapStack(size_t bytes) + { +#if EE_FIBER_HAVE_MMAP + void *base = mmap(nullptr, bytes, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (base == MAP_FAILED) + { + return nullptr; + } + if (mprotect(base, stackPageSize(), PROT_NONE) != 0) + { + munmap(base, bytes); + return nullptr; + } + return base; +#else + return std::aligned_alloc(64u, bytes); +#endif + } + + void munmapStack(void *base, size_t bytes) + { +#if EE_FIBER_HAVE_MMAP + if (base != nullptr) + { + munmap(base, bytes); + } +#else + (void)bytes; + std::free(base); +#endif + } +} // namespace + +#if defined(__x86_64__) && (defined(__linux__) || defined(__unix__) || defined(__APPLE__)) && !defined(EE_FIBER_FORCE_UCONTEXT) +#define EE_FIBER_FAST_X86_64 1 +#else +#define EE_FIBER_FAST_X86_64 0 +#endif + +#if defined(__aarch64__) && (defined(__linux__) || defined(__APPLE__)) && !defined(EE_FIBER_FORCE_UCONTEXT) +#define EE_FIBER_FAST_ARM64 1 +#else +#define EE_FIBER_FAST_ARM64 0 +#endif +#define EE_FIBER_FAST (EE_FIBER_FAST_X86_64 || EE_FIBER_FAST_ARM64) + +#if !EE_FIBER_FAST +#if defined(__APPLE__) && !defined(_XOPEN_SOURCE) +#define _XOPEN_SOURCE 700 +#endif +#include +#endif + +namespace +{ +#if EE_FIBER_FAST_X86_64 + // void eeFiberSwitch(void **saveSp, void *targetSp) + // + // Saves the callee-saved registers and the FP control words on the current + // stack, records the resulting stack pointer through saveSp, switches to + // targetSp and restores the same set from there. The final `ret` lands on + // whatever return address that stack was suspended at -- for a fresh fiber, + // the trampoline create() planted. + extern "C" void eeFiberSwitch(void **saveSp, void *targetSp); + __asm__( + ".text\n" + ".globl eeFiberSwitch\n" + ".hidden eeFiberSwitch\n" + ".type eeFiberSwitch,@function\n" + ".align 16\n" + "eeFiberSwitch:\n" + " pushq %rbp\n" + " pushq %rbx\n" + " pushq %r12\n" + " pushq %r13\n" + " pushq %r14\n" + " pushq %r15\n" + " subq $8, %rsp\n" + " stmxcsr (%rsp)\n" + " fnstcw 4(%rsp)\n" + " movq %rsp, (%rdi)\n" + " movq %rsi, %rsp\n" + " ldmxcsr (%rsp)\n" + " fldcw 4(%rsp)\n" + " addq $8, %rsp\n" + " popq %r15\n" + " popq %r14\n" + " popq %r13\n" + " popq %r12\n" + " popq %rbx\n" + " popq %rbp\n" + " ret\n" + ".size eeFiberSwitch,.-eeFiberSwitch\n"); + + struct FiberBootstrap + { + EeFiber::EntryFn entry; + void *user; + }; + + // Reached by the first `ret` out of eeFiberSwitch. %rbx carries the + // bootstrap because create() planted it in the saved-register slot. + extern "C" void eeFiberTrampoline(); + __asm__( + ".text\n" + ".globl eeFiberTrampoline\n" + ".hidden eeFiberTrampoline\n" + ".type eeFiberTrampoline,@function\n" + ".align 16\n" + "eeFiberTrampoline:\n" + // Entered by `ret`, so rsp%16 == 8 as at any function entry. SysV wants + // rsp%16 == 0 immediately before a call; without this the callee's + // 16-byte SSE spills fault. + " subq $8, %rsp\n" + " movq %rbx, %rdi\n" + " call eeFiberEnter\n" + " hlt\n" // eeFiberEnter never returns + ".size eeFiberTrampoline,.-eeFiberTrampoline\n"); + + extern "C" void eeFiberEnter(void *bootstrap) + { + auto *boot = static_cast(bootstrap); + boot->entry(boot->user); + std::abort(); // entry must never return + } +#elif EE_FIBER_FAST_ARM64 + struct FiberBootstrap + { + EeFiber::EntryFn entry; + void *user; + }; + + extern "C" void eeFiberSwitch(void **saveSp, void *targetSp); + extern "C" void eeFiberTrampoline(); + extern "C" void eeFiberEnter(void *bootstrap) + { + auto *boot = static_cast(bootstrap); + boot->entry(boot->user); + std::abort(); + } + +#if defined(__APPLE__) +#define EE_ASM_SYMBOL(name) "_" #name +#else +#define EE_ASM_SYMBOL(name) #name +#endif + // AAPCS64: x19-x30, the low halves of v8-v15, and the FP environment. + // x18 is platform-reserved and must never be used as a scratch register. + __asm__( + ".text\n" + ".p2align 2\n" + ".globl " EE_ASM_SYMBOL(eeFiberSwitch) "\n" + EE_ASM_SYMBOL(eeFiberSwitch) ":\n" + "sub sp, sp, #176\n" + "stp x19, x20, [sp, #0]\n" + "stp x21, x22, [sp, #16]\n" + "stp x23, x24, [sp, #32]\n" + "stp x25, x26, [sp, #48]\n" + "stp x27, x28, [sp, #64]\n" + "stp x29, x30, [sp, #80]\n" + "stp d8, d9, [sp, #96]\n" + "stp d10, d11, [sp, #112]\n" + "stp d12, d13, [sp, #128]\n" + "stp d14, d15, [sp, #144]\n" + "mrs x9, fpcr\n" + "mrs x10, fpsr\n" + "stp x9, x10, [sp, #160]\n" + "mov x9, sp\n" + "str x9, [x0]\n" + "mov sp, x1\n" + "ldp x11, x12, [sp, #160]\n" + "cmp x9, x11\n" + "b.eq 1f\n" + "msr fpcr, x11\n" + "1:\n" + "cmp x10, x12\n" + "b.eq 2f\n" + "msr fpsr, x12\n" + "2:\n" + "ldp d8, d9, [sp, #96]\n" + "ldp d10, d11, [sp, #112]\n" + "ldp d12, d13, [sp, #128]\n" + "ldp d14, d15, [sp, #144]\n" + "ldp x19, x20, [sp, #0]\n" + "ldp x21, x22, [sp, #16]\n" + "ldp x23, x24, [sp, #32]\n" + "ldp x25, x26, [sp, #48]\n" + "ldp x27, x28, [sp, #64]\n" + "ldp x29, x30, [sp, #80]\n" + "add sp, sp, #176\n" + "ret\n" + ".p2align 2\n" + ".globl " EE_ASM_SYMBOL(eeFiberTrampoline) "\n" + EE_ASM_SYMBOL(eeFiberTrampoline) ":\n" + "mov x0, x19\n" + "bl " EE_ASM_SYMBOL(eeFiberEnter) "\n" + "brk #0\n"); +#undef EE_ASM_SYMBOL +#else + struct FiberPlatform + { + ucontext_t fiber{}; + ucontext_t caller{}; + EeFiber::EntryFn entry = nullptr; + void *user = nullptr; + }; + + FiberPlatform *g_startingFiber = nullptr; + + void fiberTrampoline() + { + FiberPlatform *self = g_startingFiber; + self->entry(self->user); + std::abort(); + } +#endif +} // namespace + +bool EeFiber::usingFastSwitch() noexcept +{ + return EE_FIBER_FAST != 0; +} + +EeFiber::~EeFiber() +{ + destroy(); +} + +void EeFiber::destroy() +{ +#if !EE_FIBER_FAST + delete static_cast(m_platform); +#else + std::free(m_platform); +#endif + m_platform = nullptr; + munmapStack(m_stack, m_stackBytes); + m_stack = nullptr; + m_stackBytes = 0u; + m_fiberSp = nullptr; + m_returnSp = nullptr; +} + +bool EeFiber::create(EntryFn entry, void *user, size_t stackBytes) +{ + destroy(); + if (entry == nullptr || stackBytes < 64u * 1024u) + { + return false; + } + // Guest call chains nest in C++ through dispatchGuestBranch, so a fiber + // stack can run deep. Map it with a PROT_NONE guard page at the low end: + // an overflow then faults on the guard instead of quietly writing into + // whatever the allocator put underneath. + const size_t pageSize = stackPageSize(); + const size_t mapped = ((stackBytes + pageSize - 1u) / pageSize) * pageSize + pageSize; + void *base = mmapStack(mapped); + if (base == nullptr) + { + return false; + } + void *stack = static_cast(base) + pageSize; + m_stack = base; + m_stackBytes = mapped; + stackBytes = mapped - pageSize; + +#if EE_FIBER_FAST + auto *boot = static_cast(std::malloc(sizeof(FiberBootstrap))); + if (boot == nullptr) + { + destroy(); + return false; + } + boot->entry = entry; + boot->user = user; + m_platform = boot; + +#if EE_FIBER_FAST_X86_64 + // Build the frame eeFiberSwitch will pop: FP words, r15..rbx, rbp, then the + // return address it rets to. rbx carries the bootstrap into the trampoline. + auto *top = reinterpret_cast(static_cast(stack) + stackBytes); + top = reinterpret_cast(reinterpret_cast(top) & ~static_cast(15)); + // SysV wants rsp % 16 == 8 on entry, i.e. the return-address slot 16-aligned. + *--top = 0u; // alignment padding + *--top = reinterpret_cast(&eeFiberTrampoline); // ret target + *--top = 0u; // rbp + *--top = reinterpret_cast(boot); // rbx + *--top = 0u; // r12 + *--top = 0u; // r13 + *--top = 0u; // r14 + *--top = 0u; // r15 + // mxcsr (low 32) + x87 control word (next 16), matching the switch's layout. + uint32_t mxcsr = 0x1F80u; + uint16_t fcw = 0x037Fu; + __asm__ __volatile__("stmxcsr %0" : "=m"(mxcsr)); + __asm__ __volatile__("fnstcw %0" : "=m"(fcw)); + --top; + std::memcpy(top, &mxcsr, sizeof(mxcsr)); + std::memcpy(reinterpret_cast(top) + 4, &fcw, sizeof(fcw)); + m_fiberSp = top; +#else + auto *top = reinterpret_cast(static_cast(stack) + stackBytes) - 22; + std::memset(top, 0, 176); + top[0] = reinterpret_cast(boot); // x19 + top[11] = reinterpret_cast(&eeFiberTrampoline); // x30 + __asm__ __volatile__("mrs %0, fpcr" : "=r"(top[20])); + __asm__ __volatile__("mrs %0, fpsr" : "=r"(top[21])); + m_fiberSp = top; +#endif +#else + auto *platform = new (std::nothrow) FiberPlatform(); + if (platform == nullptr) + { + destroy(); + return false; + } + platform->entry = entry; + platform->user = user; + if (getcontext(&platform->fiber) != 0) + { + delete platform; + destroy(); + return false; + } + platform->fiber.uc_stack.ss_sp = stack; + platform->fiber.uc_stack.ss_size = stackBytes; + platform->fiber.uc_link = nullptr; + makecontext(&platform->fiber, fiberTrampoline, 0); + m_platform = platform; +#endif + return true; +} + +void EeFiber::resume() +{ +#if EE_FIBER_FAST + eeFiberSwitch(&m_returnSp, m_fiberSp); +#else + auto *platform = static_cast(m_platform); + g_startingFiber = platform; + swapcontext(&platform->caller, &platform->fiber); +#endif +} + +void EeFiber::suspend() +{ +#if EE_FIBER_FAST + eeFiberSwitch(&m_fiberSp, m_returnSp); +#else + auto *platform = static_cast(m_platform); + swapcontext(&platform->fiber, &platform->caller); +#endif +} diff --git a/ps2xRuntime/src/lib/Kernel/EeHostPacing.h b/ps2xRuntime/src/lib/Kernel/EeHostPacing.h new file mode 100644 index 000000000..7b988fff4 --- /dev/null +++ b/ps2xRuntime/src/lib/Kernel/EeHostPacing.h @@ -0,0 +1,48 @@ +#pragma once + +#include +#include +#include + +namespace ee_host_pacing { +using Clock = std::chrono::steady_clock; + +struct VBlankBoundary { + Clock::time_point host; + Clock::duration lateness; + bool rebased; +}; + +inline VBlankBoundary vblankBoundary(Clock::time_point scheduled, Clock::time_point now, + Clock::duration fieldPeriod) { + const auto lateness = now > scheduled ? now - scheduled : Clock::duration::zero(); + const bool rebased = lateness > fieldPeriod; + // Keep one field of catch-up credit: a slow two-field game frame should + // not incur a fresh host wait between both already-earned guest VBlanks. + return {rebased ? now - fieldPeriod : scheduled, lateness, rebased}; +} + +// Internal deterministic-clock seam. Production keeps this null; tests install +// hooks only on their executor thread, without changing scheduler object layout. +struct ClockHooks { + void *context; + Clock::time_point (*now)(void *); + void (*waitUntil)(void *, Clock::time_point); +}; +inline thread_local const ClockHooks *clockHooks = nullptr; + +inline Clock::time_point now() { + return clockHooks ? clockHooks->now(clockHooks->context) : Clock::now(); +} + +template +bool waitUntil(std::condition_variable &cv, std::unique_lock &lock, + Clock::time_point deadline, Predicate predicate) { + if (!clockHooks) + return cv.wait_until(lock, deadline, predicate); + if (predicate()) + return true; + clockHooks->waitUntil(clockHooks->context, deadline); + return predicate(); +} +} diff --git a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp index 3a6ec7d94..5b51bf67d 100644 --- a/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp +++ b/ps2xRuntime/src/lib/Kernel/EeScheduler.cpp @@ -1,13 +1,52 @@ #include "runtime/ee_scheduler.h" +#include "EeHostPacing.h" #include "ps2_log.h" #include "ps2_runtime_macros.h" #include +#include #include +#include +#include #include #include #include +#include +#include +#include + +// How many times the executor re-entered guest code. The EE clock advances +// kGuestDispatchCycles per dispatch, so this is what paces the emulated frame. +std::atomic g_eeGuestDispatchCount{0}; +// Thrown EeDispatcherTransfers. Against g_eeGuestDispatchCount this says +// whether the executor is re-entering guest code because a transfer asked it +// to, or because the call chain has to be rebuilt after one. +std::atomic g_eeTransferThrowCount{0}; +// Transfers served by suspending the guest stack instead of unwinding it. +std::atomic g_eeTransferSuspendCount{0}; +// Temporary diagnostics for the movie-speed work: how often the run loop goes +// round, how many of those resume a suspended guest stack, and how often the +// post-dispatch event pump runs. Read by the sdlgpu stats line. +std::atomic g_eeRunLoopIterations{0}; +std::atomic g_eeRunLoopResumeCount{0}; +std::atomic g_eeProcessPendingEventsCount{0}; +std::atomic g_eeEnterGuestCount{0}; + +namespace +{ + const uint64_t g_eeCycleScale = [] { + const char *value = std::getenv("DQ8_EE_CYCLE_SCALE"); + const long long parsed = value ? std::atoll(value) : 0; + return parsed > 0 ? static_cast(parsed) : 1ull; + }(); + + const uint64_t g_eeDispatchHistogramInterval = [] { + const char *value = std::getenv("DQ8_EE_DISPATCH_HISTOGRAM"); + const long long parsed = value ? std::atoll(value) : 0; + return parsed > 0 ? static_cast(parsed) : 0ull; + }(); +} namespace { @@ -54,6 +93,40 @@ namespace constexpr uint64_t kVBlankDurationCycles = microsecondsToEeCycles(500u); constexpr uint64_t kAlarmTickCycles = microsecondsToEeCycles(kAlarmTickMicroseconds); + struct VBlankPacingTrace + { + std::chrono::steady_clock::time_point window{}; + uint64_t starts = 0, rebases = 0, maxLatenessUs = 0, lines = 0; + }; + thread_local VBlankPacingTrace g_vblankPacingTrace; + const bool g_traceVBlankPacing = std::getenv("DQ8_VBLANK_PACING_TRACE") != nullptr; + + void traceVBlankPacing(uint64_t tick, uint64_t cycle, + std::chrono::steady_clock::time_point now, + const ee_host_pacing::VBlankBoundary &boundary) + { + if (!g_traceVBlankPacing || g_vblankPacingTrace.lines >= 600u) + return; + auto &trace = g_vblankPacingTrace; + ++trace.starts; + trace.rebases += boundary.rebased; + const auto latenessUs = std::chrono::duration_cast(boundary.lateness).count(); + trace.maxLatenessUs = std::max(trace.maxLatenessUs, static_cast(latenessUs)); + if (trace.lines != 0u && now - trace.window < std::chrono::seconds(2)) + return; + const auto hostUs = std::chrono::duration_cast(now.time_since_epoch()).count(); + const auto nextHostUs = std::chrono::duration_cast( + (boundary.host + kVBlankPeriod).time_since_epoch()).count(); + std::fprintf(stderr, "[VBlank pacing] tick=%llu cycle=%llu host_us=%lld starts=%llu rebases=%llu max_late_us=%llu next_host_us=%lld\n", + static_cast(tick), static_cast(cycle), + static_cast(hostUs), static_cast(trace.starts), + static_cast(trace.rebases), static_cast(trace.maxLatenessUs), + static_cast(nextHostUs)); + trace.window = now; + trace.starts = trace.rebases = trace.maxLatenessUs = 0; + ++trace.lines; + } + template int allocatePositiveId(int &nextId, const Map &objects) { @@ -87,7 +160,16 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) m_executorThread = std::this_thread::get_id(); m_rdram = rdram; m_readyQueues = {}; + m_readyMask = {}; + // Thread ids restart here, so keyed stacks would otherwise be stranded: + // the pool below cannot reclaim them either. + for (const auto &[key, top] : m_invocationStackTops) + { + m_freeInvocationStacks.push_back(top); + } + m_invocationStackTops.clear(); m_threads.clear(); + m_threadIndex.fill(nullptr); m_semaphores.clear(); m_eventFlags.clear(); m_alarms.clear(); @@ -106,6 +188,8 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) m_dmacTailOrder = 1000; m_enabledIntcMask = 0xFFFFFFFFu; m_enabledDmacMask = 0xFFFFFFFFu; + m_pendingIntcMask = 0u; + m_pendingDmacMask = 0u; m_currentThreadId = 0; m_rescheduleRequested = false; m_timeSliceExpired = false; @@ -116,15 +200,18 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) m_stopRequested.store(false, std::memory_order_release); m_checkpointPending.store(false, std::memory_order_release); m_debugPublishCountdown = 0u; + m_dispatchedThreadId = 0; { std::lock_guard lock(m_eventMutex); m_events.clear(); + m_pendingEventCount.store(0u, std::memory_order_release); m_deadlines.clear(); m_pendingInvocations.clear(); } m_eventSequence = 0; m_invocationSequence = 0; m_vsyncTick = 0; + g_vblankPacingTrace = {}; m_vsyncFlagAddress = 0; m_vsyncTickAddress = 0; m_gsVSyncCallback = 0; @@ -144,12 +231,13 @@ void EeScheduler::reset(uint8_t *rdram, const R5900Context &mainContext) main.initialPriority = 0; main.currentPriority = 0; main.status = EeThreadStatus::Ready; - m_threads.emplace(main.id, std::move(main)); + m_threadIndex[kMainThreadId] = &m_threads.emplace(main.id, std::move(main)).first->second; m_readyQueues[0].push_back(kMainThreadId); + refreshReadyMask(0); scheduleEvent(m_eeCycle + kVBlankPeriodCycles, - std::chrono::steady_clock::now() + kVBlankPeriod, + ee_host_pacing::now() + kVBlankPeriod, EeEvent{EeEventType::VBlankStart, 0, 0}); - publishSnapshot(); + publishSnapshotNow(); } void EeScheduler::run() @@ -159,6 +247,8 @@ void EeScheduler::run() while (!m_stopRequested.load(std::memory_order_acquire)) { + g_eeRunLoopIterations.fetch_add(1u, std::memory_order_relaxed); + // Service events and preemption once between guest turns. processPendingEvents(); if (m_stopRequested.load(std::memory_order_acquire)) { @@ -170,7 +260,9 @@ void EeScheduler::run() GuestThread *next = selectReady(); if (!next && m_pendingInvocations.empty()) { - publishSnapshot(); + // About to sleep: nothing after this would service a pending + // snapshot request until the guest runs again. + publishSnapshotNow(); waitForEvent(); continue; } @@ -196,16 +288,38 @@ void EeScheduler::run() GuestThread *running = currentThread(); assert(running != nullptr); + + // Suspended part way through a guest call: its stack is still live, so + // resume it rather than dispatching the saved pc, which points into the + // middle of a function the fiber is already executing. + if (running->inGuestCall) + { + g_eeRunLoopResumeCount.fetch_add(1u, std::memory_order_relaxed); + enterGuest(*running); + if (m_fiberException) + { + std::exception_ptr escaped; + std::swap(escaped, m_fiberException); + m_running.store(false, std::memory_order_release); + publishSnapshotNow(); + std::rethrow_exception(escaped); + } + continue; + } + if (running->resumeCompletion) { auto completion = std::move(running->resumeCompletion); running->resumeCompletion = {}; try { + m_dispatchedThreadId = running->id; completion(running->activeContext()); + m_dispatchedThreadId = 0; } catch (const EeDispatcherTransfer &) { + m_dispatchedThreadId = 0; } if (m_currentThreadId == 0) { @@ -252,7 +366,12 @@ void EeScheduler::run() continue; } - if (!m_pendingInvocations.empty()) + // Invocations nest onto the running thread, so a callback storm stacks + // them faster than they retire -- DQ8's movie streaming reached depth 13 + // and exhausted the stack pool. Past this the rest simply stay queued. + constexpr size_t kMaxInvocationDepth = 4u; + if (!m_pendingInvocations.empty() && + running->invocations.size() < kMaxInvocationDepth) { GuestInvocation invocation = std::move(m_pendingInvocations.front()); m_pendingInvocations.pop_front(); @@ -285,41 +404,50 @@ void EeScheduler::run() } PS2Runtime::RecompiledFunction function = m_runtime.lookupFunction(context.pc); + g_eeGuestDispatchCount.fetch_add(1u, std::memory_order_relaxed); + // DQ8_EE_DISPATCH_HISTOGRAM=N reports the guest addresses the executor + // re-enters most often, every N dispatches. A spinning guest loop shows + // up here as one address with a huge count. + if (g_eeDispatchHistogramInterval != 0u) + { + static std::unordered_map s_histogram; + static uint64_t s_seen = 0u; + ++s_histogram[context.pc]; + if (++s_seen >= g_eeDispatchHistogramInterval) + { + std::vector> top(s_histogram.begin(), s_histogram.end()); + std::partial_sort(top.begin(), top.begin() + std::min(8u, top.size()), top.end(), + [](const auto &l, const auto &r) { return l.second > r.second; }); + std::fprintf(stderr, "[ee] dispatch histogram over %llu:", + static_cast(s_seen)); + for (size_t i = 0u; i < std::min(8u, top.size()); ++i) + { + std::fprintf(stderr, " 0x%08x=%llu", top[i].first, + static_cast(top[i].second)); + } + std::fprintf(stderr, "\n"); + s_histogram.clear(); + s_seen = 0u; + } + } if (checkpointDue(kGuestDispatchCycles)) { continue; } - try - { - m_insideInterrupt = !running->invocations.empty() && running->invocations.back().kind == GuestInvocationKind::Interrupt; - m_guestExecuting.store(true, std::memory_order_release); - function(m_rdram, &context, &m_runtime); - m_guestExecuting.store(false, std::memory_order_release); - m_insideInterrupt = false; - } - catch (const EeDispatcherTransfer &) + m_pendingFunction = function; + m_pendingContext = &context; + m_pendingInsideInterrupt = + !running->invocations.empty() && + running->invocations.back().kind == GuestInvocationKind::Interrupt; + enterGuest(*running); + if (m_fiberException) { - m_guestExecuting.store(false, std::memory_order_release); - m_insideInterrupt = false; - } - catch (...) - { - m_guestExecuting.store(false, std::memory_order_release); + std::exception_ptr escaped; + std::swap(escaped, m_fiberException); m_running.store(false, std::memory_order_release); - publishSnapshot(); - throw; - } - - processPendingEvents(); - if (m_rescheduleRequested && m_currentThreadId != 0) - { - GuestThread *preempted = currentThread(); - assert(preempted != nullptr); - enqueueReady(*preempted, !m_timeSliceExpired); - m_currentThreadId = 0; - m_rescheduleRequested = false; - m_timeSliceExpired = false; + publishSnapshotNow(); + std::rethrow_exception(escaped); } } @@ -347,6 +475,7 @@ void EeScheduler::postEvent(EeEvent event) { std::lock_guard lock(m_eventMutex); m_events.push_back(event); + m_pendingEventCount.fetch_add(1u, std::memory_order_release); m_checkpointPending.store(true, std::memory_order_release); } m_eventCv.notify_one(); @@ -388,7 +517,13 @@ bool EeScheduler::checkpointDue(uint32_t cycles) noexcept void EeScheduler::accountCycles(uint32_t cycles) noexcept { - const uint64_t elapsed = std::max(1u, cycles); + // DQ8_EE_CYCLE_SCALE multiplies the charge per guest safe point. The model + // charges per inter-function branch and ignores the instructions between, + // so it undercounts badly against a real 294 MHz EE; a spinning guest loop + // then gets far more iterations per emulated frame than hardware allows. + // Only raises the floor: a frame still cannot retire before its host + // deadline, so this can never run the game faster than real time. + const uint64_t elapsed = std::max(1u, cycles) * g_eeCycleScale; m_eeCycle += elapsed; m_pendingEeTimerInterrupts |= m_runtime.memory().advanceEeTimers(elapsed); if (m_pendingEeTimerInterrupts != 0u) @@ -414,9 +549,26 @@ void EeScheduler::setupCurrentThread(uint32_t stack, uint32_t stackSize, uint32_ target->stack = stack; target->stackSize = stackSize; target->gp = gp; + reserveGuestStackFromAsyncPool(stack); publishSnapshot(); } +// Invocation stacks are carved from the top of RAM, where the EE kernel also +// puts a game's initial stack. Keep the pool below any guest stack. +void EeScheduler::reserveGuestStackFromAsyncPool(uint32_t guestStackBase) +{ + if (guestStackBase == 0u) + { + return; + } + std::lock_guard lock(m_runtime.m_asyncCallbackStackMutex); + if (guestStackBase > m_runtime.m_asyncCallbackStackFloor && + guestStackBase < m_runtime.m_asyncCallbackStackTop) + { + m_runtime.m_asyncCallbackStackTop = guestStackBase; + } +} + int EeScheduler::createThread(const EeThreadCreateParams ¶ms) { assertExecutor(); @@ -442,7 +594,8 @@ int EeScheduler::createThread(const EeThreadCreateParams ¶ms) thread.initialPriority = params.priority; thread.currentPriority = params.priority; thread.status = EeThreadStatus::Dormant; - m_threads.emplace(id, std::move(thread)); + m_threadIndex[id] = &m_threads.emplace(id, std::move(thread)).first->second; + reserveGuestStackFromAsyncPool(params.stack); publishSnapshot(); return id; } @@ -468,6 +621,8 @@ int EeScheduler::deleteThread(int id, uint32_t &ownedStack) { ownedStack = it->second.stack; } + releaseInvocationStacks(id); + m_threadIndex[id] = nullptr; m_threads.erase(it); publishSnapshot(); return KE_OK; @@ -516,6 +671,9 @@ int EeScheduler::startThread(int id, uint32_t arg, const R5900Context &caller, b m_currentThreadId = 0; if (deleteThreadRecord && id != kMainThreadId) { + releaseInvocationStacks(id); + if (static_cast(id) <= kLastThreadId) + m_threadIndex[id] = nullptr; m_threads.erase(id); } if (ownedStack != 0u) @@ -523,6 +681,7 @@ int EeScheduler::startThread(int id, uint32_t arg, const R5900Context &caller, b m_runtime.guestFree(ownedStack); } publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -645,7 +804,7 @@ void EeScheduler::sleepCurrent() setReturnS32(&self->activeContext(), KE_OK); return; } - blockCurrent(EeWaitState{EeWaitReason::Sleep, std::monostate{}}); + blockCurrentResumable(EeWaitState{EeWaitReason::Sleep, std::monostate{}}); } int EeScheduler::wakeupThread(int id, bool interruptSafe) @@ -724,13 +883,10 @@ int EeScheduler::changePriority(int id, int priority, bool interruptSafe, int &o target->currentPriority = priority; if (target->status == EeThreadStatus::Running) { - for (int p = 0; p < target->currentPriority; ++p) + const int first = firstReadyPriority(); + if (first >= 0 && first < target->currentPriority) { - if (!m_readyQueues[p].empty()) - { - m_rescheduleRequested = true; - break; - } + m_rescheduleRequested = true; } } } @@ -754,6 +910,13 @@ int EeScheduler::rotateReadyQueue(int priority, bool interruptSafe) GuestThread *self = currentThread(); if (self && self->currentPriority == priority) { + // Rotating a singleton would immediately select this same thread. + // Guest checkpoints still service timers and newly posted events. + const int firstReady = firstReadyPriority(); + if ((firstReady < 0 || firstReady > priority) && !m_rescheduleRequested) + { + return KE_OK; + } enqueueReady(*self); m_currentThreadId = 0; m_rescheduleRequested = true; @@ -809,9 +972,23 @@ void EeScheduler::transferIfRequested(bool interruptSafe) enqueueReady(*self, true); m_currentThreadId = 0; } + m_rescheduleRequested = false; m_timeSliceExpired = false; publishSnapshot(); + + // The thread stays runnable, so its C++ stack is still wanted: suspend it + // rather than unwinding and rebuilding it. Only the executor's own stack + // has nowhere to suspend to (direct-syscall tests, embedders), and it still + // throws. + if (m_activeFiber != nullptr) + { + g_eeTransferSuspendCount.fetch_add(1u, std::memory_order_relaxed); + m_activeFiber->suspend(); + return; + } + + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -927,7 +1104,7 @@ void EeScheduler::waitSemaphore(int id) GuestThread *self = currentThread(); assert(self != nullptr); object->waiters.push_back(self->id); - blockCurrent(EeWaitState{EeWaitReason::Semaphore, EeSemaphoreWait{id}}); + blockCurrentResumable(EeWaitState{EeWaitReason::Semaphore, EeSemaphoreWait{id}}); } int EeScheduler::createEventFlag(uint32_t initialBits, uint32_t attr, uint32_t option) @@ -1050,8 +1227,8 @@ void EeScheduler::waitEventFlag(int id, uint32_t bits, uint32_t mode, uint32_t r return; } flag->waiters.push_back(self->id); - blockCurrent(EeWaitState{EeWaitReason::EventFlag, - EeEventFlagWait{id, bits, mode, resultAddress}}); + blockCurrentResumable(EeWaitState{EeWaitReason::EventFlag, + EeEventFlagWait{id, bits, mode, resultAddress}}); } int EeScheduler::setAlarm(uint16_t ticks, @@ -1073,7 +1250,7 @@ int EeScheduler::setAlarm(uint16_t ticks, m_alarms.emplace(id, EeAlarm{id, ticks, handler, argument, gp, sp}); const uint64_t tickCount = ticks == 0u ? 1u : static_cast(ticks); scheduleEvent(m_eeCycle + tickCount * kAlarmTickCycles, - std::chrono::steady_clock::now() + std::chrono::microseconds(tickCount * kAlarmTickMicroseconds), + ee_host_pacing::now() + std::chrono::microseconds(tickCount * kAlarmTickMicroseconds), EeEvent{EeEventType::Alarm, static_cast(id), 0}); return id; } @@ -1115,6 +1292,7 @@ void EeScheduler::queueInvocation(GuestInvocation invocation) invocation.sequence = ++m_invocationSequence; owner->invocations.push_back(std::move(invocation)); publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1134,6 +1312,7 @@ void EeScheduler::queueInvocation(GuestInvocation invocation) owner->invocations.push_back(std::move(*it)); } publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } @@ -1168,7 +1347,16 @@ uint32_t EeScheduler::invocationStackTop() return existing->second; } constexpr uint32_t kInvocationStackSize = 0x4000u; - const uint32_t top = m_runtime.reserveAsyncCallbackStack(kInvocationStackSize, 16u); + uint32_t top = 0u; + if (!m_freeInvocationStacks.empty()) + { + top = m_freeInvocationStacks.back(); + m_freeInvocationStacks.pop_back(); + } + else + { + top = m_runtime.reserveAsyncCallbackStack(kInvocationStackSize, 16u); + } if (top == 0u) { throw std::runtime_error("EE invocation stack space exhausted"); @@ -1177,6 +1365,26 @@ uint32_t EeScheduler::invocationStackTop() return top; } +// Stacks are keyed per (thread, depth) and the pool only bumps down, so without +// this every thread that ever took an invocation holds 16 KiB forever -- DQ8 +// starts one thread per movie and drained all 16 slots on the second one. +void EeScheduler::releaseInvocationStacks(int threadId) +{ + const uint64_t prefix = static_cast(static_cast(threadId)) << 32u; + for (auto it = m_invocationStackTops.begin(); it != m_invocationStackTops.end();) + { + if ((it->first & 0xFFFFFFFF00000000ull) == prefix) + { + m_freeInvocationStacks.push_back(it->second); + it = m_invocationStackTops.erase(it); + } + else + { + ++it; + } + } +} + int EeScheduler::addIrqHandler(bool dmac, uint32_t cause, uint32_t handler, @@ -1204,6 +1412,9 @@ int EeScheduler::addIrqHandler(bool dmac, sp, true, append ? ++tail : --head}); + const uint32_t pending = dmac ? m_pendingDmacMask : m_pendingIntcMask; + if (cause < 32u && (pending & (1u << cause)) != 0u) + dispatchIrq(dmac, cause); return id; } @@ -1227,6 +1438,10 @@ int EeScheduler::setIrqHandlerEnabled(bool dmac, int id, bool enabled) if (it != handlers.end()) { it->second.enabled = enabled; + const uint32_t cause = it->second.cause; + const uint32_t pending = dmac ? m_pendingDmacMask : m_pendingIntcMask; + if (enabled && cause < 32u && (pending & (1u << cause)) != 0u) + dispatchIrq(dmac, cause); } return KE_OK; } @@ -1240,6 +1455,9 @@ int EeScheduler::setIrqCauseEnabled(bool dmac, uint32_t cause, bool enabled) if (enabled) { mask |= 1u << cause; + const uint32_t pending = dmac ? m_pendingDmacMask : m_pendingIntcMask; + if ((pending & (1u << cause)) != 0u) + dispatchIrq(dmac, cause); } else { @@ -1252,6 +1470,10 @@ int EeScheduler::setIrqCauseEnabled(bool dmac, uint32_t cause, bool enabled) void EeScheduler::dispatchIrq(bool dmac, uint32_t cause) { assertExecutor(); + // Device completion remains pending while its handler or mask is disabled. + uint32_t &pending = dmac ? m_pendingDmacMask : m_pendingIntcMask; + const uint32_t causeBit = cause < 32u ? (1u << cause) : 0u; + pending |= causeBit; const uint32_t mask = dmac ? m_enabledDmacMask : m_enabledIntcMask; if (cause < 32u && (mask & (1u << cause)) == 0u) { @@ -1270,6 +1492,8 @@ void EeScheduler::dispatchIrq(bool dmac, uint32_t cause) } std::sort(matching.begin(), matching.end(), [](const EeIrqHandler &left, const EeIrqHandler &right) { return left.order < right.order; }); + if (!matching.empty()) + pending &= ~causeBit; for (const EeIrqHandler &handler : matching) { GuestInvocation invocation{}; @@ -1278,7 +1502,9 @@ void EeScheduler::dispatchIrq(bool dmac, uint32_t cause) SET_GPR_U32(&invocation.context, 4, cause); SET_GPR_U32(&invocation.context, 5, handler.argument); SET_GPR_U32(&invocation.context, 28, handler.gp); - SET_GPR_U32(&invocation.context, 29, handler.sp); + // Not handler.sp: that thread has moved on. $sp = 0 makes the + // dispatcher hand out an invocation stack, as the EE does. + SET_GPR_U32(&invocation.context, 29, 0u); SET_GPR_U32(&invocation.context, 31, 0u); queueInvocation(std::move(invocation)); } @@ -1303,7 +1529,9 @@ void EeScheduler::setVSyncFlag(uint32_t flagAddress, uint32_t tickAddress) uint64_t EeScheduler::currentVSyncTick() const noexcept { - return m_vsyncTick; + // The host presenter also observes this clock. m_vsyncTick belongs to the + // executor; VBlankStart publishes the same value through the GS atomic. + return m_runtime.memory().gs().vsyncTick.load(std::memory_order_acquire); } uint32_t EeScheduler::setGsVSyncCallback(uint32_t callback, uint32_t gp, uint32_t sp) @@ -1391,12 +1619,16 @@ void EeScheduler::completeExternalWait(uint32_t type, uint64_t token, int result GuestThread *EeScheduler::thread(int id) { + if (static_cast(id) <= kLastThreadId) + return m_threadIndex[id]; auto it = m_threads.find(id); return it == m_threads.end() ? nullptr : &it->second; } const GuestThread *EeScheduler::thread(int id) const { + if (static_cast(id) <= kLastThreadId) + return m_threadIndex[id]; auto it = m_threads.find(id); return it == m_threads.end() ? nullptr : &it->second; } @@ -1475,11 +1707,29 @@ void EeScheduler::bindMainContextForSyscall(R5900Context &ctx, uint8_t *rdram) EeKernelSnapshot EeScheduler::snapshot() const { + // Ask the executor to refresh, then return what it published last. One + // scheduler operation of staleness is invisible to a debug panel, and the + // idle path publishes unconditionally so a quiet scheduler still converges. + m_snapshotWanted.store(true, std::memory_order_relaxed); std::lock_guard lock(m_snapshotMutex); return m_snapshot; } void EeScheduler::publishSnapshot() +{ + // Building this walks every thread, semaphore and event flag and sorts all + // three. Called from ~40 scheduler operations it was 45% of EE thread time + // during a movie, for state whose only readers are the debug panel and an + // aggressive-log tick. + if (!m_snapshotWanted.load(std::memory_order_relaxed) || + !m_snapshotWanted.exchange(false, std::memory_order_relaxed)) + { + return; + } + publishSnapshotNow(); +} + +void EeScheduler::publishSnapshotNow() { EeKernelSnapshot next{}; next.sequence = ++m_snapshotSequence; @@ -1490,13 +1740,15 @@ void EeScheduler::publishSnapshot() next.threads.reserve(m_threads.size()); for (const auto &[id, item] : m_threads) { - if (id < 0) + if (id < 0 && item.status == EeThreadStatus::Dormant) { continue; } EeThreadSnapshot snapshot{}; snapshot.id = id; snapshot.pc = item.activeContext().pc; + snapshot.ra = getRegU32(&item.activeContext(), 31); + snapshot.sp = getRegU32(&item.activeContext(), 29); snapshot.entry = item.entry; snapshot.stack = item.stack; snapshot.stackSize = item.stackSize; @@ -1576,6 +1828,33 @@ GuestThread &EeScheduler::acquireInvocationThread() return m_threads.emplace(dispatcher.id, std::move(dispatcher)).first->second; } +void EeScheduler::refreshReadyMask(int priority) noexcept +{ + const size_t word = static_cast(priority) / 64u; + const uint64_t bit = uint64_t{1} << (static_cast(priority) % 64u); + if (m_readyQueues[static_cast(priority)].empty()) + { + m_readyMask[word] &= ~bit; + } + else + { + m_readyMask[word] |= bit; + } +} + +int EeScheduler::firstReadyPriority() const noexcept +{ + for (size_t word = 0; word < m_readyMask.size(); ++word) + { + if (m_readyMask[word] != 0u) + { + return static_cast(word * 64u + + static_cast(__builtin_ctzll(m_readyMask[word]))); + } + } + return -1; +} + void EeScheduler::enqueueReady(GuestThread &item, bool front) { assert(item.currentPriority >= 0 && item.currentPriority < kPriorityCount); @@ -1589,6 +1868,7 @@ void EeScheduler::enqueueReady(GuestThread &item, bool front) { queue.push_back(item.id); } + refreshReadyMask(item.currentPriority); } void EeScheduler::removeReady(GuestThread &item) @@ -1601,18 +1881,19 @@ void EeScheduler::removeReady(GuestThread &item) auto it = std::find(queue.begin(), queue.end(), item.id); assert(it != queue.end()); queue.erase(it); + refreshReadyMask(item.currentPriority); } GuestThread *EeScheduler::selectReady() { - for (auto &queue : m_readyQueues) + const int priority = firstReadyPriority(); + if (priority >= 0) { - if (queue.empty()) - { - continue; - } + auto &queue = m_readyQueues[static_cast(priority)]; + assert(!queue.empty()); const int id = queue.front(); queue.pop_front(); + refreshReadyMask(priority); GuestThread *selected = thread(id); assert(selected != nullptr); assert(selected->status == EeThreadStatus::Ready); @@ -1621,6 +1902,85 @@ GuestThread *EeScheduler::selectReady() return nullptr; } +void EeScheduler::fiberEntry(void *user) +{ + auto *self = static_cast(user); + for (;;) + { + self->runPendingGuestCall(); + // The dispatch is over; hand the executor back its stack. Resumed when + // this thread is scheduled again with a new pending call. + self->m_activeFiber->suspend(); + } +} + +// Runs on the fiber's stack. Nothing may escape it: an exception unwinding past +// the fiber entry has no frame to land on. +void EeScheduler::runPendingGuestCall() +{ + PS2Runtime::RecompiledFunction function = m_pendingFunction; + R5900Context *context = m_pendingContext; + m_pendingFunction = nullptr; + m_pendingContext = nullptr; + try + { + m_insideInterrupt = m_pendingInsideInterrupt; + m_guestExecuting.store(true, std::memory_order_release); + if (function != nullptr && context != nullptr) + { + function(m_rdram, context, &m_runtime); + } + } + catch (const EeDispatcherTransfer &) + { + // A block, an invocation push or a thread exit. The stack is unwound, + // so the next dispatch starts from the thread's saved pc again. + } + catch (...) + { + m_fiberException = std::current_exception(); + } + m_guestExecuting.store(false, std::memory_order_release); + m_insideInterrupt = false; + m_dispatchedThreadId = 0; +} + +void EeScheduler::enterGuest(GuestThread &thread) +{ + if (!thread.fiber) + { + // Sized for deeply nested guest call chains; mapped lazily, so the + // resident cost is only the pages a thread actually touches. + static const size_t stackBytes = [] { + const char *value = std::getenv("DQ8_EE_FIBER_STACK_KB"); + const long long parsed = value ? std::atoll(value) : 0; + const size_t kb = parsed > 0 ? static_cast(parsed) : 8192u; + return kb * 1024u; + }(); + thread.fiber = std::make_unique(); + if (!thread.fiber->create(&EeScheduler::fiberEntry, this, stackBytes)) + { + thread.fiber.reset(); + throw std::runtime_error("EE scheduler could not allocate a guest fiber stack"); + } + } + + assert(m_activeFiber == nullptr && "guest fibers must not nest"); + g_eeEnterGuestCount.fetch_add(1u, std::memory_order_relaxed); + EeFiber *previous = m_activeFiber; + m_activeFiber = thread.fiber.get(); + m_dispatchedThreadId = thread.id; + thread.inGuestCall = true; + m_activeFiber->resume(); + m_activeFiber = previous; + // Cleared by the fiber when a dispatch finishes; still set means it + // suspended part way through and its stack is waiting to be resumed. + if (m_dispatchedThreadId == 0) + { + thread.inGuestCall = false; + } +} + void EeScheduler::makeRunning(GuestThread &item) { assert(m_currentThreadId == 0); @@ -1680,9 +2040,33 @@ void EeScheduler::blockCurrent(EeWaitState wait) self->status = self->suspendCount == 0 ? EeThreadStatus::Waiting : EeThreadStatus::WaitingSuspended; m_currentThreadId = 0; publishSnapshot(); + g_eeTransferThrowCount.fetch_add(1u, std::memory_order_relaxed); throw EeDispatcherTransfer{}; } +// Blocks and comes back where it left off. Only for waits that carry no +// completion -- a completion re-invokes its syscall on wake, which would run +// twice if the stack were still there -- and whose caller has nothing left to +// do afterwards, so returning here returns into the guest. blockCurrent() +// stays [[noreturn]] for everyone else, including the [[noreturn]] waitVSync +// and waitExternal, whose completions are optional. +void EeScheduler::blockCurrentResumable(EeWaitState wait) +{ + assert(!wait.completion && "a resumable block must not carry a completion"); + if (m_activeFiber == nullptr) + { + blockCurrent(std::move(wait)); + } + GuestThread *self = currentThread(); + assert(self != nullptr); + self->wait = std::move(wait); + self->status = self->suspendCount == 0 ? EeThreadStatus::Waiting : EeThreadStatus::WaitingSuspended; + m_currentThreadId = 0; + publishSnapshot(); + g_eeTransferSuspendCount.fetch_add(1u, std::memory_order_relaxed); + m_activeFiber->suspend(); +} + void EeScheduler::makeReady(GuestThread &item, int result, bool interruptSafe) { auto completion = std::move(item.wait.completion); @@ -1735,6 +2119,7 @@ void EeScheduler::applyPendingPreemption() void EeScheduler::processPendingEvents() { assertExecutor(); + g_eeProcessPendingEventsCount.fetch_add(1u, std::memory_order_relaxed); processDueDeadlines(); const uint32_t timerInterrupts = m_pendingEeTimerInterrupts; m_pendingEeTimerInterrupts = 0u; @@ -1745,35 +2130,69 @@ void EeScheduler::processPendingEvents() dispatchIrq(false, 9u + timer); } } - std::deque pending; + // INTC VIF0 (4) and VIF1 (5), raised by a VIFcode carrying the i bit. + const uint32_t vifInterrupts = m_runtime.memory().takePendingVifInterrupts(); + if ((vifInterrupts & 0x1u) != 0u) { - std::lock_guard lock(m_eventMutex); - pending.swap(m_events); + dispatchIrq(false, 4u); } - for (const EeEvent &event : pending) + if ((vifInterrupts & 0x2u) != 0u) { - processEvent(event); + dispatchIrq(false, 5u); } + // A default-constructed deque still allocates, and this path is almost + // always empty, so only build one when there is something to take. + if (m_pendingEventCount.load(std::memory_order_acquire) != 0u) { - std::lock_guard lock(m_eventMutex); - const uint64_t nextEventCycle = m_nextDeadlineCycle.load(std::memory_order_acquire); - const bool cycleEventDue = nextEventCycle != 0u && m_eeCycle >= nextEventCycle; - const bool pendingWork = !m_events.empty() || cycleEventDue || m_stopRequested.load(std::memory_order_acquire); - m_checkpointPending.store(pendingWork, std::memory_order_release); + std::unique_lock lock(m_eventMutex); + if (!m_events.empty()) + { + std::deque pending; + pending.swap(m_events); + // Same mutex postEvent counts under, so a post that races this + // either lands in the deque being drained or after this store. + m_pendingEventCount.store(0u, std::memory_order_release); + lock.unlock(); + for (const EeEvent &event : pending) + { + processEvent(event); + } + } } + + // The tail used to take m_eventMutex to ask m_events.empty(); the movie + // thread pumps twice per yield at ~4M yields a second and the mutex pair + // showed in its profile. The count plus the atomics answer the same + // question. + const uint64_t nextEventCycle = m_nextDeadlineCycle.load(std::memory_order_acquire); + const bool cycleEventDue = nextEventCycle != 0u && m_eeCycle >= nextEventCycle; + const bool pendingWork = m_pendingEventCount.load(std::memory_order_acquire) != 0u || + cycleEventDue || + m_stopRequested.load(std::memory_order_acquire); + m_checkpointPending.store(pendingWork, std::memory_order_release); applyPendingPreemption(); } void EeScheduler::processDueDeadlines() { + // Runs after every guest dispatch, so the common "nothing is due" case must + // not cost a mutex, a clock read and a walk of m_deadlines. m_nextDeadlineCycle + // is the minimum deadlineCycle, or 0 when there are none -- exactly the + // condition the loop below would fail on. + const uint64_t nextDeadline = m_nextDeadlineCycle.load(std::memory_order_acquire); + if (nextDeadline == 0u || m_eeCycle < nextDeadline) + { + return; + } + for (;;) { std::vector due; std::chrono::steady_clock::time_point pacingDeadline{}; { std::unique_lock lock(m_eventMutex); - const auto now = std::chrono::steady_clock::now(); + const auto now = ee_host_pacing::now(); for (const ScheduledEvent &item : m_deadlines) { if (item.deadlineCycle <= m_eeCycle && @@ -1792,7 +2211,7 @@ void EeScheduler::processDueDeadlines() if (now < pacingDeadline) { - m_eventCv.wait_until(lock, pacingDeadline, [this]() + ee_host_pacing::waitUntil(m_eventCv, lock, pacingDeadline, [this]() { return !m_events.empty() || m_stopRequested.load(std::memory_order_acquire); }); if (!m_events.empty() || m_stopRequested.load(std::memory_order_acquire)) @@ -1802,7 +2221,7 @@ void EeScheduler::processDueDeadlines() } } - const auto pacedNow = std::chrono::steady_clock::now(); + const auto pacedNow = ee_host_pacing::now(); auto firstFuture = std::partition(m_deadlines.begin(), m_deadlines.end(), [this, pacedNow](const ScheduledEvent &item) { return item.deadlineCycle <= m_eeCycle && @@ -1839,11 +2258,16 @@ void EeScheduler::processDueDeadlines() { if (scheduled.event.type == EeEventType::VBlankStart) { + // Preserve the guest event stream while bounding old host debt. + // Both children belong to this same host VBlank boundary. + const auto now = ee_host_pacing::now(); + const auto boundary = ee_host_pacing::vblankBoundary(scheduled.hostDeadline, now, kVBlankPeriod); + traceVBlankPacing(m_vsyncTick + 1u, scheduled.deadlineCycle, now, boundary); scheduleEvent(scheduled.deadlineCycle + kVBlankDurationCycles, - scheduled.hostDeadline + kVBlankDuration, + boundary.host + kVBlankDuration, EeEvent{EeEventType::VBlankEnd, 0, m_vsyncTick + 1u}); scheduleEvent(scheduled.deadlineCycle + kVBlankPeriodCycles, - scheduled.hostDeadline + kVBlankPeriod, + boundary.host + kVBlankPeriod, EeEvent{EeEventType::VBlankStart, 0, 0}); } processEvent(scheduled.event); @@ -1888,7 +2312,8 @@ void EeScheduler::processEvent(const EeEvent &event) invocation.context.pc = m_gsVSyncCallback; SET_GPR_U32(&invocation.context, 4, static_cast(m_vsyncTick)); SET_GPR_U32(&invocation.context, 28, m_gsVSyncCallbackGp); - SET_GPR_U32(&invocation.context, 29, m_gsVSyncCallbackSp); + // See dispatchIrq. + SET_GPR_U32(&invocation.context, 29, 0u); SET_GPR_U32(&invocation.context, 31, 0u); queueInvocation(std::move(invocation)); } @@ -1918,7 +2343,8 @@ void EeScheduler::processEvent(const EeEvent &event) SET_GPR_U32(&invocation.context, 5, static_cast(alarm.ticks)); SET_GPR_U32(&invocation.context, 6, alarm.argument); SET_GPR_U32(&invocation.context, 28, alarm.gp); - SET_GPR_U32(&invocation.context, 29, alarm.sp); + // See dispatchIrq. + SET_GPR_U32(&invocation.context, 29, 0u); SET_GPR_U32(&invocation.context, 31, 0u); queueInvocation(std::move(invocation)); break; @@ -2020,7 +2446,7 @@ void EeScheduler::waitForEvent() } if (hasTimerDeadline) { - const auto timerHostDeadline = std::chrono::steady_clock::now() + eeCyclesToHostDuration(timerCycles); + const auto timerHostDeadline = ee_host_pacing::now() + eeCyclesToHostDuration(timerCycles); if (timerHostDeadline < hostDeadline) { deadlineCycle = m_eeCycle + timerCycles; @@ -2028,7 +2454,7 @@ void EeScheduler::waitForEvent() } } - const bool signaled = m_eventCv.wait_until(lock, hostDeadline, [this]() + const bool signaled = ee_host_pacing::waitUntil(m_eventCv, lock, hostDeadline, [this]() { return !m_events.empty() || m_stopRequested.load(std::memory_order_acquire); }); if (!signaled) @@ -2080,14 +2506,8 @@ void EeScheduler::updateNextDeadline() bool EeScheduler::hasReadyAtOrAbovePriority(int priority) const { const int last = std::clamp(priority, 0, kPriorityCount - 1); - for (int p = 0; p <= last; ++p) - { - if (!m_readyQueues[static_cast(p)].empty()) - { - return true; - } - } - return false; + const int first = firstReadyPriority(); + return first >= 0 && first <= last; } void EeScheduler::renewTimeSlice() diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/Helpers/Support.h b/ps2xRuntime/src/lib/Kernel/Stubs/Helpers/Support.h index a7fae20ca..583cc818d 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/Helpers/Support.h +++ b/ps2xRuntime/src/lib/Kernel/Stubs/Helpers/Support.h @@ -1360,7 +1360,9 @@ namespace uint32_t madr = 0; uint32_t qwc = 0; uint32_t tadr = payloadPhys; - uint32_t chcr = 0x00000181u; // DIR=1, TIE=1, STR=1 (normal mode). + PS2Memory &mem = runtime->memory(); + // The SDK changes MODE, DIR and STR while preserving TTE and TIE. + uint32_t chcr = (mem.readIORegister(channelBase) & ~0xCu) | 0x101u; if (preferNormalCount) { @@ -1369,10 +1371,9 @@ namespace } else { - chcr = 0x00000185u; // MODE=1 chain, DIR=1, TIE=1, STR=1. + chcr |= 0x4u; // MODE=1 chain. } - PS2Memory &mem = runtime->memory(); mem.writeIORegister(channelBase + 0x20u, qwc & 0xFFFFu); mem.writeIORegister(channelBase + 0x10u, madr); mem.writeIORegister(channelBase + 0x30u, tadr); diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp index 44eaef4d9..77b3ff15e 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MPEG.cpp @@ -16,11 +16,51 @@ extern "C" } #endif +#include +#include #include #include #include "Syscalls/Helpers/State.h" +// Debug accounting for the movie path, read by whichever backend reports stats. +// A movie frame arrives as ~900 tile transfers, but those turned out to cost +// almost nothing; these say what the rest of the frame is doing. +std::atomic g_mpegGetPictureNanos{0}; +std::atomic g_mpegGetPictureCount{0}; +std::atomic g_mpegWriteFrameNanos{0}; +std::atomic g_mpegWriteFrameCount{0}; +std::atomic g_mpegDemuxNanos{0}; +std::atomic g_mpegDemuxCount{0}; +// How many demux calls were refused, and which guest threads drive each side: +// whether the producer and the consumer are the same thread decides whether +// parking the producer is even possible. +std::atomic g_mpegDemuxRefusedCount{0}; +std::atomic g_mpegPendingEsPeakBytes{0}; +std::atomic g_mpegDemuxThreadId{0}; +std::atomic g_mpegGetPictureThreadId{0}; + +namespace +{ + struct MpegScopedTimer + { + std::atomic &sink; + std::atomic &counter; + std::chrono::steady_clock::time_point start; + MpegScopedTimer(std::atomic &nanos, std::atomic &count) + : sink(nanos), counter(count), start(std::chrono::steady_clock::now()) {} + ~MpegScopedTimer() + { + sink.fetch_add(static_cast( + std::chrono::duration_cast( + std::chrono::steady_clock::now() - start) + .count()), + std::memory_order_relaxed); + counter.fetch_add(1, std::memory_order_relaxed); + } + }; +} + namespace ps2_stubs { namespace @@ -61,6 +101,8 @@ namespace ps2_stubs class MpegFfmpegDecoder { public: + static constexpr bool kAvailable = true; + MpegFfmpegDecoder() = default; ~MpegFfmpegDecoder() @@ -448,6 +490,8 @@ namespace ps2_stubs class MpegFfmpegDecoder { public: + static constexpr bool kAvailable = false; + bool feed(const uint8_t *, size_t, std::deque &, int64_t = -1, int64_t = -1) { static bool s_warnedNoFfmpeg = false; @@ -483,6 +527,32 @@ namespace ps2_stubs constexpr uint64_t kDefaultPictureIntervalQ32 = 2ull * kPictureClockOne; constexpr size_t kMpegTimingScanLimit = 4096u; constexpr size_t kMaxDecodedPicturesAhead = 8u; + // Games size their audio buffers for a demux the IPU pulls along in + // real time. DQ8 holds a quarter second and drops what does not fit. + constexpr int64_t kMaxAudioLead90k = 90000 * 3 / 20; + + // Lookahead is held as compressed elementary stream, not as decoded + // pictures: a 512x448 frame costs ~917 KB decoded against ~25 KB on the + // wire, so the same memory buys ~36x more of it. Refusing a demux call + // is what makes the guest's producer loop spin, and every spin costs a + // guest RotateThreadReadyQueue, so the lookahead wants to be deep. + size_t mpegMaxPendingEsBytes() + { + static const size_t bytes = [] { + const char *value = std::getenv("DQ8_MPEG_ES_BUFFER_MB"); + const long parsed = value ? std::atol(value) : 0; + return (parsed > 0 ? static_cast(parsed) : 32u) * 1024u * 1024u; + }(); + return bytes; + } + + struct MpegPendingEs + { + std::vector data; + int64_t pts90k = -1; + int64_t dts90k = -1; + }; + struct MpegPlaybackState { @@ -501,6 +571,11 @@ namespace ps2_stubs std::vector pssBuffer; std::vector pssGuestAddrs; std::deque decodedFrames; + // Demuxed but not yet decoded. Decode is pulled from here by + // sceMpegGetPicture rather than pushed by the demux call. + std::deque pendingEs; + size_t pendingEsBytes = 0u; + size_t pendingEsPeakBytes = 0u; std::unique_ptr decoder; uint8_t frameRateCode = 0u; uint8_t frameRateExtensionN = 0u; @@ -512,6 +587,11 @@ namespace ps2_stubs uint64_t presentationEndTickQ32 = std::numeric_limits::max(); int64_t firstPresentedPts90k = -1; uint64_t ptsPresentationBaseTickQ32 = 0u; + // How far audio has been handed to the game, against how much of it + // has played; see mpegDemuxBackpressured. + int64_t firstAudioPts90k = -1; + int64_t lastAudioPts90k = -1; + std::chrono::steady_clock::time_point firstAudioTime{}; }; struct MpegStreamCallbackEvent @@ -546,6 +626,44 @@ namespace ps2_stubs std::mutex g_mpeg_stub_mutex; constexpr uint32_t kMpegPictureWaitType = 1u; + + // DQ8_SKIP_MOVIES=1 serves pictures without waiting for their + // presentation tick, so a movie runs as fast as the guest will drive it + // rather than being held to 30fps. Everything else about playback is + // untouched: the frames are still decoded and written, and the movie + // still ends by itself. + // + // Two stronger skips were tried and are recorded because both looked + // right. Reporting end-of-stream from sceMpegGetPicture/sceMpegIsEnd + // leaves the streaming thread spinning on a black screen -- the guest + // state machine never learns the movie is over. Closing the caller's + // upload gate to drop the ~900 VRAM tiles a movie frame costs does cut + // them (54,000 transfers per 60 frames to 240) but the guest then DMAs + // a buffer nothing initialises, and the screen fills with garbage. + // DQ8's attract movies are not button-skippable either. + bool mpegSkipMovies() + { + static const bool skip = [] { + const char *value = std::getenv("DQ8_SKIP_MOVIES"); + return value != nullptr && *value != '\0' && *value != '0'; + }(); + return skip; + } + + // A refused demux call only has to give the consumer thread a turn -- + // it does not need one context transfer per refusal, and a transfer is + // a thrown EeDispatcherTransfer unwound through the guest's whole call + // stack. Yield on every Nth refusal instead; the spins in between are + // just a stub entry and a mutex. + size_t mpegDemuxYieldInterval() + { + static const size_t interval = [] { + const char *value = std::getenv("DQ8_MPEG_YIELD_EVERY"); + const long parsed = value ? std::atol(value) : 0; + return parsed > 0 ? static_cast(parsed) : 64u; + }(); + return interval; + } MpegStubState g_mpeg_stub_state; // TODO this resolution should follow runtime resolution @@ -1006,7 +1124,10 @@ namespace ps2_stubs void flushDecoderIfEnded(MpegPlaybackState &playback) { - if (playback.streamEnded && playback.decoder) + // Not while bytes are still queued. A program-end code arrives with + // the whole movie still buffered, and draining the decoder there + // makes every packet after it fail with AVERROR_EOF. + if (playback.streamEnded && playback.pendingEs.empty() && playback.decoder) { playback.decoder->flush(playback.decodedFrames); } @@ -1038,6 +1159,16 @@ namespace ps2_stubs } playback.sawInput = true; + + // Without a video decoder no frame can ever arrive, so parking + // sceMpegGetPicture waiters would hang the movie forever. Mark the + // stream failed instead and let them return empty-handed. + if constexpr (!MpegFfmpegDecoder::kAvailable) + { + playback.decoderFailed = true; + return; + } + updateMpegPictureTiming(playback, data, size); if (playback.waitingForVideoSequenceHeader) { @@ -1110,6 +1241,48 @@ namespace ps2_stubs flushDecoderIfEnded(playback); } + // Demux hands video payload here instead of straight to the decoder, so + // accepting the guest's bytes costs a memcpy rather than a decode. + void queueElementaryStream(MpegPlaybackState &playback, const uint8_t *data, size_t size, + int64_t pts90k, int64_t dts90k) + { + if (!data || size == 0u) + { + return; + } + playback.sawInput = true; + playback.pendingEs.push_back( + MpegPendingEs{std::vector(data, data + size), pts90k, dts90k}); + playback.pendingEsBytes += size; + playback.pendingEsPeakBytes = std::max(playback.pendingEsPeakBytes, playback.pendingEsBytes); + if (playback.pendingEsPeakBytes > g_mpegPendingEsPeakBytes.load(std::memory_order_relaxed)) + { + g_mpegPendingEsPeakBytes.store(playback.pendingEsPeakBytes, std::memory_order_relaxed); + } + } + + // Decode forward until there is something to show, or the buffer runs + // dry. One chunk is a single PES payload, so several are usually needed + // before the decoder emits a picture. + void decodePendingElementaryStream(MpegPlaybackState &playback, bool drainAll = false) + { + while (!playback.pendingEs.empty() && + (drainAll || playback.decodedFrames.empty())) + { + MpegPendingEs chunk = std::move(playback.pendingEs.front()); + playback.pendingEs.pop_front(); + playback.pendingEsBytes -= std::min(playback.pendingEsBytes, chunk.data.size()); + feedElementaryStream(playback, chunk.data.data(), chunk.data.size(), + chunk.pts90k, chunk.dts90k); + if (playback.decoderFailed) + { + break; + } + } + // The queue emptying is what makes a already-ended stream flushable. + flushDecoderIfEnded(playback); + } + void erasePssPrefix(MpegPlaybackState &playback, size_t count) { std::vector &buffer = playback.pssBuffer; @@ -1331,7 +1504,7 @@ namespace ps2_stubs pes.pts90k, pes.dts90k); } - feedElementaryStream( + queueElementaryStream( playback, buffer.data() + payloadStart, packetEnd - payloadStart, @@ -1343,6 +1516,15 @@ namespace ps2_stubs { const MpegPesHeader pes = parsePesHeader(buffer.data(), packetEnd); const size_t payloadStart = pes.payloadOffset; + if (pes.pts90k >= 0) + { + if (playback.firstAudioPts90k < 0) + { + playback.firstAudioPts90k = pes.pts90k; + playback.firstAudioTime = std::chrono::steady_clock::now(); + } + playback.lastAudioPts90k = pes.pts90k; + } if (payloadStart < packetEnd && payloadStart < playback.pssGuestAddrs.size()) { queueStreamCallbackEvent( @@ -1372,6 +1554,9 @@ namespace ps2_stubs { std::vector ignoredCallbacks; processPssBuffer(mpegAddr, playback, ignoredCallbacks, true); + // Everything still buffered has to reach the decoder before the + // stream can be called ended, or the tail of the movie is dropped. + decodePendingElementaryStream(playback, true); playback.streamEnded = true; playback.cdStreamGeneration = g_mpeg_stub_state.cdStreamGeneration; flushDecoderIfEnded(playback); @@ -1409,8 +1594,24 @@ namespace ps2_stubs // lets the game's producer loop wake the consumer again. Backpressure // still propagates naturally to sceCdStRead because the ring does not // advance while this is true. + // Bounded by buffered bytes now, not by decoded pictures: decode is + // pulled by the consumer, so pictures no longer pile up ahead of it + // and the old limit would fire on a queue that is nearly always + // empty. + // Audio goes to the game as it is demuxed, so stay a short lead ahead + // of what has played, by real time rather than by the pictures, + // which fall behind whenever decoding does. + bool audioAhead = false; + if (playback.firstAudioPts90k >= 0) + { + const auto elapsed = std::chrono::steady_clock::now() - playback.firstAudioTime; + const int64_t played = playback.firstAudioPts90k + + std::chrono::duration_cast(elapsed).count() * 9 / 100; + audioAhead = mpegPtsDelta90k(played, playback.lastAudioPts90k) > kMaxAudioLead90k; + } return !g_mpeg_stub_state.currentCdStreamEofSeen && - playback.decodedFrames.size() >= kMaxDecodedPicturesAhead; + (audioAhead || playback.pendingEsBytes >= mpegMaxPendingEsBytes() || + playback.decodedFrames.size() >= kMaxDecodedPicturesAhead); } void recordCdStreamBytesDemuxedUnlocked( @@ -1579,26 +1780,28 @@ namespace ps2_stubs return true; } - void dispatchGuestStreamCallback(uint8_t *rdram, - R5900Context *callerCtx, - PS2Runtime *runtime, - const MpegStreamCallbackEvent &event, - const MpegRegisteredCallback &callback) + bool makeGuestStreamCallback(uint8_t *rdram, + R5900Context *callerCtx, + PS2Runtime *runtime, + const MpegStreamCallbackEvent &event, + const MpegRegisteredCallback &callback, + uint32_t stackTop, + GuestInvocation &invocation) { if (!rdram || !callerCtx || !runtime || callback.func == 0u || !runtime->hasFunction(callback.func)) { - return; + return false; } const uint32_t cbDataAddr = runtime->guestMalloc(kMpegCallbackDataSize, 16u); if (cbDataAddr == 0u) { - return; + return false; } if (!writeMpegCallbackData(rdram, cbDataAddr, event)) { runtime->guestFree(cbDataAddr); - return; + return false; } R5900Context callbackCtx = *callerCtx; @@ -1606,20 +1809,22 @@ namespace ps2_stubs SET_GPR_U32(&callbackCtx, 5, cbDataAddr); SET_GPR_U32(&callbackCtx, 6, callback.data); SET_GPR_U32(&callbackCtx, 7, 0u); - SET_GPR_U32(&callbackCtx, 29, 0u); + SET_GPR_U32(&callbackCtx, 29, stackTop); SET_GPR_U32(&callbackCtx, 31, 0u); callbackCtx.pc = callback.func; - GuestInvocation invocation{}; invocation.kind = GuestInvocationKind::RpcCallback; invocation.context = callbackCtx; invocation.onComplete = [runtime, cbDataAddr](const R5900Context &, R5900Context &) { runtime->guestFree(cbDataAddr); }; - runtime->eeScheduler().queueInvocation(std::move(invocation)); + return true; } + // libmpeg calls these inside the demux call, in stream order, while the + // ring still holds their data. Queued, they ran after it was released, + // and last to first, which shuffled movie audio. void dispatchStreamCallbacks(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime, @@ -1630,11 +1835,37 @@ namespace ps2_stubs return; } + // One after another, so they can share one stack. A caller outside + // guest code (a test) gets them queued instead. + EeScheduler &scheduler = runtime->eeScheduler(); + const GuestThread *thread = scheduler.currentThread(); + const bool inGuest = thread && thread->inGuestCall; + const uint32_t stackTop = inGuest ? scheduler.invocationStackTop() : 0u; + std::vector invocations; for (const MpegStreamCallbackEvent &event : events) { for (const MpegRegisteredCallback &callback : event.callbacks) { - dispatchGuestStreamCallback(rdram, ctx, runtime, event, callback); + GuestInvocation invocation{}; + if (makeGuestStreamCallback(rdram, ctx, runtime, event, callback, stackTop, invocation)) + { + invocations.push_back(std::move(invocation)); + } + } + } + if (invocations.empty()) + { + return; + } + if (inGuest) + { + scheduler.invokeCurrentSequence(std::move(invocations)); + } + else + { + for (GuestInvocation &invocation : invocations) + { + scheduler.queueInvocation(std::move(invocation)); } } } @@ -1692,6 +1923,7 @@ namespace ps2_stubs { return; } + MpegScopedTimer timer(g_mpegWriteFrameNanos, g_mpegWriteFrameCount); const uint32_t width = static_cast(frame.width); const uint32_t height = static_cast(frame.height); @@ -2059,6 +2291,11 @@ namespace ps2_stubs setDynamicRet = *reinterpret_cast(p); mpegGuestWrite32(rdram, puVar4 + 12, setDynamicRet); + // The real sceMpegCreate ends with sceMpegReset, which clears the end flag + // the game polls. Without it a decoder made in reused memory can read as + // ended before its first picture. + sceMpegReset(rdram, ctx, runtime); + setReturnU32(ctx, setDynamicRet); } @@ -2090,6 +2327,7 @@ namespace ps2_stubs uint32_t traceIdx = 0u; bool eofChanged = false; bool backpressured = false; + bool decoderFailed = false; { std::lock_guard lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); @@ -2099,24 +2337,36 @@ namespace ps2_stubs { consumed = appendGuestBytes(mpegAddr, playback, rdram, dataAddr, byteCount, callbackEvents); recordCdStreamBytesDemuxedUnlocked(consumed, completedMpegIds, eofChanged); + // Enough to wake a consumer parked in sceMpegGetPicture; it + // stops as soon as one picture exists, so the decode rate still + // follows consumption. + decodePendingElementaryStream(playback); } decodedCount = playback.decodedFrames.size(); + decoderFailed = playback.decoderFailed; traceIdx = g_mpeg_stub_state.demuxPssTraceCount++; } if (backpressured) { + const uint64_t refusals = g_mpegDemuxRefusedCount.fetch_add(1u, std::memory_order_relaxed); if (traceIdx < 32u) { PS2_IF_AGRESSIVE_LOGS({ std::cerr << "[MPEG:DemuxPss:BACKPRESSURE] mpeg=0x" << std::hex << mpegAddr << std::dec << " decoded=" << decodedCount << std::endl; }); } + // Same reasoning and the same order as the ring variant. setReturnS32(ctx, 0); + if ((refusals % mpegDemuxYieldInterval()) == 0u) + { + runtime->eeScheduler().rotateReadyQueue(0, false); + runtime->eeScheduler().transferIfRequested(false); + } return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); - if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted) + if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted || decoderFailed) { runtime->eeScheduler().completeExternalWait(kMpegPictureWaitType, mpegAddr, KE_OK); } @@ -2141,12 +2391,17 @@ namespace ps2_stubs }); } - dispatchStreamCallbacksUnlocked(rdram, ctx, runtime, callbackEvents); + // The callbacks run before the caller resumes, so set its result first. setReturnS32(ctx, static_cast(consumed)); + dispatchStreamCallbacksUnlocked(rdram, ctx, runtime, callbackEvents); } void sceMpegDemuxPssRing(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { + MpegScopedTimer timer(g_mpegDemuxNanos, g_mpegDemuxCount); + g_mpegDemuxThreadId.store( + static_cast(runtime->eeScheduler().currentThreadId()), + std::memory_order_relaxed); static std::atomic s_demuxRingEntryCount{0u}; const uint32_t entryIdx = s_demuxRingEntryCount.fetch_add(1u, std::memory_order_relaxed); if (entryIdx < 4u) @@ -2173,6 +2428,7 @@ namespace ps2_stubs uint32_t traceIdx = 0u; bool eofChanged = false; bool backpressured = false; + bool decoderFailed = false; { std::lock_guard lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); @@ -2190,13 +2446,18 @@ namespace ps2_stubs ringSize, callbackEvents); recordCdStreamBytesDemuxedUnlocked(consumed, completedMpegIds, eofChanged); + // Same reason as the non-ring variant: keep a parked + // sceMpegGetPicture wakeable without decoding ahead. + decodePendingElementaryStream(playback); } decodedCount = playback.decodedFrames.size(); + decoderFailed = playback.decoderFailed; traceIdx = g_mpeg_stub_state.demuxRingTraceCount++; } if (backpressured) { + const uint64_t refusals = g_mpegDemuxRefusedCount.fetch_add(1u, std::memory_order_relaxed); if (traceIdx < 32u) { PS2_IF_AGRESSIVE_LOGS({ @@ -2205,11 +2466,32 @@ namespace ps2_stubs << " avail=" << availableBytes << std::endl; }); } + // Returning 0 consumed tells the producer "not now", and its loop + // asks again immediately -- DQ8 called this 22,000 times per + // presented frame, which is where the movie's frame time went. + // + // Yield, do not park. Parking the producer on a space-available + // wait and waking it from sceMpegGetPicture was tried and + // deadlocks the movie: DQ8 served 4 pictures instead of 658. That + // is exactly what the comment on mpegDemuxBackpressured predicts, + // so it is recorded here rather than left to be rediscovered. + // + // The return value has to be set before the transfer, and the + // transfer is not optional: rotateReadyQueue clears the current + // thread and asks for a reschedule, so returning normally after it + // leaves the scheduler with no running thread and trips + // bindMainContextForSyscall's assert on the next syscall. Same + // order as the RotateThreadReadyQueue syscall. setReturnS32(ctx, 0); + if ((refusals % mpegDemuxYieldInterval()) == 0u) + { + runtime->eeScheduler().rotateReadyQueue(0, false); + runtime->eeScheduler().transferIfRequested(false); + } return; } const bool currentStreamCompleted = std::find(completedMpegIds.begin(), completedMpegIds.end(), mpegAddr) != completedMpegIds.end(); - if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted) + if (decodedCount != decodedBefore || eofChanged || currentStreamCompleted || decoderFailed) { runtime->eeScheduler().completeExternalWait(kMpegPictureWaitType, mpegAddr, KE_OK); } @@ -2235,8 +2517,8 @@ namespace ps2_stubs }); } - dispatchStreamCallbacksUnlocked(rdram, ctx, runtime, callbackEvents); setReturnS32(ctx, static_cast(consumed)); + dispatchStreamCallbacksUnlocked(rdram, ctx, runtime, callbackEvents); } void sceMpegDispCenterOffX(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) @@ -2282,19 +2564,31 @@ namespace ps2_stubs void sceMpegGetPicture(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { + MpegScopedTimer timer(g_mpegGetPictureNanos, g_mpegGetPictureCount); + g_mpegGetPictureThreadId.store( + static_cast(runtime->eeScheduler().currentThreadId()), + std::memory_order_relaxed); const uint32_t mpegAddr = getRegU32(ctx, 4); const uint32_t imageAddr = getRegU32(ctx, 5); uint32_t width = kStubMovieWidth; uint32_t height = kStubMovieHeight; uint32_t frameCount = 0u; bool haveFrame = false; + bool movieEnded = false; MpegDecodedFrame frame; { std::unique_lock lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(mpegAddr); + { + // The consumer drives decode: demux only buffers. + decodePendingElementaryStream(playback); + // sawSequenceEnd is the only end-of-video signal a game streaming + // with plain sceCdRead can raise: the CD-stream EOF path belongs to + // sceCdSt*, and a PSS need not carry a program-end code. if (playback.decodedFrames.empty() && !g_mpeg_stub_state.currentCdStreamEofSeen && !playback.streamEnded && + !playback.sawSequenceEnd && !playback.decoderFailed) { if (g_mpeg_stub_state.getPictureWaitTraceCount < 32u) @@ -2341,7 +2635,7 @@ namespace ps2_stubs playback.nextPictureTickQ32 = currentTickQ32; } - if (currentTickQ32 < presentationTargetQ32) + if (currentTickQ32 < presentationTargetQ32 && !mpegSkipMovies()) { const uint64_t eligibleTick = (presentationTargetQ32 + kPictureClockOne - 1u) >> 32u; lock.unlock(); @@ -2387,17 +2681,31 @@ namespace ps2_stubs height = playback.height; frameCount = playback.picturesServed; } + + // Not on the call that serves the final frame -- the caller would + // tear the movie down before uploading it. + movieEnded = !haveFrame && playback.decodedFrames.empty() && playback.pendingEs.empty() && + (playback.sawSequenceEnd || playback.streamEnded || + playback.decoderFailed || g_mpeg_stub_state.currentCdStreamEofSeen); + } } mpegGuestWrite32(rdram, mpegAddr + 0x00u, width); mpegGuestWrite32(rdram, mpegAddr + 0x04u, height); - mpegGuestWrite32(rdram, mpegAddr + 0x08u, frameCount); + // +0x08 gates the caller's VRAM upload: DQ8 uploads its movie buffers + // only while this reads zero, so a running picture counter here stops + // the movie updating after the very first frame. + mpegGuestWrite32(rdram, mpegAddr + 0x08u, haveFrame ? 0u : 1u); if (uint8_t *base = getMemPtr(rdram, mpegAddr)) { const uint32_t iVar1 = *reinterpret_cast(base + 0x40); if (uint8_t *inner = getMemPtr(rdram, iVar1)) { + // inner[0x00] is the end-of-stream flag the game polls; its + // sceMpegIsEnd equivalent is just `return **(mpeg+0x40)`. + // Without this the movie plays out and never terminates. + *reinterpret_cast(inner + 0x00) = movieEnded ? 1u : 0u; *reinterpret_cast(inner + 0xb0) = 1; *reinterpret_cast(inner + 0xd8) = (getRegU32(ctx, 5) & 0x0FFFFFFFu) | 0x20000000u; *reinterpret_cast(inner + 0xe4) = getRegU32(ctx, 6); @@ -2486,7 +2794,10 @@ namespace ps2_stubs ++g_mpeg_stub_state.isEndTraceCount; } - setReturnS32(ctx, (ended && playback.decodedFrames.empty() && presentationComplete) ? 1 : 0); + setReturnS32(ctx, (ended && playback.decodedFrames.empty() && playback.pendingEs.empty() && + presentationComplete) + ? 1 + : 0); } void sceMpegIsRefBuffEmpty(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) @@ -2507,7 +2818,12 @@ namespace ps2_stubs std::lock_guard lock(g_mpeg_stub_mutex); MpegPlaybackState &playback = getPlaybackState(param_1); MpegPlaybackState resetState = makeFreshPlaybackStatePreservingConfig(playback); - if (playback.streamEnded || playback.decoderFailed) + // Only meaningful while sceCdSt* drives the stream: appendPssBytes + // clears a carried streamEnded when the generation advances, and + // only notifyMpegCdStreamStart advances it. Carrying it for a game + // that streams with plain sceCdRead wedges every later movie. + const bool cdStreamDriven = g_mpeg_stub_state.cdStreamBytesProduced != 0u; + if (cdStreamDriven && (playback.streamEnded || playback.decoderFailed)) { resetState.sawInput = true; resetState.streamEnded = true; diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/MemoryCard.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/MemoryCard.cpp index 564d87381..9f6d76675 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/MemoryCard.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/MemoryCard.cpp @@ -107,9 +107,35 @@ namespace ps2_stubs cvMcKilobytes(kCvMcIconInfoBytes) + cvMcKilobytes(kCvMcIconFileBytes) + 11; + // DQ8_MC_TRACE=1 logs the card calls. RUNTIME_LOG is compile-time and + // lives in a header the whole recompiled corpus includes, so turning it + // on to look at one stub costs a full rebuild. + bool mcTraceEnabled() + { + static const bool on = [] { + const char *value = std::getenv("DQ8_MC_TRACE"); + return value != nullptr && *value != '\0' && *value != '0'; + }(); + return on; + } + +#define MC_TRACE(...) \ + do \ + { \ + if (mcTraceEnabled()) \ + { \ + std::fprintf(stderr, "[mc] " __VA_ARGS__); \ + } \ + } while (0) + bool isValidMcPortSlot(int32_t port, int32_t slot) { - return port >= 0 && port < static_cast(g_mcPorts.size()) && slot == 0; + // The slot argument selects a multitap sub-slot. This runtime models + // a console with no multitap -- one card per port, and getMcRootPath() + // ignores the slot entirely -- so any slot query for a port answers + // for that port's card. Rejecting everything but slot 0 made DQ8 see + // no card at all: it passes slot=1 to every libmc call. + return port >= 0 && port < static_cast(g_mcPorts.size()) && slot >= 0; } std::filesystem::path getMcRootPath(int32_t port) @@ -338,7 +364,12 @@ namespace ps2_stubs } else if (patternPos < pattern.size() && pattern[patternPos] == '*') { - starPos = patternPos++; + while (patternPos < pattern.size() && + (pattern[patternPos] == '*' || pattern[patternPos] == '?')) + { + ++patternPos; + } + starPos = patternPos - 1u; matchPos = valuePos; } else if (starPos != std::string::npos) @@ -357,7 +388,9 @@ namespace ps2_stubs ++patternPos; } - return patternPos == pattern.size(); + // MCMAN accepts '?' at the end of a shorter filename, even if + // more pattern characters follow (mcman_checkdirpath in PS2SDK). + return patternPos == pattern.size() || pattern[patternPos] == '?'; } void setMcCommandResultLocked(int32_t cmd, int32_t result) @@ -844,6 +877,8 @@ namespace ps2_stubs } RUNTIME_LOG("[MC] GetDir port=" << port << " '" << rawPath << "' maxent=" << maxEntries << " -> result=" << result); + MC_TRACE("GetDir port=%d '%s' maxent=%d -> result=%d\n", port, rawPath.c_str(), + maxEntries, result); setReturnS32(ctx, 0); } @@ -908,6 +943,9 @@ namespace ps2_stubs RUNTIME_LOG("[MC] GetInfo port=" << port << " type=" << cardType << " free=" << freeBlocks << " format=" << format << " result=" << result); + MC_TRACE("GetInfo port=%d slot=%d -> type=%d free=%d format=%d result=%d " + "(typePtr=%08x freePtr=%08x formatPtr=%08x)\n", + port, slot, cardType, freeBlocks, format, result, typePtr, freePtr, formatPtr); setReturnS32(ctx, 0); } @@ -933,6 +971,8 @@ namespace ps2_stubs } ensureMcRootExists(0); ensureMcRootExists(1); + MC_TRACE("Init -> 0, roots %s | %s\n", getMcRootPath(0).string().c_str(), + getMcRootPath(1).string().c_str()); setReturnS32(ctx, 0); } @@ -1102,11 +1142,13 @@ namespace ps2_stubs { const std::filesystem::path oldHostPath = guestMcPathToHostPath(port, normalizeGuestMcPathLocked(port, oldPath)); - const std::filesystem::path newHostPath = - guestMcPathToHostPath(port, normalizeGuestMcPathLocked(port, newPath)); + // libmc renames an entry within its existing directory. + const std::filesystem::path newHostPath = oldHostPath.parent_path() / newPath; std::error_code ec; - if (std::filesystem::exists(oldHostPath, ec) && !ec && - std::filesystem::exists(newHostPath.parent_path(), ec) && !ec) + if (!newPath.empty() && newPath != "." && newPath != ".." && + newPath.find_first_of("/\\") == std::string::npos && + std::filesystem::exists(oldHostPath, ec) && !ec && + !std::filesystem::exists(newHostPath, ec) && !ec) { std::filesystem::rename(oldHostPath, newHostPath, ec); result = ec ? kMcResultDeniedPermit : kMcResultSucceed; @@ -1116,6 +1158,7 @@ namespace ps2_stubs setMcCommandResultLocked(kMcCmdRename, result); } + MC_TRACE("Rename port=%d '%s' -> '%s' result=%d\n", port, oldPath.c_str(), newPath.c_str(), result); setReturnS32(ctx, 0); } @@ -1209,11 +1252,13 @@ namespace ps2_stubs // on it to tell idle polling apart from command completion. if (!hadPending) { + MC_TRACE("Sync -> idle (-1)\n"); setReturnS32(ctx, -1); return; } RUNTIME_LOG("[MC] Sync cmd=" << cmd << " result=" << result); + MC_TRACE("Sync cmd=%d result=%d\n", cmd, result); if (cmdPtr != 0u) { diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/SIF.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/SIF.cpp index 6a07fe278..68d984c4c 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/SIF.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/SIF.cpp @@ -54,6 +54,13 @@ namespace ps2_stubs }; static_assert(sizeof(Ps2SifDmaTransfer) == 16u, "Unexpected SIF DMA descriptor size"); + // Once a game runs IOP modules natively, IOP memory is that IOP's. + ps2x::iop::NativeIop *nativeIop(PS2Runtime *runtime) + { + ps2x::iop::NativeIop *native = PS2IopTransport::native(runtime); + return native && native->enabled() ? native : nullptr; + } + std::mutex g_sifDmaTransferMutex; uint32_t g_nextSifDmaTransferId = 1u; std::mutex g_sifCmdStateMutex; @@ -448,19 +455,19 @@ namespace ps2_stubs void sceSifAllocIopHeap(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { (void)rdram; - (void)runtime; const uint32_t reqSize = getRegU32(ctx, 4); - setReturnU32(ctx, allocateSifHeapBlock(reqSize)); + ps2x::iop::NativeIop *native = nativeIop(runtime); + setReturnU32(ctx, native ? native->allocate(reqSize) : allocateSifHeapBlock(reqSize)); } void sceSifAllocSysMemory(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { (void)rdram; - (void)runtime; const uint32_t size = getRegU32(ctx, 5); - setReturnU32(ctx, allocateSifHeapBlock(size)); + ps2x::iop::NativeIop *native = nativeIop(runtime); + setReturnU32(ctx, native ? native->allocate(size) : allocateSifHeapBlock(size)); } void sceSifBindRpc(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) @@ -503,19 +510,20 @@ namespace ps2_stubs void sceSifFreeIopHeap(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { (void)rdram; - (void)runtime; const uint32_t addr = getRegU32(ctx, 4); + if (ps2x::iop::NativeIop *native = nativeIop(runtime)) + { + native->release(addr); + setReturnS32(ctx, 0); + return; + } setReturnS32(ctx, freeSifHeapBlock(addr) ? 0 : -1); } void sceSifFreeSysMemory(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { - (void)rdram; - (void)runtime; - - const uint32_t addr = getRegU32(ctx, 4); - setReturnS32(ctx, freeSifHeapBlock(addr) ? 0 : -1); + sceSifFreeIopHeap(rdram, ctx, runtime); } void sceSifGetDataTable(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) @@ -779,6 +787,37 @@ namespace ps2_stubs const uint32_t dmatAddr = getRegU32(ctx, 4); const uint32_t count = getRegU32(ctx, 5); + // With native modules, EE-to-IOP transfers land in that IOP's memory. + if (ps2x::iop::NativeIop *native = nativeIop(runtime); native && dmatAddr != 0u && count <= 32u) + { + for (uint32_t i = 0; i < count; ++i) + { + Ps2SifDmaTransfer xfer{}; + const uint8_t *entry = getConstMemPtr(rdram, dmatAddr + i * static_cast(sizeof(xfer))); + if (!entry) + { + break; + } + std::memcpy(&xfer, entry, sizeof(xfer)); + if (xfer.size <= 0) + { + continue; + } + const uint32_t size = static_cast(xfer.size); + // The source has to be one contiguous stretch of EE memory. + const uint8_t *first = getConstMemPtr(rdram, xfer.src); + const uint8_t *last = getConstMemPtr(rdram, xfer.src + size - 1u); + if (!first || !last || static_cast(last - first) != size - 1u) + { + continue; + } + native->write(xfer.dest, first, size); + } + ps2_syscalls::dispatchDmacHandlersForCause(rdram, runtime, 5u); + setReturnS32(ctx, static_cast(allocateSifDmaTransferId())); + return; + } + const uint32_t listAddr = getRegU32(ctx, 4); PS2_IF_AGRESSIVE_LOGS({ std::cerr << "[sceSifSetDma:CALL] pc=0x" << std::hex << ctx->pc diff --git a/ps2xRuntime/src/lib/Kernel/Stubs/VU.cpp b/ps2xRuntime/src/lib/Kernel/Stubs/VU.cpp index b7840d447..e405c41d5 100644 --- a/ps2xRuntime/src/lib/Kernel/Stubs/VU.cpp +++ b/ps2xRuntime/src/lib/Kernel/Stubs/VU.cpp @@ -168,22 +168,44 @@ namespace ps2_stubs out[3] = fullFtoi4 ? static_cast(t[3] * 16.0f) : static_cast(t[3]); } - // Guard-band proxy for the COP2 sticky clip flags: nonzero => the - // vertex is offscreen. Not the hardware per-plane flag layout. - constexpr float kScreenClipGuard = 4096.0f; int32_t screenClipCode(const float (&v)[4]) { + // The SDK returns sticky Z/S from (x,y,w)-0 and 4096-(x,y). + // Inspecting bits also preserves VU signed-zero/denormal behavior. int32_t code = 0; - if (v[0] > kScreenClipGuard) - code |= 0x1; - if (v[0] < -kScreenClipGuard) - code |= 0x2; - if (v[1] > kScreenClipGuard) - code |= 0x4; - if (v[1] < -kScreenClipGuard) - code |= 0x8; + for (unsigned lane : {0u, 1u, 3u}) + { + uint32_t bits; + std::memcpy(&bits, &v[lane], sizeof(bits)); + const uint32_t magnitude = bits & 0x7fffffffu; + if (magnitude < 0x00800000u) + code |= 0x40; + if (bits & 0x80000000u) + code |= 0x80; + else if (lane != 3u) + { + if (magnitude == 0x45800000u) + code |= 0x40; + else if (magnitude > 0x45800000u) + code |= 0x80; + } + } return code; } + + void traceScreenClip(const R5900Context *ctx, const float (&v)[4], int32_t code) + { + static const bool enabled = std::getenv("PS2_VU_TRACE_SCREEN_CLIP") != nullptr; + if (!enabled) + return; + static uint64_t calls = 0, rejected = 0; + ++calls; + rejected += code != 0; + if (calls <= 16 || (calls & (calls - 1)) == 0) + std::fprintf(stderr, "[vu-screen-clip] calls=%llu rejected=%llu ra=%08x v=(%g,%g,%g,%g) flags=%02x\n", + static_cast(calls), static_cast(rejected), + getRegU32(ctx, 31), v[0], v[1], v[2], v[3], code); + } } void sceVpu0Reset(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) @@ -340,6 +362,7 @@ namespace ps2_stubs if (readVuVec4f(rdram, vAddr, v)) { code = screenClipCode(v); + traceScreenClip(ctx, v, code); } setReturnS32(ctx, code); } @@ -354,14 +377,17 @@ namespace ps2_stubs if (readVuVec4f(rdram, v0Addr, v0)) { code |= screenClipCode(v0); + traceScreenClip(ctx, v0, screenClipCode(v0)); } if (readVuVec4f(rdram, v1Addr, v1)) { code |= screenClipCode(v1); + traceScreenClip(ctx, v1, screenClipCode(v1)); } if (readVuVec4f(rdram, v2Addr, v2)) { code |= screenClipCode(v2); + traceScreenClip(ctx, v2, screenClipCode(v2)); } setReturnS32(ctx, code); } diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp index 039da2fb0..5792b0552 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/Deci2.cpp @@ -2,6 +2,8 @@ #include "Common.h" #include "ps2_runtime.h" +#include + namespace { struct Deci2Session @@ -62,11 +64,41 @@ namespace return text; } + // Default cap on DECI2 text lines, so a game that spams kputs cannot drown + // the console. Override with PS2X_DECI2_LOG_LIMIT; 0 means unlimited, which + // is what you want when diffing a full boot trace against a reference + // capture from real hardware or an emulator -- those run to many thousands + // of lines and a silent truncation at 256 looks exactly like the guest + // stopping. + static uint32_t deci2TextLogLimit() + { + static const uint32_t limit = []() -> uint32_t + { + constexpr uint32_t kDefaultMaxDeci2TextLogs = 256u; + const char *env = std::getenv("PS2X_DECI2_LOG_LIMIT"); + if (!env || *env == '\0') + { + return kDefaultMaxDeci2TextLogs; + } + + char *parseEnd = nullptr; + const unsigned long parsed = std::strtoul(env, &parseEnd, 10); + if (parseEnd == env || *parseEnd != '\0' || parsed > 0xFFFFFFFFul) + { + return kDefaultMaxDeci2TextLogs; + } + + return static_cast(parsed); + }(); + + return limit; + } + static void logDeci2Text(const char *prefix, const std::string &text) { - constexpr uint32_t kMaxDeci2TextLogs = 256u; + const uint32_t maxDeci2TextLogs = deci2TextLogLimit(); const uint32_t logIndex = g_deci2LogCount.fetch_add(1u, std::memory_order_relaxed); - if (logIndex >= kMaxDeci2TextLogs) + if (maxDeci2TextLogs != 0u && logIndex >= maxDeci2TextLogs) { return; } diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/RPC.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/RPC.cpp index f5dfcc5af..414251689 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/RPC.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/RPC.cpp @@ -1,6 +1,7 @@ #include "Common.h" #include "RPC.h" #include "../../ps2_iop_transport.h" +#include "runtime/ps2_native_iop.h" namespace ps2_syscalls { @@ -194,6 +195,20 @@ namespace ps2_syscalls setReturnS32(ctx, -1); return; } + // A module the game runs natively also loads onto the emulated IOP. + // The tracker below still hands out the id the game sees. + if (ps2x::iop::NativeIop *native = PS2IopTransport::native(runtime); native && native->claims(modulePath)) + { + const uint32_t argLength = std::min(getRegU32(ctx, 5), 512u); + std::string arguments(argLength, '\0'); + for (uint32_t i = 0; i < argLength; ++i) + { + if (const uint8_t *byte = getConstMemPtr(rdram, getRegU32(ctx, 6) + i)) + arguments[i] = static_cast(*byte); + } + if (native->load(modulePath, arguments) > 0) + ps2_native_iop::startAudio(*runtime); + } const int32_t moduleId = trackSifModuleLoad(modulePath); if (moduleId <= 0) @@ -219,10 +234,7 @@ namespace ps2_syscalls void SifInitRpc(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime) { std::lock_guard lock(g_rpc_mutex); - if (runtime) - { - PS2IopTransport::reset(runtime); - } + // Initializing the EE RPC client must not reboot the IOP services. if (!g_rpc_initialized) { g_rpc_servers.clear(); diff --git a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp index 3530e65bb..094a4039e 100644 --- a/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp +++ b/ps2xRuntime/src/lib/Kernel/Syscalls/System.cpp @@ -578,6 +578,14 @@ namespace ps2_syscalls if (runtime) { + // A guest running its own allocator is describing *its* heap; + // moving the runtime arena there would collide with it. + if (runtime->guestHeapCeiling() != 0u) + { + setReturnU32(ctx, heapBase); + return; + } + runtime->configureGuestHeap(heapBase, heapLimit); PS2_IF_AGRESSIVE_LOGS({ @@ -604,9 +612,13 @@ namespace ps2_syscalls static constexpr uint32_t kDefaultGuestHeapEnd = 0x01F00000u; - const uint32_t ret = runtime - ? runtime->guestHeapLimit() - : kDefaultGuestHeapEnd; + // A guest running its own allocator grows right up to this value. + uint32_t ret = kDefaultGuestHeapEnd; + if (runtime) + { + const uint32_t ceiling = runtime->guestHeapCeiling(); + ret = (ceiling != 0u) ? ceiling : runtime->guestHeapLimit(); + } setReturnU32(ctx, ret); } diff --git a/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp b/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp index 9c39ae2e7..15ef50bcd 100644 --- a/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_cpu_backend.cpp @@ -420,22 +420,6 @@ namespace return {(smode2 & 0x1ull) != 0ull, ((smode2 >> 1) & 0x1ull) != 0ull}; } - void applyFieldPresentation(std::vector &pixels, uint32_t width, uint32_t height, bool oddField) - { - if (pixels.empty() || width == 0u || height < 2u) - return; - const std::vector source = pixels; - for (uint32_t y = 0; y < height; ++y) - { - uint32_t sourceY = ((y >> 1u) << 1u) + (oddField ? 1u : 0u); - if (sourceY >= height) - sourceY = height - 1u; - std::memcpy(pixels.data() + y * kHostFrameWidth * 4u, - source.data() + sourceY * kHostFrameWidth * 4u, - width * 4u); - } - } - void normalizePresentationAlpha(std::vector &pixels, uint32_t width, uint32_t height) { for (uint32_t y = 0; y < height; ++y) @@ -1687,7 +1671,8 @@ bool GSCpuBackend::CopyFrameToHostRgba(const GSFrameReg &frame, bool useLocalMemoryLayout, bool frameBaseIsPages, uint32_t sourceOriginX, - uint32_t sourceOriginY) const + uint32_t sourceOriginY, + bool doubleSourceRows) const { if (!m_vram || m_vramSize == 0u) return false; @@ -1705,7 +1690,7 @@ bool GSCpuBackend::CopyFrameToHostRgba(const GSFrameReg &frame, for (uint32_t x = 0; x < width; ++x) { const uint32_t sx = sourceOriginX + x; - const uint32_t sy = sourceOriginY + y; + const uint32_t sy = sourceOriginY + (doubleSourceRows ? (y >> 1u) : y); if (frame.psm == GS_PSM_CT32 || frame.psm == GS_PSM_CT24) { uint32_t color = 0u; @@ -1774,14 +1759,17 @@ PresentationFrame GSCpuBackend::Present(const GSPresentationRequest &request) PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationRequest &request) { PresentationFrame result{}; + result.rowPitchBytes = kHostFrameWidth * 4u; const GSPmodeState pmode = decodePmode(request.pmode); const GSSmode2State smode2 = decodeSMode2(request.smode2); - const bool fieldMode = smode2.interlaced && !smode2.frameMode; - const bool oddField = (request.vsyncTick & 1ull) != 0ull; + // SMODE2 FFMD=1 (FRAME) reads a half-height buffer whole once per field, so the + // woven host frame line-doubles it. FFMD=0 (FIELD) buffers a full frame whose + // two fields are alternate buffer rows: weaving is just reading every row. + const bool halfHeightSource = smode2.interlaced && smode2.frameMode; const GSFrameReg displayFrame1 = decodeDisplayFrame(request.dispfb1); const GSFrameReg displayFrame2 = decodeDisplayFrame(request.dispfb2); - const GSDisplayReadOrigin origin1 = decodeDisplayReadOrigin(request.dispfb1); - const GSDisplayReadOrigin origin2 = decodeDisplayReadOrigin(request.dispfb2); + GSDisplayReadOrigin origin1 = decodeDisplayReadOrigin(request.dispfb1); + GSDisplayReadOrigin origin2 = decodeDisplayReadOrigin(request.dispfb2); uint32_t width1 = 0u, height1 = 0u, width2 = 0u, height2 = 0u; decodeDisplaySize(request.display1, width1, height1); decodeDisplaySize(request.display2, width2, height2); @@ -1790,6 +1778,14 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque if (!valid1 && !valid2) return result; + // Both circuits on one buffer a single row apart is a deliberate flicker filter; + // snapping them together drops the half-row blur, as PCSX2 does by default. + const uint32_t originGap = origin1.y > origin2.y ? origin1.y - origin2.y : origin2.y - origin1.y; + const bool sameReadSource = displayFrame1.fbp == displayFrame2.fbp && displayFrame1.fbw == displayFrame2.fbw && + displayFrame1.psm == displayFrame2.psm && origin1.x == origin2.x; + if (valid1 && valid2 && sameReadSource && originGap == 1u) + origin1.y = origin2.y = std::min(origin1.y, origin2.y); + auto copySource = [&](const GSFrameReg &displayFrame, const GSDisplayReadOrigin &origin, uint32_t width, @@ -1805,12 +1801,12 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque usedPreferred = false; if (allowPreferred && request.hasPreferredSource && request.preferredDestFbp == displayFrame.fbp && (request.preferredSource.fbw != 0u || request.preferredSource.fbp != displayFrame.fbp) && - CopyFrameToHostRgba(request.preferredSource, width, height, pixels, preserveAlpha, true, false, 0u, 0u)) + CopyFrameToHostRgba(request.preferredSource, width, height, pixels, preserveAlpha, true, false, 0u, 0u, halfHeightSource)) { selected = request.preferredSource; usedPreferred = true; } - if (pixels.empty() && !CopyFrameToHostRgba(displayFrame, width, height, pixels, preserveAlpha, true, true, origin.x, origin.y)) + if (pixels.empty() && !CopyFrameToHostRgba(displayFrame, width, height, pixels, preserveAlpha, true, true, origin.x, origin.y, halfHeightSource)) return false; if (!usedPreferred && displayFrame.fbp == 0u && countNonBlackPixels(pixels, width, height) == 0u) @@ -1820,7 +1816,7 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque if (candidate.fbp == selected.fbp && candidate.fbw == selected.fbw && candidate.psm == selected.psm) continue; std::vector candidatePixels; - if (!CopyFrameToHostRgba(candidate, width, height, candidatePixels, preserveAlpha, true, true, 0u, 0u)) + if (!CopyFrameToHostRgba(candidate, width, height, candidatePixels, preserveAlpha, true, true, 0u, 0u, halfHeightSource)) continue; if (countNonBlackPixels(candidatePixels, width, height) == 0u) continue; @@ -1870,8 +1866,6 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque dst[3] = pmode.amod ? dst[3] : src[3]; } normalizePresentationAlpha(result.pixels, result.width, result.height); - if (fieldMode) - applyFieldPresentation(result.pixels, result.width, result.height, oddField); result.displayFbp = displayFrame1.fbp; result.sourceFbp = selected1.fbp; return result; @@ -1885,8 +1879,6 @@ PresentationFrame GSCpuBackend::PresentFromLocalMemory(const GSPresentationReque GSFrameReg selected = displayFrame; if (!copySource(displayFrame, origin, result.width, result.height, true, false, selected, result.pixels, result.usedPreferred)) return {}; - if (fieldMode) - applyFieldPresentation(result.pixels, result.width, result.height, oddField); normalizePresentationAlpha(result.pixels, result.width, result.height); result.displayFbp = displayFrame.fbp; result.sourceFbp = selected.fbp; diff --git a/ps2xRuntime/src/lib/gs/gs_frontend.cpp b/ps2xRuntime/src/lib/gs/gs_frontend.cpp index fd5a90d28..6c7de3b43 100644 --- a/ps2xRuntime/src/lib/gs/gs_frontend.cpp +++ b/ps2xRuntime/src/lib/gs/gs_frontend.cpp @@ -2,18 +2,25 @@ #include "runtime/gs/gs_cpu_backend.h" #include "ps2_log.h" #include "runtime/ps2_memory.h" + +// Weak on purpose: the offline gs-dump harness links this file without the pad +// backend, and there a screenshot just falls back to counting latches. +extern "C++" __attribute__((weak)) uint64_t ps2PadCurrentGuestFrame(); #include #include +#include #include #include +#include #include +#include #include +#include #include +#include namespace { - static constexpr uint32_t kHostFrameWidth = 640u; - GSPrimReg decodePrimRegister(uint64_t value) { GSPrimReg prim{}; @@ -125,6 +132,7 @@ void GS::init(uint8_t *vram, uint32_t vramSize, GSRegisters *privRegs) void GS::reset() { std::lock_guard lock(m_stateMutex); + std::lock_guard backendLock(m_backendLifetimeMutex); std::memset(m_ctx, 0, sizeof(m_ctx)); m_prim = {}; m_primRegister = {}; @@ -170,6 +178,7 @@ void GS::reset() m_hostPresentationFrame.clear(); m_hostPresentationWidth = 0u; m_hostPresentationHeight = 0u; + m_hostPresentationRowPitchBytes = 0u; m_hostPresentationDisplayFbp = 0u; m_hostPresentationSourceFbp = 0u; m_hostPresentationUsedPreferred = false; @@ -524,9 +533,13 @@ GSPresentationRequest GS::buildPresentationRequestUnlocked() const return request; } +static void maybeWriteScreenshot(GS &gs); + void GS::latchHostPresentationFrame() { GSPresentationRequest request{}; + GSPresentationTicket ticket; + std::unique_lock backendLock(m_backendLifetimeMutex, std::defer_lock); { std::lock_guard lock(m_stateMutex); if (!m_backend || !m_privRegs) @@ -535,46 +548,138 @@ void GS::latchHostPresentationFrame() m_hostPresentationFrame.clear(); m_hasHostPresentationFrame = false; m_hostPresentationWidth = m_hostPresentationHeight = 0u; + m_hostPresentationRowPitchBytes = 0u; return; } + backendLock.lock(); request = buildPresentationRequestUnlocked(); + if (m_backend->QueuesPreparedPresentation()) + ticket = m_backend->PreparePresentation(request); } PresentationFrame frame{}; + if (ticket) + frame = m_backend->DisplayPreparedPresentation(ticket); + else { - std::lock_guard backendLock(m_backendLifetimeMutex); - if (m_backend) - { - m_backend->Flush(); - m_backend->Sync(GSSyncReason::Presentation); - frame = m_backend->Present(request); - } + m_backend->Flush(); + m_backend->Sync(GSSyncReason::Presentation); + frame = m_backend->Present(request); } - const bool hasFrame = static_cast(frame); + const bool presented = static_cast(frame); + const bool hasHostFrame = presented && frame.mode == GSPresentationMode::HostPixels; const uint32_t displayFbp = frame.displayFbp; const uint32_t sourceFbp = frame.sourceFbp; const uint32_t width = frame.width; const uint32_t height = frame.height; + uint32_t rowPitchBytes = frame.rowPitchBytes; + if (rowPitchBytes == 0u && width <= std::numeric_limits::max() / 4u) + rowPitchBytes = width * 4u; const bool usedPreferred = frame.usedPreferred; { std::lock_guard presentationLock(m_presentationMutex); - m_hostPresentationFrame = std::move(frame.pixels); - m_hostPresentationWidth = width; - m_hostPresentationHeight = height; + if (hasHostFrame) + m_hostPresentationFrame = std::move(frame.pixels); + else + m_hostPresentationFrame.clear(); + m_hostPresentationWidth = hasHostFrame ? width : 0u; + m_hostPresentationHeight = hasHostFrame ? height : 0u; + m_hostPresentationRowPitchBytes = hasHostFrame ? rowPitchBytes : 0u; m_hostPresentationDisplayFbp = displayFbp; m_hostPresentationSourceFbp = sourceFbp; m_hostPresentationUsedPreferred = usedPreferred; - m_hasHostPresentationFrame = hasFrame; + m_hasHostPresentationFrame = hasHostFrame; } + ticket.reset(); + backendLock.unlock(); - if (hasFrame) + if (hasHostFrame) + { + maybeWriteScreenshot(*this); + } + + if (presented) { std::lock_guard lock(m_stateMutex); recordPresentDebugEventUnlocked(displayFbp, sourceFbp, width, height, usedPreferred); } } +// DQ8_GFX_SCREENSHOT_DIR + DQ8_GFX_SCREENSHOT_EVERY=N dump the presented frame +// as a binary PPM every N frames. PPM keeps this dependency-free; convert with +// `ffmpeg -i shot.ppm shot.png` when a viewer is wanted. A backend that presents +// natively skips the CPU compose, so it disables that fast path while this is on. +// +// A free function on purpose: gs_frontend.h reaches the whole recompiled corpus +// through ps2_runtime.h, so declaring this on GS would cost a full rebuild. +static void maybeWriteScreenshot(GS &gs) +{ + struct Config + { + std::string directory; + uint64_t every = 0u; + }; + static const Config config = [] { + Config parsed{}; + const char *dir = std::getenv("DQ8_GFX_SCREENSHOT_DIR"); + if (dir == nullptr || *dir == '\0') + { + return parsed; + } + parsed.directory = dir; + const char *every = std::getenv("DQ8_GFX_SCREENSHOT_EVERY"); + const long long value = every ? std::atoll(every) : 0; + parsed.every = value > 0 ? static_cast(value) : 60ull; + return parsed; + }(); + if (config.every == 0u) + { + return; + } + + // Named by guest vsync tick, the same clock DQ8_PAD_SCRIPT uses, so a + // screenshot tells you directly which frame to script an input at. + static uint64_t s_latches = 0u; + const uint64_t frame = + ps2PadCurrentGuestFrame != nullptr ? ps2PadCurrentGuestFrame() : s_latches++; + static uint64_t s_lastWritten = std::numeric_limits::max(); + if (s_lastWritten != std::numeric_limits::max() && + frame < s_lastWritten + config.every) + { + return; + } + s_lastWritten = frame; + + std::vector pixels; + uint32_t width = 0u; + uint32_t height = 0u; + if (!gs.copyLatchedHostPresentationFrame(pixels, width, height, nullptr, nullptr, nullptr) || + width == 0u || height == 0u) + { + return; + } + + char path[512]; + std::snprintf(path, sizeof(path), "%s/frame_%06llu.ppm", config.directory.c_str(), + static_cast(frame)); + std::ofstream out(path, std::ios::binary); + if (!out) + { + return; + } + out << "P6\n" << width << " " << height << "\n255\n"; + // copyLatchedHostPresentationFrame hands back tightly packed RGBA. + std::vector rgb(static_cast(width) * height * 3u); + for (size_t i = 0u, n = static_cast(width) * height; i < n; ++i) + { + rgb[i * 3u + 0u] = pixels[i * 4u + 0u]; + rgb[i * 3u + 1u] = pixels[i * 4u + 1u]; + rgb[i * 3u + 2u] = pixels[i * 4u + 2u]; + } + out.write(reinterpret_cast(rgb.data()), static_cast(rgb.size())); +} + bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, uint32_t &outWidth, uint32_t &outHeight, @@ -610,7 +715,20 @@ bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, outPixels.resize(packedRowBytes * static_cast(outHeight)); if (outWidth != 0u && outHeight != 0u) { - const size_t sourceRowBytes = static_cast(kHostFrameWidth) * 4u; + const size_t sourceRowBytes = m_hostPresentationRowPitchBytes; + if (sourceRowBytes < packedRowBytes) + { + outPixels.clear(); + outWidth = 0u; + outHeight = 0u; + if (outDisplayFbp) + *outDisplayFbp = 0u; + if (outSourceFbp) + *outSourceFbp = 0u; + if (outUsedPreferred) + *outUsedPreferred = false; + return false; + } for (uint32_t y = 0; y < outHeight; ++y) { const size_t srcOffset = static_cast(y) * sourceRowBytes; @@ -638,8 +756,41 @@ bool GS::copyLatchedHostPresentationFrame(std::vector &outPixels, return true; } +// Debug accounting: total time decoding GIF packets, and how many arrived. +// A backend can subtract its own time from this to see what packet decoding +// costs on its own, which is the difference between "the renderer is slow" and +// "the game is sending an enormous number of tiny packets". +std::atomic g_gsFrontendPacketNanos{0}; +std::atomic g_gsFrontendPacketCount{0}; +// Off unless a backend reports the numbers: the field sends over ten thousand +// packets a frame, and two clock reads each add up. +std::atomic g_gsFrontendTiming{false}; +// The same for the native image-upload fast path, which bypasses +// processGIFPacket entirely -- DQ8's movie tiles arrive this way, one DMA +// chain per 16x16 tile, so without a separate counter they are invisible. +std::atomic g_gsUploadNativeNanos{0}; +std::atomic g_gsUploadNativeCount{0}; + void GS::processGIFPacket(const uint8_t *data, uint32_t sizeBytes) { + struct PacketTimer + { + bool timed = g_gsFrontendTiming.load(std::memory_order_relaxed); + std::chrono::steady_clock::time_point start = + timed ? std::chrono::steady_clock::now() : std::chrono::steady_clock::time_point{}; + ~PacketTimer() + { + if (!timed) + return; + g_gsFrontendPacketNanos.fetch_add( + static_cast(std::chrono::duration_cast( + std::chrono::steady_clock::now() - start) + .count()), + std::memory_order_relaxed); + g_gsFrontendPacketCount.fetch_add(1, std::memory_order_relaxed); + } + } packetTimer; + std::lock_guard lock(m_stateMutex); if (!data || sizeBytes < 16 || !m_backend) return; @@ -783,8 +934,17 @@ void GS::uploadImageNative(uint64_t bitbltbuf, const uint8_t *data, uint32_t sizeBytes) { - std::lock_guard lock(m_stateMutex); - uploadImageNativeUnlocked(bitbltbuf, trxpos, trxreg, trxdir, data, sizeBytes); + const auto start = std::chrono::steady_clock::now(); + { + std::lock_guard lock(m_stateMutex); + uploadImageNativeUnlocked(bitbltbuf, trxpos, trxreg, trxdir, data, sizeBytes); + } + g_gsUploadNativeNanos.fetch_add( + static_cast(std::chrono::duration_cast( + std::chrono::steady_clock::now() - start) + .count()), + std::memory_order_relaxed); + g_gsUploadNativeCount.fetch_add(1, std::memory_order_relaxed); } void GS::uploadImageNativeUnlocked(uint64_t bitbltbuf, diff --git a/ps2xRuntime/src/lib/gs/gs_threaded_backend.cpp b/ps2xRuntime/src/lib/gs/gs_threaded_backend.cpp new file mode 100644 index 000000000..e5ba622f4 --- /dev/null +++ b/ps2xRuntime/src/lib/gs/gs_threaded_backend.cpp @@ -0,0 +1,344 @@ +#include "runtime/gs/gs_threaded_backend.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#if defined(__APPLE__) || defined(__linux__) +#include +#endif + +struct GSThreadedBackend::Impl +{ + struct PresentationResult + { + std::mutex mutex; + std::condition_variable ready; + bool completed = false; + GSPresentationTicket frame; + std::exception_ptr error; + void fail(std::exception_ptr failure) + { + std::lock_guard lock(mutex); + if (completed) return; + error = failure; + completed = true; + ready.notify_all(); + } + }; + struct PresentationTicket final : GSPreparedPresentation + { + std::shared_ptr owner; + std::shared_ptr result; + PresentationTicket(std::shared_ptr owner, std::shared_ptr result) + : owner(std::move(owner)), result(std::move(result)) {} + }; + struct Prepare + { + GSPresentationRequest request; + std::shared_ptr result; + }; + enum class Op { Submit, Transfer, Upload, Flush, TextureFlush, Write, Prepare }; + using Pixel = std::array; + struct Command + { + Op op; + std::variant, Pixel, Prepare> data; + int rounding = std::fegetround(); + size_t bytes() const + { + return sizeof(Command) + (op == Op::Upload ? std::get>(data).size() : 0u); + } + }; + + explicit Impl(std::unique_ptr target, size_t queueBytes) + : backend(std::move(target)), capacity(std::max(queueBytes, sizeof(Command))) + { + if (!backend) + throw std::invalid_argument("GS worker requires a backend"); + worker = std::thread([this] { run(); }); + } + + ~Impl() + { + backend->CancelPreparedPresentations(); + { + std::lock_guard lock(mutex); + stopping = true; + } + work.notify_one(); + worker.join(); + } + + void enqueue(Command command, bool flush = false) + { + // A presenter may also submit. Holding this lock across a drain keeps + // newer commands behind the synchronous operation, not merely its fence. + std::lock_guard submitLock(submission); + const size_t size = command.bytes(); + std::unique_lock lock(mutex); + if (outstandingBytes != 0u && size > capacity - std::min(capacity, outstandingBytes)) + { + urgent = true; + work.notify_one(); + progress.wait(lock, [&] { + return error || outstandingBytes == 0u || + size <= capacity - std::min(capacity, outstandingBytes); + }); + } + rethrow(); + pending.push_back(std::move(command)); + outstandingBytes += size; + urgent = urgent || flush; + if (pending.size() == 1u || pending.size() == 64u || urgent) + work.notify_one(); + } + + template + auto observe(F &&fn) + { + std::lock_guard submitLock(submission); + { + std::unique_lock lock(mutex); + urgent = true; + work.notify_one(); + progress.wait(lock, [&] { return outstandingBytes == 0u || error; }); + rethrow(); + } + return fn(*backend); + } + + void rethrow() const + { + if (error) + std::rethrow_exception(error); + } + + void execute(const Command &command) + { + // XGKICK submits while the VU has selected rounding toward zero. + if (rounding != command.rounding) + { + std::fesetround(command.rounding); + rounding = command.rounding; + } + switch (command.op) + { + case Op::Submit: backend->Submit(std::get(command.data)); break; + case Op::Transfer: backend->BeginTransfer(std::get(command.data)); break; + case Op::Upload: + { + const auto &bytes = std::get>(command.data); + backend->UploadImage(bytes.data(), static_cast(bytes.size())); + break; + } + case Op::Flush: backend->Flush(); break; + case Op::TextureFlush: backend->TextureFlush(); break; + case Op::Write: + { + const auto &p = std::get(command.data); + backend->WriteVram(p[0], p[1], p[2], p[3], p[4], p[5]); + break; + } + case Op::Prepare: + { + const auto &prepare = std::get(command.data); + backend->Flush(); + backend->Sync(GSSyncReason::Presentation); + auto frame = backend->PreparePresentation(prepare.request); + if (!frame) + throw std::runtime_error("GS presentation preparation returned no ticket"); + std::lock_guard lock(prepare.result->mutex); + prepare.result->frame = std::move(frame); + prepare.result->completed = true; + prepare.result->ready.notify_all(); + break; + } + } + } + + void run() + { + rounding = std::fegetround(); +#if defined(__APPLE__) + pthread_setname_np("GSWorker"); +#elif defined(__linux__) + pthread_setname_np(pthread_self(), "GSWorker"); +#endif + std::vector batch; + for (;;) + { + size_t bytes = 0u; + { + std::unique_lock lock(mutex); + work.wait(lock, [&] { return stopping || !pending.empty(); }); + if (pending.empty() && stopping) + return; + // Batch small primitives without stranding a short final packet. + work.wait_for(lock, std::chrono::milliseconds(1), [&] { + return stopping || urgent || pending.size() >= 64u; + }); + batch.swap(pending); + urgent = false; + } + try + { + for (const auto &command : batch) + { + execute(command); + bytes += command.bytes(); + // A batch can hold most of a frame. Returning its space as + // it drains keeps the producer going instead of stopping + // it for the whole batch and then leaving this thread idle. + if (bytes >= kReleaseBytes) + { + { + std::lock_guard lock(mutex); + outstandingBytes -= bytes; + } + bytes = 0u; + progress.notify_all(); + } + } + } + catch (...) + { + std::lock_guard lock(mutex); + error = std::current_exception(); + for (const auto &command : batch) + if (command.op == Op::Prepare) + std::get(command.data).result->fail(error); + for (const auto &command : pending) + if (command.op == Op::Prepare) + std::get(command.data).result->fail(error); + pending.clear(); + outstandingBytes = 0u; + progress.notify_all(); + return; + } + batch.clear(); + { + std::lock_guard lock(mutex); + outstandingBytes -= bytes; + } + progress.notify_all(); + } + } + + static constexpr size_t kReleaseBytes = 64u * 1024u; + + std::unique_ptr backend; + const std::shared_ptr identity = std::make_shared(0); + const size_t capacity; + std::mutex submission, mutex; + std::condition_variable work, progress; + std::vector pending; + size_t outstandingBytes = 0u; + bool urgent = false, stopping = false; + std::exception_ptr error; + int rounding = FE_TONEAREST; + std::thread worker; +}; + +GSThreadedBackend::GSThreadedBackend(std::unique_ptr backend, size_t queueBytes) + : m_impl(std::make_unique(std::move(backend), queueBytes)) {} +GSThreadedBackend::~GSThreadedBackend() = default; + +void GSThreadedBackend::Initialize(uint8_t *vram, uint32_t size) +{ + m_impl->observe([&](auto &b) { b.Initialize(vram, size); }); +} +void GSThreadedBackend::Reset() { m_impl->observe([](auto &b) { b.Reset(); }); } +void GSThreadedBackend::Submit(const GSPrimitiveBatch &batch) +{ + m_impl->enqueue({Impl::Op::Submit, batch}); +} +void GSThreadedBackend::BeginTransfer(const GSTransferCommand &command) +{ + m_impl->enqueue({Impl::Op::Transfer, command}); +} +void GSThreadedBackend::UploadImage(const uint8_t *data, uint32_t size) +{ + std::vector bytes; + if (size != 0u) + { + if (!data) + throw std::invalid_argument("GS upload data is null"); + bytes.assign(data, data + size); + } + m_impl->enqueue({Impl::Op::Upload, std::move(bytes)}); +} +void GSThreadedBackend::Flush() { m_impl->enqueue({Impl::Op::Flush, {}}, true); } +void GSThreadedBackend::TextureFlush() { m_impl->enqueue({Impl::Op::TextureFlush, {}}); } +void GSThreadedBackend::Sync(GSSyncReason reason) +{ + m_impl->observe([&](auto &b) { b.Sync(reason); }); +} +PresentationFrame GSThreadedBackend::Present(const GSPresentationRequest &request) +{ + if (SupportsPreparedPresentation()) + return DisplayPreparedPresentation(PreparePresentation(request)); + return m_impl->observe([&](auto &b) { return b.Present(request); }); +} +bool GSThreadedBackend::SupportsPreparedPresentation() const +{ + return m_impl->backend->SupportsPreparedPresentation(); +} +GSPresentationTicket GSThreadedBackend::PreparePresentation(const GSPresentationRequest &request) +{ + if (!SupportsPreparedPresentation()) return {}; + auto result = std::make_shared(); + auto ticket = std::make_shared(m_impl->identity, result); + ticket->sourceVsyncTick = request.vsyncTick; + m_impl->enqueue({Impl::Op::Prepare, Impl::Prepare{request, std::move(result)}}, true); + return ticket; +} +PresentationFrame GSThreadedBackend::DisplayPreparedPresentation(const GSPresentationTicket &ticket) +{ + const auto *prepared = dynamic_cast(ticket.get()); + if (!prepared || prepared->owner != m_impl->identity) + throw std::invalid_argument("GS presentation ticket belongs to another backend"); + GSPresentationTicket frame; + { + std::unique_lock lock(prepared->result->mutex); + prepared->result->ready.wait(lock, [&] { return prepared->result->completed; }); + if (prepared->result->error) std::rethrow_exception(prepared->result->error); + frame = prepared->result->frame; + } + return m_impl->backend->DisplayPreparedPresentation(frame); +} +void GSThreadedBackend::CancelPreparedPresentations() noexcept +{ + m_impl->backend->CancelPreparedPresentations(); +} +bool GSThreadedBackend::ClearFramebuffer(const GSContext &context, uint32_t rgba) +{ + return m_impl->observe([&](auto &b) { return b.ClearFramebuffer(context, rgba); }); +} +uint32_t GSThreadedBackend::ConsumeLocalToHostBytes(uint8_t *dst, uint32_t size) +{ + return m_impl->observe([&](auto &b) { return b.ConsumeLocalToHostBytes(dst, size); }); +} +uint32_t GSThreadedBackend::ReadVram(uint32_t psm, uint32_t base, uint32_t bw, uint32_t x, uint32_t y) const +{ + return m_impl->observe([&](auto &b) { return b.ReadVram(psm, base, bw, x, y); }); +} +void GSThreadedBackend::WriteVram(uint32_t psm, uint32_t base, uint32_t bw, uint32_t x, uint32_t y, uint32_t value) +{ + m_impl->enqueue({Impl::Op::Write, Impl::Pixel{psm, base, bw, x, y, value}}); +} +void GSThreadedBackend::SnapshotVram(std::vector &out) const +{ + m_impl->observe([&](auto &b) { b.SnapshotVram(out); }); +} +GSTransferSnapshot GSThreadedBackend::GetTransferSnapshot() const +{ + return m_impl->observe([](auto &b) { return b.GetTransferSnapshot(); }); +} diff --git a/ps2xRuntime/src/lib/gs/ps2_gs_memory.cpp b/ps2xRuntime/src/lib/gs/ps2_gs_memory.cpp index fecf9fd2c..1bfe50ad0 100644 --- a/ps2xRuntime/src/lib/gs/ps2_gs_memory.cpp +++ b/ps2xRuntime/src/lib/gs/ps2_gs_memory.cpp @@ -225,6 +225,41 @@ namespace GSMem PixelStorageTraits::Write(PageTableC32, data, bp, bw, x, y, value); } + namespace + { + // Shared body for the 32-bit run writers. Defined here so the storage + // traits and their page table inline into the loop. + template + inline void WriteRun32(const TableT& table, u8* data, u32 bp, u32 bw, u32 dsax, + u32 rowEnd, u32 x, u32 y, const u8* src, u32 count) + { + for (u32 i = 0; i < count; ++i) + { + u32 value = 0; + std::memcpy(&value, src, sizeof(value)); + src += sizeof(value); + PixelStorageTraits::Write(table, data, bp, bw, x, y, value); + if (++x >= rowEnd) + { + x = dsax; + ++y; + } + } + } + } + + void WriteRunCT32(u8* data, u32 bp, u32 bw, u32 dsax, u32 rowEnd, u32 x, u32 y, + const u8* src, u32 count) + { + WriteRun32(PageTableC32, data, bp, bw, dsax, rowEnd, x, y, src, count); + } + + void WriteRunZ32(u8* data, u32 bp, u32 bw, u32 dsax, u32 rowEnd, u32 x, u32 y, + const u8* src, u32 count) + { + WriteRun32(PageTableZ32, data, bp, bw, dsax, rowEnd, x, y, src, count); + } + void WriteCT24(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 value) { PixelStorageTraits::Write(PageTableC32, data, bp, bw, x, y, value); @@ -358,6 +393,66 @@ namespace GSMem { return 0; } + + namespace + { + // Body for the consecutive-x row readers. Along a row only the + // in-page column and the page column move, so the walk is increments; + // the swizzle table lookup is the only per-pixel work left. The + // address mask matches Read()'s wrap guard for regions past 4 MiB. + template + inline void ReadRowImpl(const TableT& table, u8* data, u32 bp, u32 bw, + u32 x, u32 y, u32 count, u8* dst) + { + constexpr u32 kBlocksPerPage = 32; + constexpr u32 kPixelsPerPage = PageW * PageH; + constexpr u32 kAddressMask = u32(MEMORY_SIZE) - BytesPerPixel; + + const u32 basePage = bp / kBlocksPerPage; + const u32 block = bp % kBlocksPerPage; + const u32 pagesPerRow = (bw * 64u) / PageW; + const u32 pageRowBase = basePage + (y / PageH) * pagesPerRow; + const u32 yInPage = y % PageH; + const auto& row = table[block][yInPage]; + + u32 page = pageRowBase + x / PageW; + u32 xInPage = x % PageW; + while (count > 0) + { + const u32 segment = count < (PageW - xInPage) ? count : (PageW - xInPage); + const u32 pageAddress = page * kPixelsPerPage; + for (u32 i = 0; i < segment; ++i) + { + const u32 byteAddress = + ((pageAddress + row[xInPage + i]) * BytesPerPixel) & kAddressMask; + std::memcpy(dst, data + byteAddress, BytesPerPixel); + dst += BytesPerPixel; + } + xInPage += segment; + if (xInPage == PageW) + { + xInPage = 0; + ++page; + } + count -= segment; + } + } + } + + void ReadRowCT32(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 count, u8* dst) + { + ReadRowImpl(PageTableC32, data, bp, bw, x, y, count, dst); + } + + void ReadRowZ32(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 count, u8* dst) + { + ReadRowImpl(PageTableZ32, data, bp, bw, x, y, count, dst); + } + + void ReadRowP8(u8* data, u32 bp, u32 bw, u32 x, u32 y, u32 count, u8* dst) + { + ReadRowImpl(PageTableP8, data, bp, bw, x, y, count, dst); + } } diff --git a/ps2xRuntime/src/lib/ps2_iop_host.cpp b/ps2xRuntime/src/lib/ps2_iop_host.cpp index 8709ed1d7..d9ee52bf6 100644 --- a/ps2xRuntime/src/lib/ps2_iop_host.cpp +++ b/ps2xRuntime/src/lib/ps2_iop_host.cpp @@ -472,6 +472,23 @@ bool PS2IopHostAdapter::hasGuestFunction(uint32_t address) const return m_runtime.hasFunction(address); } +bool PS2IopHostAdapter::hasGuestRpcServer(uint32_t sid) const +{ + uint32_t record = 0u; + { + std::lock_guard lock(g_rpc_mutex); + const auto found = g_rpc_servers.find(sid); + if (found == g_rpc_servers.end()) + { + return false; + } + record = found->second.sd_ptr; + } + // Binding from the EE side leaves a placeholder with no function. + t_SifRpcServerData server{}; + return record != 0u && readGuest(record, &server, sizeof(server)) && server.func != 0u; +} + bool PS2IopHostAdapter::invokeGuestFunction(uint64_t callToken, uint32_t address, uint32_t a0, diff --git a/ps2xRuntime/src/lib/ps2_iop_host.h b/ps2xRuntime/src/lib/ps2_iop_host.h index 48c5b2264..7e529c256 100644 --- a/ps2xRuntime/src/lib/ps2_iop_host.h +++ b/ps2xRuntime/src/lib/ps2_iop_host.h @@ -71,6 +71,7 @@ class PS2IopHostAdapter final : public ps2x::iop::IopHost int32_t memoryCard(const ps2x::iop::MemoryCardRequest &request) override; bool hasGuestFunction(uint32_t address) const override; + bool hasGuestRpcServer(uint32_t sid) const override; bool invokeGuestFunction(uint64_t callToken, uint32_t address, uint32_t a0, diff --git a/ps2xRuntime/src/lib/ps2_iop_transport.h b/ps2xRuntime/src/lib/ps2_iop_transport.h index 988d6b290..6f47b84ad 100644 --- a/ps2xRuntime/src/lib/ps2_iop_transport.h +++ b/ps2xRuntime/src/lib/ps2_iop_transport.h @@ -37,6 +37,11 @@ class PS2IopTransport : ps2x::iop::RpcResult{}; } + [[nodiscard]] static ps2x::iop::NativeIop *native(PS2Runtime *runtime) + { + return runtime && runtime->m_iopSubsystem ? &runtime->m_iopSubsystem->native() : nullptr; + } + static void notifyTransfer( PS2Runtime *runtime, uint8_t *rdram, diff --git a/ps2xRuntime/src/lib/ps2_memory.cpp b/ps2xRuntime/src/lib/ps2_memory.cpp index 7cb2ba463..bf0c88455 100644 --- a/ps2xRuntime/src/lib/ps2_memory.cpp +++ b/ps2xRuntime/src/lib/ps2_memory.cpp @@ -2,6 +2,7 @@ #include "runtime/ps2_address.h" #include "runtime/gs/gs_frontend.h" #include "ps2_log.h" +#include #include #include #include @@ -402,11 +403,38 @@ bool PS2Memory::initialize(size_t ramSize) void PS2Memory::resetEeTimers() noexcept { m_eeTimers = {}; + m_anyEeTimerCued = false; +} + +void PS2Memory::refreshEeTimersCued() noexcept +{ + // Write path only: advanceEeTimers is the hot side and just reads the + // flag. + m_anyEeTimerCued = false; + for (const EeTimer &timer : m_eeTimers) + { + if ((timer.mode & kEeTimerModeCue) != 0u) + { + m_anyEeTimerCued = true; + return; + } + } +} + +uint32_t PS2Memory::takePendingVifInterrupts() noexcept +{ + // Runs after every guest dispatch. The exchange is a locked RMW; the + // pending set is almost always empty, so the plain load answers first. + if (m_pendingVifInterrupts.load(std::memory_order_acquire) == 0u) + { + return 0u; + } + return m_pendingVifInterrupts.exchange(0u, std::memory_order_acq_rel); } uint32_t PS2Memory::advanceEeTimers(uint64_t eeCycles) noexcept { - if (eeCycles == 0u) + if (eeCycles == 0u || !m_anyEeTimerCued) { return 0u; } @@ -421,11 +449,30 @@ uint32_t PS2Memory::advanceEeTimers(uint64_t eeCycles) noexcept } const uint64_t clockHz = kEeTimerClockHz[timer.mode & kEeTimerModeClksMask]; - const uint64_t wholeSeconds = eeCycles / kEeClockHz; - const uint64_t remainingCycles = eeCycles % kEeClockHz; - const uint64_t scaled = remainingCycles * clockHz + timer.clockRemainder; - const uint64_t ticks = wholeSeconds * clockHz + scaled / kEeClockHz; - timer.clockRemainder = scaled % kEeClockHz; + uint64_t ticks = 0u; + if (eeCycles < kEeClockHz) + { + // The hot case by far: this runs on every guest safe point with a + // handful of cycles, so the two 64-bit divisions the general form + // needs are pure overhead. Below one EE second wholeSeconds is + // zero, and a sub-tick accumulation needs no division at all. + const uint64_t scaled = eeCycles * clockHz + timer.clockRemainder; + if (scaled < kEeClockHz) + { + timer.clockRemainder = scaled; + continue; + } + ticks = scaled / kEeClockHz; + timer.clockRemainder = scaled % kEeClockHz; + } + else + { + const uint64_t wholeSeconds = eeCycles / kEeClockHz; + const uint64_t remainingCycles = eeCycles % kEeClockHz; + const uint64_t scaled = remainingCycles * clockHz + timer.clockRemainder; + ticks = wholeSeconds * clockHz + scaled / kEeClockHz; + timer.clockRemainder = scaled % kEeClockHz; + } if (ticks == 0u) { continue; @@ -476,6 +523,10 @@ uint32_t PS2Memory::advanceEeTimers(uint64_t eeCycles) noexcept uint64_t PS2Memory::cyclesUntilNextEeTimerInterrupt() const noexcept { + if (!m_anyEeTimerCued) + { + return std::numeric_limits::max(); + } uint64_t nearest = std::numeric_limits::max(); for (const EeTimer &timer : m_eeTimers) { @@ -1123,6 +1174,7 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value) { timer.clockRemainder = 0u; } + refreshEeTimersCued(); break; } case kEeTimerCompareOffset: @@ -1315,7 +1367,13 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value) uint32_t asr1 = m_ioRegisters[channelBase + 0x50]; uint32_t asp = (chcr >> 4) & 0x3u; const bool tieEnabled = (chcr & (1u << 7)) != 0u; - const int kMaxChainTags = 4096; + const bool transferVifTagData = (chcr & (1u << 6)) != 0u && + (channelBase == 0x10009000u || channelBase == 0x10008000u); + // Brent's cycle check includes the return stack, so finite CALL + // chains can reuse a subchain without an arbitrary tag limit. + std::array cycleCheckpoint{tagAddr, asr0, asr1, asp}; + uint64_t cyclePower = 1u; + uint64_t cycleDistance = 0u; std::vector chainBuf; auto appendData = [&](uint32_t srcAddr, uint32_t qwCount) @@ -1353,7 +1411,7 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value) } }; - auto appendCompactVif1TagData = [&](uint32_t localTagAddr, uint32_t qwCount) + auto appendVifTagData = [&](uint32_t localTagAddr) { uint32_t tagPhys = 0u; const bool tagScratch = isScratchpad(localTagAddr); @@ -1364,15 +1422,13 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value) if (tagPhys + 16u > localMax) return; - // VIF packet helpers embed 8 bytes of VIF stream in the DMAtag's upper half. + // With TTE, every tag supplies two VIFcodes before its payload. chainBuf.insert(chainBuf.end(), localBase + tagPhys + 8u, localBase + tagPhys + 16u); - appendData(localTagAddr + 16u, qwCount); }; - int tagsProcessed = 0; uint32_t lastTagUpper = (chcr >> 16) & 0xFFFFu; - while (tagsProcessed < kMaxChainTags) + while (true) { const uint32_t currentTagAddr = tagAddr; const bool tagInSPR = isScratchpad(tagAddr); @@ -1407,7 +1463,6 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value) const bool irq = ((tag >> 31) & 0x1ull) != 0ull; uint32_t addr = static_cast((tag >> 32) & 0x7FFFFFFF); lastTagUpper = static_cast((tag >> 16) & 0xFFFFu); - ++tagsProcessed; uint32_t dataAddr = 0; bool hasPayload = (tagQwc > 0); @@ -1477,23 +1532,30 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value) break; } - const bool compactVifLocalTag = - (channelBase == 0x10009000u || channelBase == 0x10008000u) && - (id == 1u || id == 2u || id == 5u || id == 6u || id == 7u); - if (compactVifLocalTag) - appendCompactVif1TagData(currentTagAddr, 0u); + if (transferVifTagData) + appendVifTagData(currentTagAddr); if (hasPayload) { - if (compactVifLocalTag) - appendData(currentTagAddr + 16u, tagQwc); - else - appendData(dataAddr, tagQwc); + appendData(dataAddr, tagQwc); } if (irq && tieEnabled) endChain = true; if (endChain) break; + const std::array nextState{tagAddr, asr0, asr1, asp}; + if (nextState == cycleCheckpoint) + { + RUNTIME_ERROR("[DMA] cyclic source chain at 0x" << std::hex + << tagAddr << " on channel 0x" << channelBase << std::dec << '\n'); + break; + } + if (++cycleDistance == cyclePower) + { + cycleCheckpoint = nextState; + cyclePower *= 2u; + cycleDistance = 0u; + } } m_ioRegisters[channelBase + 0x30] = tagAddr; diff --git a/ps2xRuntime/src/lib/ps2_native_iop.cpp b/ps2xRuntime/src/lib/ps2_native_iop.cpp new file mode 100644 index 000000000..41b01d5f9 --- /dev/null +++ b/ps2xRuntime/src/lib/ps2_native_iop.cpp @@ -0,0 +1,86 @@ +#include "runtime/ps2_native_iop.h" + +#include "ps2_iop_transport.h" +#include "raylib.h" + +#include +#include +#include +#include + +namespace ps2_native_iop +{ + namespace + { + // raylib's stream callback carries no user pointer. + std::atomic g_source{nullptr}; + AudioStream g_stream{}; + bool g_streaming = false; + float g_volume = 1.0f; + + void fill(void *buffer, unsigned int frames) + { + if (ps2x::iop::NativeIop *source = g_source.load(std::memory_order_acquire)) + source->render(static_cast(buffer), frames); + // PS2X_AUDIO_DUMP=file keeps a copy of everything played, as raw + // 48 kHz stereo s16, to compare against a reference offline. + static FILE *dump = [] { + const char *path = std::getenv("PS2X_AUDIO_DUMP"); + return path && *path ? std::fopen(path, "wb") : nullptr; + }(); + if (dump) + std::fwrite(buffer, 4u, frames, dump); + } + } + + void setModules(PS2Runtime &runtime, std::vector names) + { + if (ps2x::iop::NativeIop *native = PS2IopTransport::native(&runtime)) + native->setModules(std::move(names)); + } + + void setVolume(float volume) + { + g_volume = volume; +#if !defined(PLATFORM_VITA) + if (g_streaming) + SetAudioStreamVolume(g_stream, g_volume); +#endif + } + + void startAudio(PS2Runtime &runtime) + { +#if !defined(PLATFORM_VITA) + ps2x::iop::NativeIop *native = PS2IopTransport::native(&runtime); + if (g_streaming || !native || !IsAudioDeviceReady()) + return; + // 1024 frames is about 21 ms, and the IOP renders a buffer in well under that. + SetAudioStreamBufferSizeDefault(1024); + g_stream = LoadAudioStream(48000u, 16u, 2u); + if (!IsAudioStreamValid(g_stream)) + { + std::cerr << "[iop] could not open an audio stream; the native IOP runs silently" << std::endl; + return; + } + g_source.store(native, std::memory_order_release); + SetAudioStreamCallback(g_stream, &fill); + SetAudioStreamVolume(g_stream, g_volume); + PlayAudioStream(g_stream); + g_streaming = true; +#else + (void)runtime; +#endif + } + + void stopAudio() + { +#if !defined(PLATFORM_VITA) + if (!g_streaming) + return; + // Unloading takes the mixer lock, so no callback is running after it. + UnloadAudioStream(g_stream); + g_source.store(nullptr, std::memory_order_release); + g_streaming = false; +#endif + } +} diff --git a/ps2xRuntime/src/lib/ps2_pad.cpp b/ps2xRuntime/src/lib/ps2_pad.cpp index 8590f1424..c8e77d18f 100644 --- a/ps2xRuntime/src/lib/ps2_pad.cpp +++ b/ps2xRuntime/src/lib/ps2_pad.cpp @@ -1,11 +1,29 @@ #include "runtime/ps2_pad.h" +#include "runtime/ps2_pad_host.h" #include "ps2_host_backend.h" +#include +#include +#include +#include +#include +#include #include +#include +#include +#include +#include +#include namespace { constexpr uint8_t kPadAnalogMarker = 0x73; constexpr uint8_t kPadStickCenter = 0x80; + constexpr uint32_t kPadStickNeutral = 0x80808080u; + constexpr int kGamepad = 0; + + // Drop a latch the guest never came back for, so a tap during a long + // non-polling stretch (loading, cutscene) does not fire much later. + constexpr uint64_t kLatchTimeoutMs = 1000u; constexpr uint16_t PAD_LEFT = 0x0080u; constexpr uint16_t PAD_DOWN = 0x0040u; @@ -23,6 +41,369 @@ namespace constexpr uint16_t PAD_L1 = 0x0400u; constexpr uint16_t PAD_R2 = 0x0200u; constexpr uint16_t PAD_L2 = 0x0100u; + + // Published by the render thread in ps2PadPollHost(), consumed on the EE thread. + std::atomic g_hostPolled{false}; + std::atomic g_held{0u}; // active-high PAD_* mask + std::atomic g_latched{0u}; // press edges not consumed yet + std::atomic g_sticks{kPadStickNeutral}; // packed rx,ry,lx,ly + std::atomic g_latchStampMs{0u}; + + struct KeyBinding + { + int key; + uint16_t mask; + }; + + constexpr KeyBinding kKeyBindings[] = { + {KEY_UP, PAD_UP}, + {KEY_DOWN, PAD_DOWN}, + {KEY_LEFT, PAD_LEFT}, + {KEY_RIGHT, PAD_RIGHT}, + {KEY_X, PAD_CROSS}, {KEY_SPACE, PAD_CROSS}, + {KEY_C, PAD_CIRCLE}, {KEY_ESCAPE, PAD_CIRCLE}, + {KEY_Z, PAD_SQUARE}, {KEY_KP_0, PAD_SQUARE}, + {KEY_V, PAD_TRIANGLE}, {KEY_KP_1, PAD_TRIANGLE}, + {KEY_Q, PAD_L1}, + {KEY_E, PAD_R1}, + {KEY_LEFT_SHIFT, PAD_L2}, + {KEY_RIGHT_SHIFT, PAD_R2}, + {KEY_ENTER, PAD_START}, + {KEY_TAB, PAD_SELECT}, + {KEY_R, PAD_L3}, + {KEY_F, PAD_R3}, + }; + + struct PadBinding + { + int button; + uint16_t mask; + }; + + constexpr PadBinding kPadBindings[] = { + {GAMEPAD_BUTTON_LEFT_FACE_UP, PAD_UP}, + {GAMEPAD_BUTTON_LEFT_FACE_DOWN, PAD_DOWN}, + {GAMEPAD_BUTTON_LEFT_FACE_LEFT, PAD_LEFT}, + {GAMEPAD_BUTTON_LEFT_FACE_RIGHT, PAD_RIGHT}, + {GAMEPAD_BUTTON_RIGHT_FACE_DOWN, PAD_CROSS}, + {GAMEPAD_BUTTON_RIGHT_FACE_RIGHT, PAD_CIRCLE}, + {GAMEPAD_BUTTON_RIGHT_FACE_LEFT, PAD_SQUARE}, + {GAMEPAD_BUTTON_RIGHT_FACE_UP, PAD_TRIANGLE}, + {GAMEPAD_BUTTON_LEFT_TRIGGER_1, PAD_L1}, + {GAMEPAD_BUTTON_RIGHT_TRIGGER_1, PAD_R1}, + {GAMEPAD_BUTTON_LEFT_TRIGGER_2, PAD_L2}, + {GAMEPAD_BUTTON_RIGHT_TRIGGER_2, PAD_R2}, + {GAMEPAD_BUTTON_MIDDLE_RIGHT, PAD_START}, + {GAMEPAD_BUTTON_MIDDLE_LEFT, PAD_SELECT}, + {GAMEPAD_BUTTON_LEFT_THUMB, PAD_L3}, + {GAMEPAD_BUTTON_RIGHT_THUMB, PAD_R3}, + }; + + uint8_t axisToByte(float axis) + { + if (axis > -0.125f && axis < 0.125f) + return kPadStickCenter; + const float mapped = 128.0f + axis * (axis < 0.0f ? 128.0f : 127.0f); + return static_cast(mapped < 0.0f ? 0.0f : (mapped > 255.0f ? 255.0f : mapped)); + } + + uint8_t keyboardAxis(int negative, int positive, uint8_t fallback) + { + const bool low = IsKeyDown(negative); + const bool high = IsKeyDown(positive); + return low || high ? (low == high ? kPadStickCenter : low ? 0u : 255u) : fallback; + } + + uint64_t nowMs() + { + using namespace std::chrono; + return static_cast( + duration_cast(steady_clock::now().time_since_epoch()).count()); + } + + uint32_t keyMask(int key) + { + uint32_t mask = 0u; + for (const KeyBinding &binding : kKeyBindings) + { + if (binding.key == key) + { + mask |= binding.mask; + } + } + return mask; + } + + // drainQueue must only be true on the render thread: GetKeyPressed() mutates + // raylib's queue, while every other call here is a plain read. + void sampleHost(bool drainQueue, uint32_t &held, uint32_t &pressed, uint32_t &sticks) + { + held = 0u; + pressed = 0u; + sticks = kPadStickNeutral; + + if (IsGamepadAvailable(kGamepad)) + { + for (const PadBinding &binding : kPadBindings) + { + if (IsGamepadButtonDown(kGamepad, binding.button)) + { + held |= binding.mask; + } + if (IsGamepadButtonPressed(kGamepad, binding.button)) + { + pressed |= binding.mask; + } + } + + const uint8_t rx = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_RIGHT_X)); + const uint8_t ry = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_RIGHT_Y)); + const uint8_t lx = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_LEFT_X)); + const uint8_t ly = axisToByte(GetGamepadAxisMovement(kGamepad, GAMEPAD_AXIS_LEFT_Y)); + sticks = (static_cast(rx) << 24) | (static_cast(ry) << 16) | + (static_cast(lx) << 8) | static_cast(ly); + } + + for (const KeyBinding &binding : kKeyBindings) + { + if (IsKeyDown(binding.key)) + { + held |= binding.mask; + } + } + + const uint8_t rx = keyboardAxis(KEY_J, KEY_L, static_cast(sticks >> 24u)); + const uint8_t ry = keyboardAxis(KEY_I, KEY_K, static_cast(sticks >> 16u)); + const uint8_t lx = keyboardAxis(KEY_A, KEY_D, static_cast(sticks >> 8u)); + const uint8_t ly = keyboardAxis(KEY_W, KEY_S, static_cast(sticks)); + sticks = (uint32_t(rx) << 24u) | (uint32_t(ry) << 16u) | (uint32_t(lx) << 8u) | ly; + + if (drainQueue) + { + // raylib queues every GLFW press, including one released again inside + // the same poll, which IsKeyPressed() would already have missed. + for (int key = GetKeyPressed(); key != 0; key = GetKeyPressed()) + { + pressed |= keyMask(key); + } + } + } +} + +namespace +{ + // Scripted input. DQ8_PAD_SCRIPT is either a file path or the script text + // itself, entries separated by ';' or newlines: + // + //