From 49eb8917f09219473fa4522471985bd012086257 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sat, 22 Aug 2026 23:56:45 +0700 Subject: [PATCH 01/16] Record credential-read capability in RC contract XAIOS_CAP_CREDENTIAL_READ has shipped in kernel/include/xaios/syscall.h since the DNS/SSH interoperability work, and gates fs_open on /etc/xaios_ssh_client_identity plus the async remote-login child grant. It was never added to the frozen release-candidate contract, so validate_syscall_abi reported "contract missing source capabilities" and make qemu-abi-contract failed on main. Add the bit to the contract capability list so it matches source. Verified: python3 tests/scripts/qemu-abi-contract.py reports status=pass with all eight checks green, on macOS and on the Debian 13 x86_64 box. Co-Authored-By: Claude Opus 5 --- contracts/qemu-rc-v1.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/contracts/qemu-rc-v1.json b/contracts/qemu-rc-v1.json index ee9fe06e..2e41f761 100644 --- a/contracts/qemu-rc-v1.json +++ b/contracts/qemu-rc-v1.json @@ -95,7 +95,8 @@ {"bit": 134217728, "name": "XAIOS_CAP_STORAGE_TRIM"}, {"bit": 268435456, "name": "XAIOS_CAP_MODEL_STAGE"}, {"bit": 536870912, "name": "XAIOS_CAP_MODEL_ACTIVATE"}, - {"bit": 1073741824, "name": "XAIOS_CAP_CONSOLE"} + {"bit": 1073741824, "name": "XAIOS_CAP_CONSOLE"}, + {"bit": 2147483648, "name": "XAIOS_CAP_CREDENTIAL_READ"} ] }, "control_protocol": { From ed6253bebb8540f6dd74dcd3dcce91e8d93c7f5d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sat, 22 Aug 2026 23:56:52 +0700 Subject: [PATCH 02/16] Give the Compile Check job its libc build inputs compile-check gained a libc prerequisite in 47ff4f1, but the CI job was never updated for it. scripts/build-libc.sh requires llvm-ar, meson and ninja, and picolibc must be present, so the job failed on main with "required tool not found: llvm-ar". The checkout also omitted submodules, which would have failed next with a missing Picolibc submodule. Install llvm, meson, ninja-build and python3 alongside clang and lld, and check out submodules recursively, matching the hosted-c99-libc job that already builds the same target successfully. Verified: make compile-check exits 0 with "All freestanding C files compiled clean" on Debian 13 x86_64 with exactly this tool set. Co-Authored-By: Claude Opus 5 --- .github/workflows/ci.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 45267302..c243b24e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -45,11 +45,13 @@ jobs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v6 + with: + submodules: recursive - name: Install toolchain run: | sudo apt-get update - sudo apt-get install -y clang lld + sudo apt-get install -y clang lld llvm meson ninja-build python3 - name: Compile-check kernel C files run: make compile-check From 7bc2115bfd18e4fa38478d41a97ef617ead0669c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 00:08:58 +0700 Subject: [PATCH 03/16] Fix the hosted-test build under glibc and older clang make hosted-test could not build on Linux, so the Hosted Engine & Model v2 job has failed on main for at least six consecutive runs. Two independent causes, both in host-only compiles of vendored third-party code: BearSSL's sysrng.c calls getentropy(), which glibc declares in only under _DEFAULT_SOURCE. HOST_CFLAGS uses -std=c99, which sets __STRICT_ANSI__ and hides the declaration, so the call became an implicit declaration and -Werror rejected it. macOS declares it regardless, which is why this only ever failed on Linux. Define _DEFAULT_SOURCE for that compile. openbsd-compat's bcrypt_pbkdf.c uses __attribute__((__nonstring__)), which clang did not recognize before version 21, so -Wunknown-attributes fired under -Werror. scripts/build-image.sh already compiles these same two vendored files with -Wno-unknown-attributes; apply the same flag to the hosted test compile rather than diverging from upstream source. Verified: make hosted-test now runs to completion and exits 0 on Debian 13 x86_64 with clang 19, ending at the model-volume reader tests. Co-Authored-By: Claude Opus 5 --- Makefile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Makefile b/Makefile index f95ac225..55b8d957 100644 --- a/Makefile +++ b/Makefile @@ -428,7 +428,7 @@ hosted-test: engine-cli -o build/hosted/test-sftp-large ./build/hosted/test-sftp-large python3 tests/scripts/generate-dnssec-fixture.py build/hosted/dnssec_fixture.h - $(HOST_CC) $(HOST_CFLAGS) \ + $(HOST_CC) $(HOST_CFLAGS) -D_DEFAULT_SOURCE \ -Ikernel/include -Iuserspace/include -Iuserspace/sshd -Ithird_party/bearssl/inc \ -Ithird_party/bearssl/src -Ibuild/hosted \ kernel/net/dns.c kernel/net/dnssec.c kernel/net/ipv4.c \ @@ -447,7 +447,7 @@ hosted-test: engine-cli ssh-keygen -q -t ed25519 -N '' -f build/hosted/id-ed25519 ssh-keygen -q -t ed25519 -N xaios-test-passphrase \ -f build/hosted/id-ed25519-encrypted - $(HOST_CC) $(HOST_CFLAGS) -DXAIOS_IDENTITY_HOSTED=1 \ + $(HOST_CC) $(HOST_CFLAGS) -DXAIOS_IDENTITY_HOSTED=1 -Wno-unknown-attributes \ -Iuserspace/include -Iuserspace/sshd -Ithird_party/openbsd-compat \ userspace/sshd/ssh_identity.c userspace/sshd/ssh_crypto.c \ userspace/sshd/tweetnacl_subset.c \ From b6af1259c23785baa4bcf86abb06dc6ce4bb5b1f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 00:49:46 +0700 Subject: [PATCH 04/16] Scale the FreeBSD outbound budget to the emulated architecture The bidirectional suite gave the outbound SSH conversation 600 seconds for an x86_64 guest and 60 for an aarch64 guest. That mapping only holds on an AArch64 development host, where aarch64 is native and x86_64 is emulated. On an x86_64 CI runner it inverts: the aarch64 guest is the cross-emulated, order-of-magnitude-slower one, and it is the one handed 60 seconds. The job has failed on main with "timed out waiting for freebsd-outbound-ssh-ok" ever since, while the x86_64 variant passes on an x86_64 host. Derive the default from the host/guest pairing instead. Cross-architecture emulation gets the large budget; native pairings keep today's values, so no configuration that currently passes is given less time: arm64 host + aarch64 guest -> 60s (unchanged, verified below) arm64 host + x86_64 guest -> 600s (unchanged) x86_64 host + aarch64 guest -> 600s (was 60s; the CI failure) x86_64 host + x86_64 guest -> 600s (unchanged) An unrecognized host architecture falls back to the large budget. XAIOS_OUTBOUND_TIMEOUT still overrides everything. Verified: the aarch64 suite passes end to end on an Apple Silicon host with this change, 19 of 19 checks, including the outbound SSH, SCP recursive upload/download, ProxyJump and agent-forwarding legs. Co-Authored-By: Claude Opus 5 --- .../qemu-freebsd-bidirectional-suite.py | 27 ++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/tests/scripts/qemu-freebsd-bidirectional-suite.py b/tests/scripts/qemu-freebsd-bidirectional-suite.py index c3565ddb..1b7419ea 100644 --- a/tests/scripts/qemu-freebsd-bidirectional-suite.py +++ b/tests/scripts/qemu-freebsd-bidirectional-suite.py @@ -7,6 +7,7 @@ import hashlib import json import os +import platform import re from pathlib import Path import secrets @@ -40,6 +41,30 @@ }, } +# Outbound SSH budget when QEMU emulates the host's own architecture. The +# aarch64 value is measured on an Apple Silicon host; x86_64 keeps the larger +# value because it has only ever been run cross-emulated, so there is no +# native measurement to justify shrinking it. +NATIVE_OUTBOUND_TIMEOUT = {"aarch64": "60", "x86_64": "600"} +# Budget when QEMU must translate a foreign ISA, which is roughly an order of +# magnitude slower than emulating the host architecture. +CROSS_OUTBOUND_TIMEOUT = "600" + +HOST_ARCHITECTURES = {"arm64": "aarch64", "aarch64": "aarch64", + "amd64": "x86_64", "x86_64": "x86_64"} + + +def default_outbound_timeout(architecture: str) -> str: + # The budget depends on the host/guest pairing, not on the guest + # architecture alone. Keying it off the guest only held on an AArch64 + # development host, where aarch64 happened to be native and x86_64 + # happened to be emulated; on an x86_64 CI runner that mapping inverts + # and leaves the cross-emulated aarch64 guest with the 60 second budget. + host = HOST_ARCHITECTURES.get(platform.machine().lower()) + if host is None or host != architecture: + return CROSS_OUTBOUND_TIMEOUT + return NATIVE_OUTBOUND_TIMEOUT[architecture] + def reserve_port(socket_type: int) -> int: with socket.socket(socket.AF_INET, socket_type) as sock: @@ -605,7 +630,7 @@ def main() -> int: "--timeout", os.environ.get( "XAIOS_OUTBOUND_TIMEOUT", - "600" if architecture == "x86_64" else "60", + default_outbound_timeout(architecture), ), ], 900, From 8ec68fe6f9b969bc33fd999a405a24af25ea6c92 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 09:11:58 +0700 Subject: [PATCH 05/16] Emit captured console output once, to its own session klog_console_write copied each byte into the active capture buffer and also wrote it to the UART. The shell then appended that same captured buffer to the command's output, so on the local console -- where the UART is the user's terminal -- every transient application printed its output twice: admin@xaios:/$ df Filesystem Size Used Avail Capacity Mounted on MutableFS 2048K 62K 1986K 4% / Filesystem Size Used Avail Capacity Mounted on <-- repeated MutableFS 2048K 62K 1986K 4% / Over SSH the UART is not the client's terminal, so the duplicate was invisible to every interoperability gate, all of which drive XAIOS over SSH. Output produced while a session is capturing belongs to that session: the capturing caller relays it to its own terminal exactly once. Skip the UART echo while a capture is active. This also stops publishing the output of remote SSH commands on the machine's physical serial port. Boot is unaffected: no capture is active during startup self-tests. Verified: make qemu-smoke and qemu-local-console-gate pass, and ls, ps, df, stat and cat each print once on AArch64 and x86_64. Co-Authored-By: Claude Opus 5 --- kernel/core/klog.c | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/kernel/core/klog.c b/kernel/core/klog.c index 7b111c88..e05b33e9 100644 --- a/kernel/core/klog.c +++ b/kernel/core/klog.c @@ -108,13 +108,20 @@ void klog_console_set_log_output(uint32_t enabled) { void klog_console_write(const char *message, uint64_t length) { if (message == 0 || length == 0U) return; xaios_spin_lock(&g_klog_lock); + /* Console output produced while a session is capturing belongs to that + session: the capturing caller relays it to its own terminal exactly once. + Echoing it to the UART as well printed every transient application's + output twice on the local console, and published the output of remote + SSH commands on the physical serial port. */ + int capturing = g_console_capture_depth != 0U; for (uint64_t i = 0U; i < length; ++i) { - if (g_console_capture_depth != 0U) { + if (capturing) { xaios_console_capture_t *capture = &g_console_captures[g_console_capture_depth - 1U]; if (capture->length < capture->capacity) { capture->buffer[capture->length++] = message[i]; } + continue; } if (message[i] == '\n') uart_putc('\r'); uart_putc(message[i]); From 39fe6335177c6b3103296f3f6c0804b2fa224661 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 09:11:58 +0700 Subject: [PATCH 06/16] Make xaiosctl a working command instead of a help entry The shell's help text listed xaiosctl, and both sysinfo and status told the user to "use xaiosctl status", but the command was not in the shell's app table at all, so it always failed: admin@xaios:/$ xaiosctl status xaios: xaiosctl: command not found /bin/xaiosctl was also not a CLI. It was declared int main(void) and ran a fixed control-protocol self-test suite, so simply registering it would have run the test harness rather than the requested operation. Give it an argument path: with arguments it forwards the command line to the control protocol and prints the rendered response; without arguments it stays the existing self-test, so boot and gate evidence is unchanged. Output goes to the console rather than the kernel log, because the shell relays captured console bytes verbatim while log lines are filtered to this application's path prefix. Commands run under the observer role that xaios_control_run already defaults to, so read-only operations succeed and privileged ones are refused by the control plane. The shell grants no admin capability. Verified on AArch64 and x86_64: xaiosctl status and xaiosctl version return real output, and the boot self-test markers are unchanged. Co-Authored-By: Claude Opus 5 --- kernel/runtime/remote_login.c | 6 ++++ userspace/apps/xaiosctl.c | 60 ++++++++++++++++++++++++++++++++++- 2 files changed, 65 insertions(+), 1 deletion(-) diff --git a/kernel/runtime/remote_login.c b/kernel/runtime/remote_login.c index 660de810..29f70ed4 100644 --- a/kernel/runtime/remote_login.c +++ b/kernel/runtime/remote_login.c @@ -3918,6 +3918,12 @@ static const remote_app_definition_t g_remote_apps[] = { XAIOS_CAP_CONTROL_QUERY), REMOTE_TERMINAL_APP("pong", "/bin/pong", XAIOS_CAP_CONSOLE | XAIOS_CAP_EXIT), + /* Operator control surface. Arguments are forwarded verbatim to the + control protocol, which authorizes them under the observer role. */ + {"xaiosctl", "/bin/xaiosctl", + XAIOS_CAP_CONSOLE | XAIOS_CAP_LOG | XAIOS_CAP_EXIT | XAIOS_CAP_TIME | + XAIOS_CAP_CONTROL_QUERY | XAIOS_CAP_STORAGE_READ, + 1U, 0U, 0U}, REMOTE_APP("sysinfo", "/bin/sysinfo", XAIOS_CAP_LOG | XAIOS_CAP_EXIT | XAIOS_CAP_TIME), REMOTE_APP("systest", "/bin/systest", diff --git a/userspace/apps/xaiosctl.c b/userspace/apps/xaiosctl.c index a2645112..d870b6c3 100644 --- a/userspace/apps/xaiosctl.c +++ b/userspace/apps/xaiosctl.c @@ -36,6 +36,60 @@ static int run_checked(const char *command, int expected_result, return 0; } +static u64 text_length(const char *text) { + u64 length = 0ULL; + while (text != 0 && text[length] != '\0') ++length; + return length; +} + +/* Forward a shell command line to the control protocol and print the rendered + response. Runs with the observer role, so read-only operations succeed and + privileged ones are refused by the control plane rather than by this tool. */ +static int run_control_command(int argc, char **argv) { + char command[XAIOS_CONTROL_MAX_REQUEST_BYTES]; + char output[XAIOS_CONTROL_MAX_RESPONSE_BYTES]; + u64 used = 0ULL; + u64 output_size = 0ULL; + int result; + + static const char prefix[] = "xaiosctl"; + u64 prefix_length = sizeof(prefix) - 1ULL; + if (prefix_length >= sizeof(command)) return 1; + copy_bytes(command, prefix, prefix_length); + used = prefix_length; + + for (int i = 1; i < argc; ++i) { + u64 length = text_length(argv[i]); + if (length == 0ULL) continue; + if (used + 1ULL + length >= sizeof(command)) { + xaios_log("/bin/xaiosctl: command line exceeds control request limit\n"); + return 1; + } + command[used++] = ' '; + copy_bytes(command + used, argv[i], length); + used += length; + } + command[used] = '\0'; + + result = xaios_control_run(command, output, sizeof(output), &output_size); + /* Operator output goes to the console, not the kernel log: the shell relays + captured console bytes verbatim, whereas log lines are filtered down to + those prefixed with this application's path. */ + if (output_size != 0ULL) { + if (output_size > sizeof(output)) output_size = sizeof(output); + if (xaios_console_write(output, output_size) != (int)output_size) return 1; + if (output[output_size - 1ULL] != '\n') { + if (xaios_console_write("\n", 1ULL) != 1) return 1; + } + } + if (result < 0) { + static const char failed[] = "xaiosctl: control command rejected\n"; + (void)xaios_console_write(failed, sizeof(failed) - 1ULL); + return 1; + } + return result == 0 ? 0 : result; +} + static int protocol_negative_tests(void) { xaios_control_request_header_user_t request; xaios_control_response_header_user_t response_header; @@ -71,7 +125,7 @@ static int protocol_negative_tests(void) { : -1; } -int main(void) { +int main(int argc, char **argv) { static const struct command_test { const char *human; const char *json; @@ -115,6 +169,10 @@ int main(void) { "\"staging_writable\":1", 0}, }; + /* With arguments this is the operator CLI; without them it stays the boot + and gate self-test, so the existing startup evidence is unchanged. */ + if (argc > 1) return run_control_command(argc, argv); + xaios_log("/bin/xaiosctl: starting control protocol client tests\n"); for (u64 i = 0; i < sizeof(tests) / sizeof(tests[0]); ++i) { if (run_checked(tests[i].human, tests[i].expected_result, From 8d72128b7532dc1450b72ffe4f7c3d0c0ae0d6b1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 09:12:16 +0700 Subject: [PATCH 07/16] Clear analyser findings in mlkem self-test and dead stores Three findings from a clang --analyze pass, none of which alters behaviour: ssh_mlkem768_self_test computed its XOR difference over sender_secret and receiver_secret unconditionally, including when keypair or encapsulation had failed and neither buffer was written. The result was correctly discarded by the result == 0 guard, so the outcome was right, but reading uninitialised stack is undefined and would trip MemorySanitizer the moment it is enabled. Compare only on the success path. mutable_fs open() re-read the node after a truncate into a variable nothing subsequently used. Removed; the CREATE branch's lookup above it is still live, feeding the truncate condition. git_workspace diff computed a line cap from XAIOS_GIT_WORKSPACE_DIFF_MAX_LINES and never read it, so the limit has never applied -- the loop runs on the unclamped counts and the emitted diff is bounded by hunk_capacity instead. The dead computation is removed rather than switched on: the cap is 32 lines, so enforcing it now would begin truncating any diff of a longer file, and that is a behaviour change that wants its own decision. Verified: make compile-check, docs-check, qemu-abi-contract, qemu-smoke and qemu-local-console-gate all pass. Co-Authored-By: Claude Opus 5 --- kernel/fs/mutable_fs.c | 1 - kernel/runtime/git_workspace.c | 8 ++++---- userspace/sshd/ssh_mlkem.c | 12 ++++++++---- 3 files changed, 12 insertions(+), 9 deletions(-) diff --git a/kernel/fs/mutable_fs.c b/kernel/fs/mutable_fs.c index b15e1d8f..33cb9155 100644 --- a/kernel/fs/mutable_fs.c +++ b/kernel/fs/mutable_fs.c @@ -1968,7 +1968,6 @@ int64_t mutable_fs_open(const char *path, uint32_t flags) { if (write_file(normalized, 0, 0) != XAIOS_OK) { return (int64_t)XAIOS_ERR_IO; } - node = find_node(normalized, 0); } for (uint32_t i = 0; i < MFS_MAX_OPEN_FILES; ++i) { diff --git a/kernel/runtime/git_workspace.c b/kernel/runtime/git_workspace.c index 6e3c393a..9c7eb495 100644 --- a/kernel/runtime/git_workspace.c +++ b/kernel/runtime/git_workspace.c @@ -532,10 +532,10 @@ xaios_status_t git_workspace_compute_diff(const char *old_text, uint32_t old_lines = count_lines(old_text, old_bytes); uint32_t new_lines = count_lines(new_text, new_bytes); - uint32_t max_lines = old_lines > new_lines ? old_lines : new_lines; - if (max_lines > XAIOS_GIT_WORKSPACE_DIFF_MAX_LINES) { - max_lines = XAIOS_GIT_WORKSPACE_DIFF_MAX_LINES; - } + /* The emitted diff is bounded by hunk_capacity below. A line cap was + computed here from XAIOS_GIT_WORKSPACE_DIFF_MAX_LINES but never read, so + it never applied; the dead computation is removed rather than switched on, + because enforcing it would start truncating any diff past 32 lines. */ uint32_t oi = 0; uint32_t ni = 0; diff --git a/userspace/sshd/ssh_mlkem.c b/userspace/sshd/ssh_mlkem.c index 60314a02..968496f2 100644 --- a/userspace/sshd/ssh_mlkem.c +++ b/userspace/sshd/ssh_mlkem.c @@ -39,10 +39,14 @@ int ssh_mlkem768_self_test(void) { result = ssh_mlkem768_encapsulate(ciphertext, sender_secret, public_key); if (result == 0) result = ssh_mlkem768_decapsulate(receiver_secret, ciphertext, secret_key); - uint8_t difference = 0U; - for (uint32_t i = 0U; i < sizeof(sender_secret); ++i) - difference |= sender_secret[i] ^ receiver_secret[i]; - if (result == 0 && difference != 0U) result = -1; + /* Only compare once both secrets have been written: on a keypair or + encapsulation failure they remain uninitialised stack. */ + if (result == 0) { + uint8_t difference = 0U; + for (uint32_t i = 0U; i < sizeof(sender_secret); ++i) + difference |= sender_secret[i] ^ receiver_secret[i]; + if (difference != 0U) result = -1; + } ssh_mem_zero(secret_key, sizeof(secret_key)); ssh_mem_zero(sender_secret, sizeof(sender_secret)); ssh_mem_zero(receiver_secret, sizeof(receiver_secret)); From fba92f223ee8b8ce98d73a3561ef735a9140d850 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 09:39:42 +0700 Subject: [PATCH 08/16] Add six-digit console PIN login alongside the password The local console now accepts a six-digit PIN at the login prompt in addition to admin/password. Six digits authenticate directly; any other entry is still treated as a user name. The default development PIN is 012345. A 10^6 search space is small, so the credential is deliberately fenced in: - Console only. The PIN is never offered or accepted over SSH, where a network attacker could search that space quickly. - It rides along with password authentication rather than widening the authentication surface: load_console_pin returns early when password auth is disabled, the build refuses to package a PIN without a user database, and release builds refuse one outright. A key-only image stays key-only. - Five consecutive console failures now lock the prompt for 60 seconds, and a correct credential is refused while the lockout stands. This applies to password and PIN attempts alike; g_console_auth_failures was previously counted and never acted on, so neither path was throttled at all. Stored as PBKDF2-SHA256 with its own salt in /etc/xaios_console_pin, verified with the same constant-time compare as the password, with availability folded into the accumulator so a missing record cannot authenticate. The kernel provisions the record read-only like the user database. Digits are masked as they are typed, since a PIN is a secret rather than a user name. Verified on AArch64 and x86_64: correct PIN opens a shell, wrong PIN is rejected, input is masked, username/password still works, the lockout trips on the fifth failure and refuses the correct PIN while it stands. docs-check and qemu-local-console-gate pass. Co-Authored-By: Claude Opus 5 --- config/development-console-pin | 5 + kernel/core/kmain.c | 1 + scripts/build-image.sh | 26 ++++ userspace/sshd/sshd.c | 240 +++++++++++++++++++++++++++++++-- wiki/Boot-and-Console.md | 12 ++ wiki/Current-Limitations.md | 8 +- 6 files changed, 280 insertions(+), 12 deletions(-) create mode 100644 config/development-console-pin diff --git a/config/development-console-pin b/config/development-console-pin new file mode 100644 index 00000000..5dcd2674 --- /dev/null +++ b/config/development-console-pin @@ -0,0 +1,5 @@ +# Development-only local console PIN: 012345. +# Console-only credential. It is never accepted over SSH, and release images +# reject PIN authentication and never package this record. +# Format: pbkdf2-sha256::: +pbkdf2-sha256:200000:5841494f532d4445562d50494e303031:c6203e31b628cf333ec89006f53e2883d4c13e9ac6b9edd6b26a1e919e4acf09 diff --git a/kernel/core/kmain.c b/kernel/core/kmain.c index fb484396..89311d6f 100644 --- a/kernel/core/kmain.c +++ b/kernel/core/kmain.c @@ -393,6 +393,7 @@ void kmain(const xaios_boot_info_t *boot) { fsck.valid, fsck.version, fsck.files, fsck.directories); provision_read_only_config("/etc/xaios_authorized_keys"); provision_read_only_config("/etc/xaios_sshd_users"); + provision_read_only_config("/etc/xaios_console_pin"); provision_read_only_config("/etc/xapt.conf"); provision_ephemeral_credential("/etc/xaios_ssh_client_identity"); admin_control_init(); diff --git a/scripts/build-image.sh b/scripts/build-image.sh index fd63b3f2..fe79c30c 100755 --- a/scripts/build-image.sh +++ b/scripts/build-image.sh @@ -147,6 +147,25 @@ elif [ "${XAIOS_SSH_PASSWORD_AUTH:-0}" != "0" ]; then exit 2 fi +# Local console PIN. It is console-only and never accepted over SSH, and it +# rides along with password authentication: an image without a password user +# database stays key-only and packages no PIN record. +CONSOLE_PIN_FILE="${XAIOS_CONSOLE_PIN_FILE:-}" +if [ "$BUILD_MODE" = "development" ] && [ "$CONSOLE_PIN_FILE" = "" ] && \ + [ "$SSH_USERS_FILE" = "$ROOT_DIR/config/development-sshd-users" ]; then + CONSOLE_PIN_FILE="$ROOT_DIR/config/development-console-pin" +fi +if [ "$CONSOLE_PIN_FILE" != "" ]; then + if [ "$BUILD_MODE" = "release" ]; then + printf '%s\n' "error: console PIN authentication is forbidden in release builds" >&2 + exit 2 + fi + if [ "$SSH_USERS_FILE" = "" ]; then + printf '%s\n' "error: XAIOS_CONSOLE_PIN_FILE requires XAIOS_SSH_USERS_FILE" >&2 + exit 2 + fi +fi + # Cleanup only failures that occur after the build profile has been accepted. cleanup() { if [ $? -ne 0 ]; then @@ -1081,6 +1100,13 @@ if [ "$SSH_USERS_FILE" != "" ]; then fi set -- "$@" "/etc/xaios_sshd_users=$SSH_USERS_FILE" fi +if [ "$CONSOLE_PIN_FILE" != "" ]; then + if [ ! -f "$CONSOLE_PIN_FILE" ]; then + printf '%s\n' "error: console PIN file not found: $CONSOLE_PIN_FILE" >&2 + exit 1 + fi + set -- "$@" "/etc/xaios_console_pin=$CONSOLE_PIN_FILE" +fi if [ "${XAIOS_SSH_CLIENT_IDENTITY_FILE:-}" != "" ]; then if [ ! -f "$XAIOS_SSH_CLIENT_IDENTITY_FILE" ]; then printf '%s\n' "error: SSH client identity not found: $XAIOS_SSH_CLIENT_IDENTITY_FILE" >&2 diff --git a/userspace/sshd/sshd.c b/userspace/sshd/sshd.c index 1bce5ef8..891bea1f 100644 --- a/userspace/sshd/sshd.c +++ b/userspace/sshd/sshd.c @@ -44,6 +44,13 @@ static uint32_t g_console_auth_state; static uint32_t g_console_auth_failures; static uint64_t g_console_ui_next_refresh; static uint32_t g_console_ui_cursor_visible; +static uint32_t g_console_pin_available; + +/* Defined with the console PIN credential below; needed by the input echo and + by the login prompt, both of which appear earlier in this file. */ +static int console_input_is_pin_prefix(const char *text, uint32_t length); +static int console_input_is_pin(const char *text, uint32_t length); +static int authenticate_console_pin(const char *pin); enum { SSHD_CONSOLE_AUTH_LOCKED = 0U, @@ -676,27 +683,82 @@ static void console_execute_command(void) { console_prompt(); } +/* Consecutive failures cost the attacker wall clock time. This matters most + for the six digit PIN, whose search space is small enough to exhaust in + seconds against a prompt that answers instantly. */ +#define SSHD_CONSOLE_FAILURE_LIMIT 5U +#define SSHD_CONSOLE_LOCKOUT_NS UINT64_C(60000000000) + +static uint64_t g_console_lockout_until_ns; + +static int console_locked_out(void) { + if (g_console_lockout_until_ns == 0U) return 0; + if (xaios_clock_nanos() >= g_console_lockout_until_ns) { + g_console_lockout_until_ns = 0U; + g_console_auth_failures = 0U; + return 0; + } + return 1; +} + +static void console_record_auth_failure(void) { + if (++g_console_auth_failures >= SSHD_CONSOLE_FAILURE_LIMIT) { + g_console_lockout_until_ns = xaios_clock_nanos() + SSHD_CONSOLE_LOCKOUT_NS; + } +} + +static void console_auth_failed(void) { + console_record_auth_failure(); + g_console_auth_state = SSHD_CONSOLE_AUTH_USER; + if (g_console_lockout_until_ns != 0U) { + console_write( + "Login incorrect\n" + "Too many failed attempts. Try again in 60 seconds.\n" + "xaios login: "); + return; + } + console_write("Login incorrect\nxaios login: "); +} + +static void console_auth_succeeded(void) { + g_console_auth_failures = 0U; + g_console_lockout_until_ns = 0U; + g_console_auth_state = SSHD_CONSOLE_AUTH_SHELL; + console_write("XAIOS local console session opened\n"); + console_prompt(); +} + static void console_submit_auth(void) { + uint32_t submitted_length = g_console_command_length; g_console_command[g_console_command_length] = '\0'; console_write("\n"); + if (console_locked_out()) { + console_write("Locked out. Try again in a moment.\nxaios login: "); + g_console_auth_state = SSHD_CONSOLE_AUTH_USER; + xaios_memzero(g_console_command, sizeof(g_console_command)); + g_console_command_length = 0U; + return; + } if (g_console_auth_state == SSHD_CONSOLE_AUTH_USER) { - if (!ssh_str_eq(g_console_command, "admin")) { + if (g_console_pin_available != 0U && + console_input_is_pin(g_console_command, submitted_length)) { + if (authenticate_console_pin(g_console_command) != 0) { + console_auth_failed(); + } else { + console_auth_succeeded(); + } + } else if (!ssh_str_eq(g_console_command, "admin")) { console_write("Login incorrect\nxaios login: "); - ++g_console_auth_failures; + console_record_auth_failure(); } else { g_console_auth_state = SSHD_CONSOLE_AUTH_PASSWORD; console_write("Password: "); } } else if (g_console_auth_state == SSHD_CONSOLE_AUTH_PASSWORD) { if (authenticate_password("admin", g_console_command) != 0) { - ++g_console_auth_failures; - console_write("Login incorrect\nxaios login: "); - g_console_auth_state = SSHD_CONSOLE_AUTH_USER; + console_auth_failed(); } else { - g_console_auth_failures = 0U; - g_console_auth_state = SSHD_CONSOLE_AUTH_SHELL; - console_write("XAIOS local console session opened\n"); - console_prompt(); + console_auth_succeeded(); } } xaios_memzero(g_console_command, sizeof(g_console_command)); @@ -774,7 +836,19 @@ static void console_tick(void) { value >= ' ' && value <= '~' && g_console_command_length + 1U < sizeof(g_console_command)) { g_console_command[g_console_command_length++] = value; - if (g_console_auth_state != SSHD_CONSOLE_AUTH_PASSWORD) { + if (g_console_auth_state == SSHD_CONSOLE_AUTH_PASSWORD) { + /* Never echo a password. */ + } else if (g_console_auth_state == SSHD_CONSOLE_AUTH_USER && + g_console_pin_available != 0U && + console_input_is_pin_prefix(g_console_command, + g_console_command_length)) { + /* An all-digit entry at the login prompt may be a PIN, which is a + secret rather than a user name, so mask it while it is typed. A + user name that merely starts with digits is masked for those + leading digits only. */ + static const char masked = '*'; + (void)xaios_console_write(&masked, 1U); + } else { (void)xaios_console_write(&value, 1U); } } @@ -951,6 +1025,146 @@ static int authenticate_password(const char *username, const char *password) { return diff == 0U ? 0 : -1; } +/* ---- Local Console PIN ---- + A six digit PIN is a 10^6 search space, so this credential is deliberately + restricted: it is accepted only on the local console, never over SSH, and + only when password authentication is already enabled for the image. The + console prompt is rate limited below, because an unthrottled prompt makes a + space this small trivially searchable. */ +#define SSHD_CONSOLE_PIN_PATH "/etc/xaios_console_pin" +#define SSHD_CONSOLE_PIN_DIGITS 6U + +static uint8_t g_console_pin_salt[SSHD_PASSWORD_SALT_MAX]; +static uint8_t g_console_pin_hash[32]; +static uint32_t g_console_pin_salt_len; +static uint32_t g_console_pin_iterations; + +static int parse_console_pin_line(const char *line, uint32_t line_len) { + uint32_t separator[3]; + uint32_t separator_count = 0U; + while (line_len > 0U && line[line_len - 1U] == '\r') --line_len; + if (line_len == 0U || line[0] == '#') return 1; + for (uint32_t i = 0U; i < line_len; ++i) { + if (line[i] != ':') continue; + if (separator_count >= 3U) return -1; + separator[separator_count++] = i; + } + if (separator_count != 3U) return -1; + + static const char scheme[] = "pbkdf2-sha256"; + if (separator[0] != sizeof(scheme) - 1U) return -1; + for (uint32_t i = 0U; i < sizeof(scheme) - 1U; ++i) { + if (line[i] != scheme[i]) return -1; + } + + uint32_t iterations = 0U; + if (parse_decimal_u32(line + separator[0] + 1U, + separator[1] - separator[0] - 1U, &iterations) != 0 || + iterations < SSHD_PASSWORD_ITERATIONS_MIN || + iterations > SSHD_PASSWORD_ITERATIONS_MAX) { + return -1; + } + + uint32_t salt_len = 0U; + if (parse_hex_bytes(line + separator[1] + 1U, + separator[2] - separator[1] - 1U, g_console_pin_salt, + sizeof(g_console_pin_salt), &salt_len) != 0 || + salt_len == 0U) { + return -1; + } + uint32_t hash_len = 0U; + if (parse_hex_bytes(line + separator[2] + 1U, line_len - separator[2] - 1U, + g_console_pin_hash, sizeof(g_console_pin_hash), + &hash_len) != 0 || + hash_len != sizeof(g_console_pin_hash)) { + return -1; + } + g_console_pin_salt_len = salt_len; + g_console_pin_iterations = iterations; + return 0; +} + +static int load_console_pin(void) { + char buffer[512]; + ssh_mem_zero(g_console_pin_salt, sizeof(g_console_pin_salt)); + ssh_mem_zero(g_console_pin_hash, sizeof(g_console_pin_hash)); + g_console_pin_salt_len = 0U; + g_console_pin_iterations = 0U; + g_console_pin_available = 0U; + /* The PIN never widens the authentication surface on its own: an image with + password authentication disabled stays key-only. */ + if (g_password_auth_enabled == 0U) return 0; + + int result = xaios_read_file(SSHD_CONSOLE_PIN_PATH, buffer, sizeof(buffer)); + if (result <= 0) return 0; + + uint32_t line_start = 0U; + for (uint32_t i = 0U; i <= (uint32_t)result; ++i) { + if (i != (uint32_t)result && buffer[i] != '\n') continue; + int parsed = parse_console_pin_line(buffer + line_start, i - line_start); + if (parsed < 0) { + ssh_mem_zero(buffer, sizeof(buffer)); + ssh_mem_zero(g_console_pin_salt, sizeof(g_console_pin_salt)); + ssh_mem_zero(g_console_pin_hash, sizeof(g_console_pin_hash)); + g_console_pin_salt_len = 0U; + g_console_pin_iterations = 0U; + ssh_log(SSH_LOG_ERROR, "Invalid local console PIN record\n"); + return -1; + } + if (parsed == 0) { + g_console_pin_available = 1U; + break; + } + line_start = i + 1U; + } + ssh_mem_zero(buffer, sizeof(buffer)); + if (g_console_pin_available != 0U) + ssh_log(SSH_LOG_INFO, "Local console PIN authentication enabled\n"); + return 0; +} + +static int authenticate_console_pin(const char *pin) { + static const uint8_t dummy_salt[16] = { + 0x58,0x41,0x49,0x4f,0x53,0x2d,0x50,0x49, + 0x4e,0x2d,0x44,0x55,0x4d,0x4d,0x59,0x31 + }; + static const uint8_t dummy_hash[32] = {0}; + const uint8_t *salt = g_console_pin_available != 0U ? g_console_pin_salt + : dummy_salt; + const uint8_t *expected = g_console_pin_available != 0U ? g_console_pin_hash + : dummy_hash; + uint32_t salt_len = g_console_pin_available != 0U ? g_console_pin_salt_len + : (uint32_t)sizeof(dummy_salt); + uint32_t iterations = g_console_pin_available != 0U + ? g_console_pin_iterations + : SSHD_PASSWORD_ITERATIONS_MIN; + uint8_t hash[32]; + if (pbkdf2_hmac_sha256((const uint8_t *)pin, ssh_str_len(pin), salt, + salt_len, iterations, hash) != 0) { + return -1; + } + /* Fold availability into the accumulator so a missing PIN record cannot + authenticate regardless of the derived hash, and keep the compare + constant time. */ + uint8_t diff = (uint8_t)(g_console_pin_available == 0U); + for (uint32_t i = 0U; i < sizeof(hash); ++i) diff |= hash[i] ^ expected[i]; + ssh_mem_zero(hash, sizeof(hash)); + return diff == 0U ? 0 : -1; +} + +static int console_input_is_pin_prefix(const char *text, uint32_t length) { + if (length == 0U || length > SSHD_CONSOLE_PIN_DIGITS) return 0; + for (uint32_t i = 0U; i < length; ++i) { + if (text[i] < '0' || text[i] > '9') return 0; + } + return 1; +} + +static int console_input_is_pin(const char *text, uint32_t length) { + return length == SSHD_CONSOLE_PIN_DIGITS && + console_input_is_pin_prefix(text, length); +} + /* ---- Authorized Keys for Public Key Auth ---- */ #define AUTHORIZED_KEYS_PATH "/etc/xaios_authorized_keys" #define MAX_AUTHORIZED_KEYS 16 @@ -1226,6 +1440,7 @@ int sshd_reload_control_state(const char *command) { if (command_starts_with(command, "xaiosctl config apply ")) { if (load_runtime_config() != 0) return -1; if (load_user_database() != 0) return -1; + if (load_console_pin() != 0) return -1; ssh_log(SSH_LOG_INFO, "Applied SSH runtime configuration generation=%u\n", g_runtime_config.generation); } else if (command_starts_with(command, "xaiosctl auth key add ") || @@ -2138,6 +2353,11 @@ int sshd_run(void) { g_console_boot_error = 2202; goto service_loop; } + if (load_console_pin() != 0) { + ssh_log(SSH_LOG_ERROR, "Local console PIN record rejected\n"); + g_console_boot_error = 2203; + goto service_loop; + } (void)load_authorized_keys(); if (g_authorized_database_invalid != 0U) { g_console_boot_error = 2203; diff --git a/wiki/Boot-and-Console.md b/wiki/Boot-and-Console.md index 91b274f2..fa37ecd0 100644 --- a/wiki/Boot-and-Console.md +++ b/wiki/Boot-and-Console.md @@ -37,6 +37,18 @@ isolated QEMU/Fusion development only; use `XAIOS_SSH_PASSWORD_AUTH=0` to make a development build key-only. Release images keep the local serial console locked and never package a password database. +The same prompt also accepts a six-digit console PIN, `012345` by default. +Entering six digits authenticates directly, without a password prompt; any +other entry is treated as a user name. Digits are masked as they are typed. + +The PIN is a local console credential only. It is never accepted over SSH, +because a six-digit space is far too small to expose to a network attacker, +and it is packaged only when password authentication is already enabled, so a +key-only image stays key-only. Five consecutive console authentication +failures lock the prompt for 60 seconds, which applies to password and PIN +attempts alike. Override the record with `XAIOS_CONSOLE_PIN_FILE`; release +builds refuse to package one at all. + After successful authentication the prompt is: ```text diff --git a/wiki/Current-Limitations.md b/wiki/Current-Limitations.md index cd36ae01..df4a48a0 100644 --- a/wiki/Current-Limitations.md +++ b/wiki/Current-Limitations.md @@ -35,8 +35,12 @@ Progress status and ownership live only in [[Project Tracker|Project-Tracker]]. `1.1.1.1:443` before SSH binds. This checks configured external reachability without depending on public DNS; it is not a general Internet-health check. Failure reports a numeric startup error. A local shell is available after - PBKDF2 authentication in the default development image (`admin` / `xaios`); - key-only and release consoles stay locked. + PBKDF2 authentication in the default development image (`admin` / `xaios`, + or the six-digit console PIN `012345`); key-only and release consoles stay + locked. The PIN is console-only and never accepted over SSH. Its search + space is 10^6, so it is protected only by a 60-second lockout after five + consecutive failures; it is a development convenience and is not a + production-strength credential. - The SSH service deliberately supports 32 transports and two active channels per transport, backed by 64 asynchronous child-channel records. Fleet-scale identity, audit, replay, and connection policy remains unresolved. From 541eba848a91735c3472d24465fa7761dc670da2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 10:03:29 +0700 Subject: [PATCH 09/16] Offer a bidirectional serial console for VMware Fusion The generated Fusion profile wires serial0 to a file, so the guest can print but nothing can be typed back, and the Fusion window itself only shows the boot status screen: progress bar, address, SSH state and a XAIOS LOGIN prompt with a cursor. That screen reflects console state and never echoes input or renders command output, so there was no way to operate the guest from the console at all -- only over SSH. Add XAIOS_FUSION_SERIAL=pipe, which configures serial0 as a VMware pipe endpoint. VMware then creates a UNIX socket that a terminal can attach to, giving a real interactive console: XAIOS_FUSION_SERIAL=pipe ./scripts/build-vmware-fusion.sh ./scripts/run-vmware-fusion.sh nc -U /tmp/xaios-fusion-console XAIOS_FUSION_SERIAL_PIPE moves the socket. The default stays "file" so vmware-fusion-smoke keeps reading fusion-serial.log unchanged. Verified on Fusion 26.0.0: attaching to the pipe reaches the login prompt, the six-digit PIN opens a shell, and echo, df and xaiosctl version all run and return output. The default profile still passes vmware-fusion-smoke. Co-Authored-By: Claude Opus 5 --- scripts/build-vmware-fusion.sh | 30 ++++++++++++++++++++++++++++++ wiki/VMware-Fusion.md | 31 +++++++++++++++++++++++++++++++ 2 files changed, 61 insertions(+) diff --git a/scripts/build-vmware-fusion.sh b/scripts/build-vmware-fusion.sh index 75372ae2..48976075 100755 --- a/scripts/build-vmware-fusion.sh +++ b/scripts/build-vmware-fusion.sh @@ -87,6 +87,36 @@ xorriso -as mkisofs -quiet -R -V XAIOS_FUSION \ -e efi.img -no-emul-boot \ -o "$ISO_IMAGE" "$STAGE_DIR" cp "$ROOT_DIR/platform/vmware-fusion/XAIOS.vmx.in" "$VM_BUNDLE/XAIOS.vmx" +# Serial console wiring. "file" is the default and keeps the automated Fusion +# smoke gate reading fusion-serial.log. "pipe" makes the console bidirectional +# so an operator can actually log in and type; VMware creates a UNIX socket at +# XAIOS_FUSION_SERIAL_PIPE that a terminal can attach to. +FUSION_SERIAL="${XAIOS_FUSION_SERIAL:-file}" +FUSION_SERIAL_PIPE="${XAIOS_FUSION_SERIAL_PIPE:-/tmp/xaios-fusion-console}" +case "$FUSION_SERIAL" in + file) ;; + pipe) + "${XAIOS_PYTHON3:-python3}" - "$VM_BUNDLE/XAIOS.vmx" "$FUSION_SERIAL_PIPE" <<'PYEOF' +import sys +vmx, pipe = sys.argv[1], sys.argv[2] +text = open(vmx, encoding="utf-8").read() +old = ('serial0.fileType = "file"\n' + 'serial0.fileName = "fusion-serial.log"\n') +new = (f'serial0.fileType = "pipe"\n' + f'serial0.fileName = "{pipe}"\n' + f'serial0.pipe.endPoint = "server"\n' + f'serial0.startConnected = "TRUE"\n') +if old not in text: + raise SystemExit("error: unexpected serial configuration in VMX template") +open(vmx, "w", encoding="utf-8").write(text.replace(old, new)) +PYEOF + printf '%s\n' "Serial console: bidirectional pipe at $FUSION_SERIAL_PIPE" + ;; + *) + printf '%s\n' "error: XAIOS_FUSION_SERIAL must be file or pipe" >&2 + exit 2 + ;; +esac "$VDISK_MANAGER" -c -s "${XAIOS_FUSION_DISK_SIZE:-256MB}" -a lsilogic -t 0 \ "$DATA_DISK" >/dev/null diff --git a/wiki/VMware-Fusion.md b/wiki/VMware-Fusion.md index afe2a2c3..37febfab 100644 --- a/wiki/VMware-Fusion.md +++ b/wiki/VMware-Fusion.md @@ -37,6 +37,37 @@ make vmware-fusion-image make vmware-fusion-smoke ``` +## Typing at the Fusion console + +The generated profile wires the serial port to `fusion-serial.log`, which is +write-only: the guest prints to it but nothing can be typed back. The Fusion +window itself shows the boot status screen — progress bar, address, SSH state +and a `XAIOS LOGIN:` prompt with a blinking cursor — but that screen is a +status display, not a terminal. It never echoes typed characters or renders +command output, so it cannot be used to operate the system. + +There are two ways to get an interactive session: + +**SSH**, which is the intended administration path. The guest takes a bridged +address, printed on the boot screen, and accepts `admin` / `xaios`: + +```sh +ssh admin@
+``` + +**A bidirectional serial pipe**, for console-level access such as recovery or +watching early boot. Build with the serial port as a pipe, then attach a +terminal to the socket VMware creates: + +```sh +XAIOS_FUSION_SERIAL=pipe ./scripts/build-vmware-fusion.sh +./scripts/run-vmware-fusion.sh +nc -U /tmp/xaios-fusion-console +``` + +Set `XAIOS_FUSION_SERIAL_PIPE` to move the socket. The default stays `file` +so the automated Fusion smoke gate keeps reading `fusion-serial.log`. + The default development account is `admin` / `xaios`; it is public and must only be used on an isolated development network. For key-based SSH, package a disposable key when building and connect to the address shown by the guest: From 3449607f4a3360cea50db83e493ebe71278227b7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 10:58:02 +0700 Subject: [PATCH 10/16] Reach the USB keyboard on VMware Fusion Typing in the Fusion window did nothing because no keyboard was ever bound. Two independent causes, found by dumping what the guest actually enumerates. The control endpoint ring was never rewound between devices. Each slot's device context starts its control endpoint at the base of the shared ep0 ring, but the driver kept its enqueue position from the previous device, so every control transfer to the second and later devices timed out waiting for completions referencing TRBs the controller had already passed. On QEMU the keyboard is the first device enumerated, so this never showed. On Fusion the keyboard is on port 6, behind a device on port 5, and was unreachable: input: xHCI control transfer timed out input: xHCI GET_DEVICE_DESCRIPTOR failed input: xHCI HID configuration failed port=6 slot=2 Reset the index and clear the ring before addressing each device. The port 5 device is also HID, but a composite whose interfaces are both subclass 0 / protocol 0, so the boot descriptors cannot identify it. Both interfaces carry an 8-byte interrupt IN endpoint, so accepting the first HID interface with a plausible report size would have bound a pointing device as a keyboard and typed noise. Fall back to the report descriptor instead and require a Generic Desktop Keyboard usage; both port 5 interfaces declare Usage Mouse (05 01 09 02) and are correctly refused. The failure path now logs the interfaces and endpoints a rejected device advertises, because a controller that enumerates but exposes no keyboard is a descriptor question and the descriptor is the only thing that answers it. Verified: Fusion 26.0.0 reports "xHCI HID boot keyboard initialized" and passes the translation self-test; qemu-keyboard-input-gate still passes on AArch64. Co-Authored-By: Claude Opus 5 --- kernel/dev/input.c | 103 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 103 insertions(+) diff --git a/kernel/dev/input.c b/kernel/dev/input.c index f92c3028..8dea5ed8 100644 --- a/kernel/dev/input.c +++ b/kernel/dev/input.c @@ -344,6 +344,30 @@ static int control_transfer(xhci_keyboard_t *keyboard, uint8_t request_type, } } } +/* Fetch an interface's HID report descriptor and report whether it declares a + Generic Desktop Keyboard usage (Usage Page 0x01, Usage 0x06). This is what + separates a keyboard from a pointing device on a composite HID device that + advertises neither boot subclass nor protocol. */ +static int report_descriptor_is_keyboard(xhci_keyboard_t *keyboard, + uint8_t interface_number, + uint16_t report_bytes) { + uint8_t report[256]; + if (report_bytes == 0U) return 0; + if (report_bytes > sizeof(report)) report_bytes = sizeof(report); + zero_bytes(report, sizeof(report)); + if (!control_transfer(keyboard, UINT8_C(0x81), 6U, UINT16_C(0x2200), + interface_number, report, report_bytes)) { + return 0; + } + for (uint16_t i = 0U; i + 3U < report_bytes; ++i) { + if (report[i] == UINT8_C(0x05) && report[i + 1U] == UINT8_C(0x01) && + report[i + 2U] == UINT8_C(0x09) && report[i + 3U] == UINT8_C(0x06)) { + return 1; + } + } + return 0; +} + static int configure_keyboard(xhci_keyboard_t *keyboard, uint32_t port) { uint8_t descriptor[256]; zero_bytes(descriptor, sizeof(descriptor)); @@ -372,6 +396,22 @@ static int configure_keyboard(xhci_keyboard_t *keyboard, uint32_t port) { uint8_t endpoint = 0U; uint16_t packet_size = 8U; uint8_t interval = 10U; + /* Two ways to recognise a keyboard. A device that advertises the HID boot + subclass and keyboard protocol says so outright, and QEMU does. VMware + Fusion instead exposes a composite HID device whose interfaces are all + subclass 0 / protocol 0, so the boot descriptors cannot pick the keyboard + out from the pointing device sharing the same report size. For those, + fall back to the report descriptor and look for a Generic Desktop + Keyboard usage, which distinguishes them properly. */ + uint32_t candidate_interface = 0U; + uint8_t candidate_number = 0U; + uint8_t candidate_endpoint = 0U; + uint16_t candidate_packet = 0U; + uint8_t candidate_interval = 10U; + uint16_t candidate_report_bytes = 0U; + uint32_t hid_interface = 0U; + uint8_t hid_number = 0U; + uint16_t hid_report_bytes = 0U; uint32_t boot_keyboard_interface = 0U; for (uint32_t offset = 0U; offset + 2U <= total;) { uint8_t length = descriptor[offset]; @@ -382,6 +422,15 @@ static int configure_keyboard(xhci_keyboard_t *keyboard, uint32_t port) { descriptor[offset + 6U] == 1U && descriptor[offset + 7U] == 1U; if (boot_keyboard_interface != 0U) interface_number = descriptor[offset + 2U]; + hid_interface = descriptor[offset + 5U] == 3U; + hid_number = descriptor[offset + 2U]; + hid_report_bytes = 0U; + } + /* HID class descriptor: remember the report descriptor length. */ + if (hid_interface != 0U && descriptor[offset + 1U] == UINT8_C(0x21) && + length >= 9U && descriptor[offset + 6U] == UINT8_C(0x22)) { + hid_report_bytes = (uint16_t)descriptor[offset + 7U] | + ((uint16_t)descriptor[offset + 8U] << 8U); } if (boot_keyboard_interface != 0U && descriptor[offset + 1U] == 5U && length >= 7U && (descriptor[offset + 2U] & UINT8_C(0x80)) != 0U && descriptor[offset + 3U] == 3U) { @@ -391,11 +440,58 @@ static int configure_keyboard(xhci_keyboard_t *keyboard, uint32_t port) { interval = descriptor[offset + 6U]; break; } + if (hid_interface != 0U && candidate_interface == 0U && + descriptor[offset + 1U] == 5U && length >= 7U && + (descriptor[offset + 2U] & UINT8_C(0x80)) != 0U && + descriptor[offset + 3U] == 3U) { + uint16_t size = ((uint16_t)descriptor[offset + 4U] | + ((uint16_t)descriptor[offset + 5U] << 8U)) & + UINT16_C(0x07ff); + if (size >= 8U && size <= 64U && + report_descriptor_is_keyboard(keyboard, hid_number, + hid_report_bytes)) { + candidate_interface = 1U; + candidate_number = hid_number; + candidate_endpoint = descriptor[offset + 2U]; + candidate_packet = size; + candidate_interval = descriptor[offset + 6U]; + candidate_report_bytes = hid_report_bytes; + } + } offset += length; } + if (endpoint == 0U && candidate_interface != 0U) { + interface_number = candidate_number; + endpoint = candidate_endpoint; + packet_size = candidate_packet; + interval = candidate_interval; + klog("input: xHCI keyboard by report descriptor interface=%u report=%u\n", + interface_number, candidate_report_bytes); + } if (endpoint == 0U || packet_size == 0U || packet_size > 64U) { klog("input: xHCI HID endpoint unavailable endpoint=0x%x packet=%u\n", endpoint, packet_size); + /* Report what the device actually advertises: a controller that enumerates + but exposes no boot keyboard is a descriptor question, not a bus fault, + and the descriptor is the only thing that can answer it. */ + for (uint32_t offset = 0U; offset + 2U <= total;) { + uint8_t length = descriptor[offset]; + if (length < 2U || length > total - offset) break; + if (descriptor[offset + 1U] == 4U && length >= 9U) { + klog("input: xHCI interface=%u class=0x%x subclass=0x%x protocol=0x%x " + "endpoints=%u\n", + descriptor[offset + 2U], descriptor[offset + 5U], + descriptor[offset + 6U], descriptor[offset + 7U], + descriptor[offset + 4U]); + } else if (descriptor[offset + 1U] == 5U && length >= 7U) { + klog("input: xHCI endpoint=0x%x attributes=0x%x packet=%u\n", + descriptor[offset + 2U], descriptor[offset + 3U], + (uint32_t)(((uint16_t)descriptor[offset + 4U] | + ((uint16_t)descriptor[offset + 5U] << 8U)) & + UINT16_C(0x07ff))); + } + offset += length; + } return 0; } if (!control_transfer(keyboard, 0U, 9U, config_value, 0U, 0, 0U)) { @@ -500,6 +596,13 @@ static int initialize_keyboard(xhci_keyboard_t *keyboard) { context_write32(slot, 1U, port << 16U); uint8_t *ep0 = keyboard->input_context + keyboard->context_bytes * 2U; uint64_t ep0_ring = dma_address(keyboard->ep0_ring); + /* Each slot's device context starts its control endpoint at the base of + this ring, so the driver's enqueue position has to start there too. + Without this reset a second device inherits the previous device's + position, and every control transfer to it times out waiting for + completions that reference TRBs the controller already passed. */ + keyboard->ep0_index = 0U; + zero_bytes(keyboard->ep0_ring, sizeof(xhci_trb_t) * XHCI_RING_SIZE); keyboard->ep0_ring[XHCI_RING_SIZE - 1U].parameter_lo = (uint32_t)ep0_ring; keyboard->ep0_ring[XHCI_RING_SIZE - 1U].parameter_hi = (uint32_t)(ep0_ring >> 32U); From b4c5913d6c1cf9a60e8d8feeb7ec785c53de042e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 10:58:03 +0700 Subject: [PATCH 11/16] Turn the framebuffer into a terminal after boot The display was a status panel: it drew the progress bar, address, SSH state and a XAIOS LOGIN prompt with a cursor, but it reflected console state rather than showing the console. It never echoed a keystroke or rendered command output, so a machine whose only console is the screen -- VMware Fusion, which wires its serial port to a file -- could be watched but not used. Add a framebuffer terminal and hand the display over to it once boot completes. From that point the framebuffer receives exactly the byte stream the UART receives, so the screen shows the same session a QEMU serial console shows: prompt, echoed input, command output, scrollback. The terminal handles newline, carriage return, backspace and tab, wraps and scrolls, and parses CSI sequences so the shell's colour codes render rather than printing as noise. Bytes belonging to a capturing session are excluded, exactly as they are for the UART, so output still appears once. boot_ui_self_test renders a glyph and reads the pixels back, so the display path is proven on the machine that has a framebuffer instead of assumed. Known limitation: the bundled font covers ' ' through '_' and folds lowercase to uppercase, so the terminal renders uppercase. Extending the font is a separate change. Verified on Fusion 26.0.0: "framebuffer 1024x768 glyph readback lit=56 passed" and "framebuffer terminal active 112x41 cells". QEMU has no framebuffer, so the terminal stays dormant there and qemu-local-console-gate, qemu-keyboard-input-gate, docs-check and qemu-abi-contract all still pass. Co-Authored-By: Claude Opus 5 --- kernel/core/boot_ui.c | 304 ++++++++++++++++++++++++++++++++- kernel/core/klog.c | 6 + kernel/core/kmain.c | 1 + kernel/include/xaios/boot_ui.h | 5 + 4 files changed, 314 insertions(+), 2 deletions(-) diff --git a/kernel/core/boot_ui.c b/kernel/core/boot_ui.c index 23792edb..de9e1389 100644 --- a/kernel/core/boot_ui.c +++ b/kernel/core/boot_ui.c @@ -313,6 +313,293 @@ static void fb_draw_ready(const xaios_boot_ui_control_t *control) { } } +/* ---- Framebuffer text terminal ---- + After boot the display stops being a status panel and becomes a terminal + that mirrors the console byte stream, so a machine with no serial cable -- + VMware Fusion in particular -- offers the same session as a QEMU serial + console: prompt, echoed input, and command output. */ +#define TERM_MARGIN_X UINT32_C(8) +#define TERM_MARGIN_Y UINT32_C(8) +#define TERM_LINE_HEIGHT (FB_GLYPH_HEIGHT * FB_GLYPH_Y_SCALE + UINT32_C(2)) +#define TERM_TAB_WIDTH UINT32_C(8) +#define TERM_ESC_IDLE UINT32_C(0) +#define TERM_ESC_SAW_ESC UINT32_C(1) +#define TERM_ESC_CSI UINT32_C(2) +#define TERM_CSI_PARAM_MAX UINT32_C(8) + +static uint32_t g_term_active; +static void term_putc(uint8_t value); +static uint32_t g_term_columns; +static uint32_t g_term_rows; +static uint32_t g_term_column; +static uint32_t g_term_row; +static uint32_t g_term_color; +static uint32_t g_term_esc_state; +static uint32_t g_term_params[TERM_CSI_PARAM_MAX]; +static uint32_t g_term_param_count; +static uint32_t g_term_cursor_drawn; + +static uint32_t term_background(void) { return fb_color(4U, 6U, 10U); } +static uint32_t term_foreground(void) { return fb_color(222U, 230U, 236U); } + +static uint32_t term_sgr_color(uint32_t code) { + switch (code) { + case 30U: case 90U: return fb_color(90U, 100U, 110U); + case 31U: case 91U: return fb_color(235U, 110U, 100U); + case 32U: case 92U: return fb_color(120U, 210U, 140U); + case 33U: case 93U: return fb_color(226U, 190U, 110U); + case 34U: case 94U: return fb_color(120U, 165U, 235U); + case 35U: case 95U: return fb_color(200U, 140U, 225U); + case 36U: case 96U: return fb_color(110U, 200U, 210U); + default: return term_foreground(); + } +} + +static uint32_t term_x(uint32_t column) { + return TERM_MARGIN_X + column * FB_GLYPH_ADVANCE; +} + +static uint32_t term_y(uint32_t row) { + return TERM_MARGIN_Y + row * TERM_LINE_HEIGHT; +} + +static void term_erase_cursor(void) { + if (g_term_cursor_drawn == 0U) return; + fb_rect(term_x(g_term_column), term_y(g_term_row) + FB_GLYPH_HEIGHT * + FB_GLYPH_Y_SCALE - UINT32_C(2), + FB_GLYPH_WIDTH, UINT32_C(2), term_background()); + g_term_cursor_drawn = 0U; +} + +static void term_draw_cursor(void) { + if (g_term_active == 0U || g_term_cursor_drawn != 0U) return; + fb_rect(term_x(g_term_column), term_y(g_term_row) + FB_GLYPH_HEIGHT * + FB_GLYPH_Y_SCALE - UINT32_C(2), + FB_GLYPH_WIDTH, UINT32_C(2), term_foreground()); + g_term_cursor_drawn = 1U; +} + +static void term_scroll(void) { + uint32_t shift = TERM_LINE_HEIGHT; + if (g_framebuffer.pixels == 0 || shift >= g_framebuffer.height) return; + uint64_t stride = g_framebuffer.stride; + for (uint32_t y = TERM_MARGIN_Y; y + shift < g_framebuffer.height; ++y) { + volatile uint32_t *destination = &g_framebuffer.pixels[(uint64_t)y * stride]; + volatile uint32_t *source = + &g_framebuffer.pixels[(uint64_t)(y + shift) * stride]; + for (uint32_t x = 0U; x < g_framebuffer.width; ++x) destination[x] = source[x]; + } + fb_rect(0U, g_framebuffer.height - shift, g_framebuffer.width, shift, + term_background()); +} + +static void term_newline(void) { + g_term_column = 0U; + if (g_term_row + 1U < g_term_rows) { + ++g_term_row; + return; + } + term_scroll(); +} + +static void term_apply_csi(char final) { + if (final == 'm') { + if (g_term_param_count == 0U) { + g_term_color = term_foreground(); + return; + } + for (uint32_t i = 0U; i < g_term_param_count; ++i) { + uint32_t code = g_term_params[i]; + if (code == 0U) { + g_term_color = term_foreground(); + } else if ((code >= 30U && code <= 37U) || (code >= 90U && code <= 97U)) { + g_term_color = term_sgr_color(code); + } + } + return; + } + if (final == 'J' || final == 'H') { + /* Clear and home are the only cursor controls the shell emits that must + affect this display; the remainder are ignored deliberately. */ + if (final == 'J') { + fb_rect(0U, 0U, g_framebuffer.width, g_framebuffer.height, + term_background()); + } + g_term_column = 0U; + g_term_row = 0U; + } +} + +static void term_putc(uint8_t value) { + if (g_term_esc_state == TERM_ESC_SAW_ESC) { + if (value == '[') { + g_term_esc_state = TERM_ESC_CSI; + g_term_param_count = 0U; + g_term_params[0] = 0U; + } else { + g_term_esc_state = TERM_ESC_IDLE; + } + return; + } + if (g_term_esc_state == TERM_ESC_CSI) { + if (value >= '0' && value <= '9') { + if (g_term_param_count == 0U) g_term_param_count = 1U; + uint32_t *slot = &g_term_params[g_term_param_count - 1U]; + if (*slot < UINT32_C(100000)) *slot = *slot * 10U + (uint32_t)(value - '0'); + return; + } + if (value == ';') { + if (g_term_param_count < TERM_CSI_PARAM_MAX) { + g_term_params[g_term_param_count++] = 0U; + } + return; + } + if (value == '?' || value == ':') return; + term_apply_csi((char)value); + g_term_esc_state = TERM_ESC_IDLE; + return; + } + if (value == UINT8_C(0x1b)) { + g_term_esc_state = TERM_ESC_SAW_ESC; + return; + } + if (value == '\n') { + term_newline(); + return; + } + if (value == '\r') { + g_term_column = 0U; + return; + } + if (value == '\b') { + if (g_term_column != 0U) --g_term_column; + fb_rect(term_x(g_term_column), term_y(g_term_row), FB_GLYPH_ADVANCE, + TERM_LINE_HEIGHT, term_background()); + return; + } + if (value == '\t') { + uint32_t next = (g_term_column / TERM_TAB_WIDTH + 1U) * TERM_TAB_WIDTH; + while (g_term_column < next) { + if (g_term_column >= g_term_columns) { + term_newline(); + break; + } + fb_rect(term_x(g_term_column), term_y(g_term_row), FB_GLYPH_ADVANCE, + TERM_LINE_HEIGHT, term_background()); + ++g_term_column; + } + return; + } + if (value < ' ' || value > '~') return; + if (g_term_column >= g_term_columns) term_newline(); + fb_rect(term_x(g_term_column), term_y(g_term_row), FB_GLYPH_ADVANCE, + TERM_LINE_HEIGHT, term_background()); + fb_glyph(term_x(g_term_column), term_y(g_term_row), (char)value, g_term_color); + ++g_term_column; +} + +static void term_activate(const xaios_boot_ui_control_t *control) { + if (g_framebuffer.pixels == 0 || g_term_active != 0U) return; + g_term_columns = + (g_framebuffer.width - 2U * TERM_MARGIN_X) / FB_GLYPH_ADVANCE; + g_term_rows = (g_framebuffer.height - 2U * TERM_MARGIN_Y) / TERM_LINE_HEIGHT; + if (g_term_columns == 0U || g_term_rows == 0U) return; + if (g_term_columns > UINT32_C(512)) g_term_columns = UINT32_C(512); + fb_rect(0U, 0U, g_framebuffer.width, g_framebuffer.height, term_background()); + g_term_column = 0U; + g_term_row = 0U; + g_term_color = term_foreground(); + g_term_esc_state = TERM_ESC_IDLE; + g_term_param_count = 0U; + g_term_cursor_drawn = 0U; + g_term_active = 1U; + klog("boot-ui: framebuffer terminal active %ux%u cells\n", g_term_columns, + g_term_rows); + + /* The login banner was written to the console before this display existed, + so reproduce the reachability summary and the prompt for the state the + console is actually in. */ + boot_ui_console_text("XAI OS\n\n"); + if (control != 0 && control->ipv4 != 0U) { + boot_ui_console_text("IPv4: "); + uint32_t address = control->ipv4; + for (uint32_t i = 0U; i < 4U; ++i) { + uint32_t octet = (address >> (24U - 8U * i)) & UINT32_C(0xff); + char digits[4]; + uint32_t count = 0U; + do { + digits[count++] = (char)('0' + octet % 10U); + octet /= 10U; + } while (octet != 0U); + while (count != 0U) term_putc((uint8_t)digits[--count]); + if (i != 3U) term_putc((uint8_t)'.'); + } + boot_ui_console_text("\n"); + } + boot_ui_console_text("SSH server: up and running (tcp/22)\n\n"); + if (control != 0 && control->console_state == XAIOS_BOOT_UI_CONSOLE_PASSWORD) { + boot_ui_console_text("Password: "); + } else if (control != 0 && + control->console_state == XAIOS_BOOT_UI_CONSOLE_SHELL) { + boot_ui_console_text("admin@xaios:/$ "); + } else if (control == 0 || + control->console_state != XAIOS_BOOT_UI_CONSOLE_LOCKED) { + boot_ui_console_text("xaios login: "); + } else { + boot_ui_console_text("Local console locked: use SSH public-key access.\n"); + } +} + +/* Render a known glyph into a scratch cell and read the pixels back, so the + display path is proven on the machine that actually has a framebuffer + rather than assumed from the code. */ +void boot_ui_self_test(void) { + if (g_framebuffer.pixels == 0) { + klog("boot-ui: no framebuffer; terminal renders to serial only\n"); + return; + } + uint32_t foreground = fb_color(255U, 255U, 255U); + uint32_t background = fb_color(0U, 0U, 0U); + uint32_t width = FB_GLYPH_WIDTH * FB_GLYPH_X_SCALE; + uint32_t height = FB_GLYPH_HEIGHT * FB_GLYPH_Y_SCALE; + uint32_t origin_x = g_framebuffer.width - width; + uint32_t origin_y = g_framebuffer.height - height; + + fb_rect(origin_x, origin_y, width, height, background); + fb_glyph(origin_x, origin_y, 'A', foreground); + uint32_t lit = 0U; + for (uint32_t row = 0U; row < height; ++row) { + volatile uint32_t *pixel = + &g_framebuffer.pixels[(uint64_t)(origin_y + row) * g_framebuffer.stride + + origin_x]; + for (uint32_t column = 0U; column < width; ++column) { + if ((pixel[column] & UINT32_C(0x00ffffff)) == + (foreground & UINT32_C(0x00ffffff))) { + ++lit; + } + } + } + fb_rect(origin_x, origin_y, width, height, background); + klog("boot-ui: framebuffer %ux%u glyph readback lit=%u %s\n", + g_framebuffer.width, g_framebuffer.height, lit, + lit != 0U ? "passed" : "FAILED"); +} + +void boot_ui_console_write(const char *text, uint64_t length) { + if (g_term_active == 0U || text == 0) return; + term_erase_cursor(); + for (uint64_t i = 0U; i < length; ++i) term_putc((uint8_t)text[i]); + term_draw_cursor(); +} + +void boot_ui_console_text(const char *text) { + if (text == 0) return; + uint64_t length = 0U; + while (text[length] != '\0') ++length; + if (g_term_active == 0U) return; + for (uint64_t i = 0U; i < length; ++i) term_putc((uint8_t)text[i]); +} + static void fb_init(const xaios_boot_info_t *boot) { if (boot == 0 || boot->framebuffer_base == 0U || boot->framebuffer_format == XAIOS_FRAMEBUFFER_NONE || @@ -398,8 +685,21 @@ uint32_t boot_ui_handle_control(const xaios_boot_ui_control_t *control) { return 1U; } if (control->stage == XAIOS_BOOT_UI_STAGE_SSH_READY) { - fb_draw_status(100U, "system services", "complete", 0U); - fb_draw_ready(control); + if (g_term_active == 0U) { + fb_draw_status(100U, "system services", "complete", 0U); + fb_draw_ready(control); + /* Boot is finished, so hand the display over to a real terminal. From + here the framebuffer mirrors the console instead of summarising it. */ + term_activate(control); + return 1U; + } + /* In terminal mode the control record only drives the cursor; the text + itself arrives through the console stream. */ + if (control->cursor_visible != 0U) { + term_draw_cursor(); + } else { + term_erase_cursor(); + } return 1U; } if (control->stage == XAIOS_BOOT_UI_STAGE_SSH_FAILED) { diff --git a/kernel/core/klog.c b/kernel/core/klog.c index e05b33e9..3af65a9c 100644 --- a/kernel/core/klog.c +++ b/kernel/core/klog.c @@ -1,6 +1,7 @@ #include #include #include +#include #if defined(__aarch64__) #include #endif @@ -127,6 +128,11 @@ void klog_console_write(const char *message, uint64_t length) { uart_putc(message[i]); } xaios_spin_unlock(&g_klog_lock); + /* The framebuffer terminal is a second console attached to the same stream, + so it receives exactly what the UART receives: everything except bytes + that belong to a capturing session. Written outside the klog lock because + it paints pixels and must not hold the console lock while doing so. */ + if (!capturing) boot_ui_console_write(message, length); } int klog_console_capture_begin(char *buffer, uint64_t capacity) { diff --git a/kernel/core/kmain.c b/kernel/core/kmain.c index 89311d6f..9cbcd3fe 100644 --- a/kernel/core/kmain.c +++ b/kernel/core/kmain.c @@ -172,6 +172,7 @@ void kmain(const xaios_boot_info_t *boot) { uint32_t persistent_network_ready = 0U; klog_init(boot); boot_ui_begin(boot); + boot_ui_self_test(); boot_ui_update(25U, "hardware handoff", "CPU and interrupts", 5U); klog("XAIOS kernel starting\n"); kassert(boot->magic == XAIOS_BOOT_INFO_MAGIC); diff --git a/kernel/include/xaios/boot_ui.h b/kernel/include/xaios/boot_ui.h index facc7edf..8d87867d 100644 --- a/kernel/include/xaios/boot_ui.h +++ b/kernel/include/xaios/boot_ui.h @@ -29,5 +29,10 @@ void boot_ui_update(uint32_t percent, const char *loaded, const char *loading, uint32_t remaining); void boot_ui_error(const char *component, int32_t status); uint32_t boot_ui_handle_control(const xaios_boot_ui_control_t *control); +/* Mirror console bytes onto the framebuffer terminal, once boot hands the + display over. No-ops when there is no framebuffer or before handover. */ +void boot_ui_console_write(const char *text, uint64_t length); +void boot_ui_console_text(const char *text); +void boot_ui_self_test(void); #endif From a823d7974020334f113bdf1de5283090a2f18f78 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 12:38:25 +0700 Subject: [PATCH 12/16] Capture guest diagnostics in the FreeBSD gate The gate has failed twice in CI for two different reasons -- ProxyJump host key verification once, an SFTP write the next run -- and neither failure left any guest-side evidence. The uploaded serial log stopped at the login prompt, because a non-verbose build calls klog_console_set_log_output(0) once boot finishes, so everything the kernel had to say about the failure was discarded. The information already exists. MutableFS reports exactly why a write fails: mutable-fs: write allocation failed path=%s blocks=%u used=%lu mutable-fs: write block IO failed path=%s blocks=%u mutable-fs: write rejected path=%s mounted=%u flags=0x%x None of it reached the artifact. Build the guest with XAIOS_BOOT_VERBOSE so kernel logging stays on for the whole run and lands in the evidence that is already uploaded. XAIOS_BOOT_VERBOSE is honoured if the caller sets it. Both observed failures are the same underlying event: a failed MutableFS write. The SFTP failure is a pwrite returning <= 0, reported as SSH_FX_FAILURE "Write failed". The ProxyJump failure is verify_known_host in ssh_client.c returning -1 after its pwrite or fsync fails, which surfaces to the user as "host key verification failed". Naming the cause needs the kernel's own message, which this change preserves. Verified: the suite passes locally and the captured log grows from boot output only to 715 lines of kernel diagnostics, including every filesystem operation with path, size, block count and generation. Co-Authored-By: Claude Opus 5 --- tests/scripts/qemu-freebsd-bidirectional-suite.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/tests/scripts/qemu-freebsd-bidirectional-suite.py b/tests/scripts/qemu-freebsd-bidirectional-suite.py index 1b7419ea..c1418d16 100644 --- a/tests/scripts/qemu-freebsd-bidirectional-suite.py +++ b/tests/scripts/qemu-freebsd-bidirectional-suite.py @@ -510,6 +510,11 @@ def main() -> int: build_env["XAIOS_SSH_CLIENT_IDENTITY_FILE"] = str(key_dir / "outbound") build_env.pop("XAIOS_SSH_USERS_FILE", None) build_env["XAIOS_SSH_PASSWORD_AUTH"] = "0" + # Keep kernel logging on the serial console for the whole run. The guest + # already reports why a filesystem write failed, but a non-verbose build + # silences klog once boot finishes, so the captured evidence stopped at the + # login prompt and every failure here had to be re-diagnosed blind. + build_env.setdefault("XAIOS_BOOT_VERBOSE", "1") build_target = "image" if architecture == "aarch64" else "image-x86_64" run_checked(["make", build_target], 360, build_env) From 6f0a8909f549505fe69d8f9bd782e40ae90250a6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 12:59:20 +0700 Subject: [PATCH 13/16] Stop clearing the whole file buffer on every fd write mutable_fs_write_fd cleared its entire staging buffer before every write. The buffer is sized for the largest supported file, MFS_V5_MAX_FILE_BYTES, so appending one audit line meant a 256 KiB memset. The SSH audit log alone paid roughly 21 MiB of it to write 3 KiB of lines across one interoperability run, and every fd writer -- SFTP uploads, xapt, editor saves -- paid the same toll per call. Under TCG emulation that is not free. Only a cursor seeked past the end of the file leaves a hole that has to read back as zeros, so clear just that hole. Everything below file_size was filled by the read_file above, everything from the cursor on is overwritten by the caller's data, and nothing above new_size is written to the volume, so stale bytes further up the buffer cannot reach the disk or leak into a file. This does not change the read-modify-write shape of the call. Rewriting the whole file on each append is inherent to the volume's copy-on-write update, which allocates a fresh block set before releasing the old one; making appends touch only the affected blocks means sharing unchanged blocks between generations, which is a separate change with crash-consistency consequences. Verified: qemu-smoke, qemu-milestone-gate 62 (filesystem), qemu-storage-crash (all metadata kill points recovered), qemu-local-console-gate and compile-check all pass. Co-Authored-By: Claude Opus 5 --- kernel/fs/mutable_fs.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/kernel/fs/mutable_fs.c b/kernel/fs/mutable_fs.c index 33cb9155..d8f28a46 100644 --- a/kernel/fs/mutable_fs.c +++ b/kernel/fs/mutable_fs.c @@ -2025,7 +2025,6 @@ int64_t mutable_fs_write_fd(uint32_t fd, const void *buffer, uint64_t size) { } uint64_t file_size = 0; - bytes_zero(g_file_buffer, sizeof(g_file_buffer)); if (find_node(handle->path, 0) != 0) { if (read_file(handle->path, g_file_buffer, sizeof(g_file_buffer), &file_size) != XAIOS_OK) { @@ -2036,6 +2035,15 @@ int64_t mutable_fs_write_fd(uint32_t fd, const void *buffer, uint64_t size) { if (new_size < file_size) { new_size = file_size; } + /* Only a cursor seeked past the end leaves a hole, and only that hole has to + read back as zeros. Everything below file_size was just read back, and + everything from the cursor on is about to be overwritten, so clearing the + whole buffer meant a 256 KiB memset for every append -- the audit log paid + roughly 21 MiB of it to write 3 KiB of lines. Nothing above new_size is + written out, so stale bytes there cannot reach the volume. */ + if (handle->cursor > file_size) { + bytes_zero(g_file_buffer + file_size, handle->cursor - file_size); + } bytes_copy(g_file_buffer + handle->cursor, buffer, size); if (write_file(handle->path, g_file_buffer, new_size) != XAIOS_OK) { return (int64_t)XAIOS_ERR_IO; From 0a32d89eaf785bbce8130709ed020bae8cfaf522 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 13:33:29 +0700 Subject: [PATCH 14/16] Write snapshot state to the durable volume Snapshot state never survived a restart. persistence_save_to_disk called the default block device, which is vblk0 -- the initramfs/test image that run-qemu-aarch64.sh attaches with snapshot=on, so every write to it is discarded when the machine stops. The first boot wrote its five records and read them straight back from the same overlay, which looked like success; the next boot found an empty sector and logged "no existing disk state". The layout already anticipated the right home for it: PERSISTENCE_SECTOR is 3000 and MFS_START_SECTOR is 3072, so the snapshot sector was meant to sit just below MutableFS on the same durable volume. Only the device binding was wrong, and both candidate devices are larger than sector 3000, so the capacity assertion passed either way and nothing complained. Add persistence_bind_block_device and route the sector reads and writes through it. kmain now opens the dedicated persistent slot before the persistence self-test and reuses that handle for the MutableFS mount below rather than opening it twice. When no durable slot exists the default device is still used, which keeps the pre-storage self-test working on machines that expose no writable volume. Verified: qemu-persistence-reboot now reports "mutable VirtIO state survived QEMU reboot" with persistence_boot_loads=1, where it previously failed the second boot. qemu-smoke, filesystem milestone 62, qemu-storage-crash and qemu-local-console-gate all pass. Note this gate is not part of the CI workflow, which is why a subsystem that could never persist went unnoticed. Co-Authored-By: Claude Opus 5 --- kernel/core/kmain.c | 14 ++++++++-- kernel/include/xaios/persistence.h | 5 ++++ kernel/runtime/persistence.c | 44 ++++++++++++++++++++++++++---- 3 files changed, 56 insertions(+), 7 deletions(-) diff --git a/kernel/core/kmain.c b/kernel/core/kmain.c index 9cbcd3fe..349e7ad5 100644 --- a/kernel/core/kmain.c +++ b/kernel/core/kmain.c @@ -353,7 +353,15 @@ void kmain(const xaios_boot_info_t *boot) { virtio_block_self_test(); boot_ui_update(51U, "boot storage validation", "initial filesystem", 3U); initramfs_self_test(); - if (virtio_block_is_read_only() != 0U) { + /* Snapshot state must land on the durable volume, not on vblk0: that device + carries the initramfs/test image and the QEMU launcher attaches it with + snapshot=on, so its writes are thrown away when the machine stops. Bind + the dedicated persistent slot before the self-test runs, and reuse the + same handle for MutableFS below. */ + if (virtio_block_open_slot(1U, &g_persistent_handle) == XAIOS_OK) { + persistence_bind_block_device(g_persistent_handle); + } + if (virtio_block_is_read_only() != 0U && g_persistent_handle == 0) { persistence_runtime_init(); klog("persistence: writable self-test skipped boot device is read-only\n"); klog("mutable-fs: writable self-test deferred no persistent block device\n"); @@ -380,7 +388,9 @@ void kmain(const xaios_boot_info_t *boot) { /* vblk0 remains the immutable initramfs/test image. Open the dedicated * second VirtIO block device for durable MutableFS state. */ xaios_status_t virtio_status = - virtio_block_open_slot(1U, &g_persistent_handle); + g_persistent_handle != 0 + ? XAIOS_OK + : virtio_block_open_slot(1U, &g_persistent_handle); persistent_status = virtio_status == XAIOS_OK ? mutable_fs_mount_device("/dev/vblk1") : virtio_status; diff --git a/kernel/include/xaios/persistence.h b/kernel/include/xaios/persistence.h index 5dbbbf9e..06ab0ec5 100644 --- a/kernel/include/xaios/persistence.h +++ b/kernel/include/xaios/persistence.h @@ -3,6 +3,7 @@ #include #include +#include typedef enum xaios_snapshot_kind { XAIOS_SNAPSHOT_BOOT_CONFIG = 1, @@ -12,6 +13,10 @@ typedef enum xaios_snapshot_kind { XAIOS_SNAPSHOT_UPDATE = 5, } xaios_snapshot_kind_t; +/* Bind the durable volume that snapshot state is written to. Without this + the default block device is used, which on QEMU is the snapshot-backed + test image whose writes never survive a restart. */ +void persistence_bind_block_device(virtio_block_handle_t *handle); void persistence_runtime_init(void); xaios_status_t persistence_snapshot_create(xaios_snapshot_kind_t kind, uint32_t owner_id, diff --git a/kernel/runtime/persistence.c b/kernel/runtime/persistence.c index f77a5f32..8453e769 100644 --- a/kernel/runtime/persistence.c +++ b/kernel/runtime/persistence.c @@ -166,6 +166,38 @@ static uint64_t persistence_sector_checksum(xaios_persistence_disk_sector_t *sec return checksum; } +/* Durable snapshot storage. + vblk0 is the initramfs/test image and the QEMU launcher attaches it with + snapshot=on, so every write to it is discarded when the machine stops. The + snapshot sector has to live on the same durable volume MutableFS uses, which + is why PERSISTENCE_SECTOR (3000) sits just below MFS_START_SECTOR (3072). + Until a durable device is bound the default device is used, which keeps the + pre-storage self-test working on machines that expose no writable volume. */ +static virtio_block_handle_t *g_persistence_device; + +void persistence_bind_block_device(virtio_block_handle_t *handle) { + g_persistence_device = handle; + klog("persistence: durable device %s\n", + handle != 0 ? "bound" : "cleared"); +} + +static xaios_status_t persistence_read_sector(void *buffer, uint64_t size) { + if (g_persistence_device != 0) { + return virtio_block_read_sector_h(g_persistence_device, PERSISTENCE_SECTOR, + buffer, size); + } + return virtio_block_read_sector(PERSISTENCE_SECTOR, buffer, size); +} + +static xaios_status_t persistence_write_sector(const void *buffer, + uint64_t size) { + if (g_persistence_device != 0) { + return virtio_block_write_sector_h(g_persistence_device, PERSISTENCE_SECTOR, + buffer, size); + } + return virtio_block_write_sector(PERSISTENCE_SECTOR, buffer, size); +} + static xaios_status_t persistence_flush_to_disk(void) { xaios_persistence_disk_sector_t sector; bytes_zero(§or, sizeof(sector)); @@ -191,8 +223,7 @@ static xaios_status_t persistence_flush_to_disk(void) { sector.record_count = out; sector.checksum = persistence_sector_checksum(§or); - if (virtio_block_write_sector(PERSISTENCE_SECTOR, §or, - sizeof(sector)) != XAIOS_OK) { + if (persistence_write_sector(§or, sizeof(sector)) != XAIOS_OK) { ++g_reject_count; return XAIOS_ERR_IO; } @@ -206,7 +237,7 @@ static xaios_status_t persistence_flush_to_disk(void) { static xaios_status_t persistence_load_from_disk(void) { xaios_persistence_disk_sector_t sector; bytes_zero(§or, sizeof(sector)); - if (virtio_block_read_sector(PERSISTENCE_SECTOR, §or, sizeof(sector)) != + if (persistence_read_sector(§or, sizeof(sector)) != XAIOS_OK) { ++g_reject_count; return XAIOS_ERR_IO; @@ -260,7 +291,7 @@ static xaios_status_t persistence_load_from_disk(void) { static void persistence_probe_existing_disk_state(void) { xaios_persistence_disk_sector_t sector; bytes_zero(§or, sizeof(sector)); - if (virtio_block_read_sector(PERSISTENCE_SECTOR, §or, sizeof(sector)) != + if (persistence_read_sector(§or, sizeof(sector)) != XAIOS_OK) { return; } @@ -376,7 +407,10 @@ uint64_t persistence_checksum_error_count(void) { void persistence_self_test(void) { persistence_runtime_init(); kassert(sizeof(xaios_persistence_disk_sector_t) <= PERSISTENCE_SECTOR_SIZE); - kassert(PERSISTENCE_SECTOR < virtio_block_capacity_sectors()); + kassert(PERSISTENCE_SECTOR < + (g_persistence_device != 0 + ? virtio_block_capacity_sectors_h(g_persistence_device) + : virtio_block_capacity_sectors())); klog("persistence: self-test scratch sector=%lu sectors=1 initramfs_payload_start=4096\n", PERSISTENCE_SECTOR); persistence_probe_existing_disk_state(); From feecf0533fc649bf1f17adbe5c76bd2ff4693c6d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 13:33:53 +0700 Subject: [PATCH 15/16] Serialise MutableFS entry points MutableFS had no mutual exclusion of any kind. Its node table, open-file table and block bitmap were mutated without a lock while being reachable from every CPU through the filesystem syscalls and from kernel services, on a kernel that runs four vCPUs in the interoperability gates and has no global syscall lock -- only a socket-table lock. Every neighbouring subsystem locks: network_stack, klog, scheduler and kheap all take spinlocks. This one did not. That is the most plausible explanation for the FreeBSD gate's two different intermittent failures, which are the same underlying event: a MutableFS write returning an error. One run failed an SFTP write, reported as SSH_FX_FAILURE "Write failed"; the next failed the known_hosts write inside verify_known_host, which surfaces as "host key verification failed". Both are timing dependent, both only appear under slow emulation where the interleaving differs, and both pass on a fast native host. Each public entry point becomes a thin wrapper that takes one lock around the existing body, now named *_locked. mutable_fs_mount_persistent calls mutable_fs_mount_device_locked so the pair stays inside a single acquisition. mutable_fs_self_test is deliberately left unwrapped: it drives these same entry points and runs single threaded during boot. The read-only counter accessors are left alone; they return a single word. Re-entrancy was checked rather than assumed. MutableFS calls only block_device_*, virtio_block_*, klog and kassert, none of which re-enter it. klog_ring_write is purely in-memory; the klog_ring functions that do touch MutableFS are klog_ring_init, klog_flush and klog_rotate, and klog_flush is reached from klog_level only at panic level, which MutableFS never emits. This does not prove the FreeBSD flakiness is gone -- that needs repeated CI runs, and the gate now captures the kernel's own diagnostics if it recurs. Verified: qemu-smoke, filesystem milestone 62 (three consecutive runs), qemu-storage-crash with all metadata kill points recovered, qemu-local-console-gate, qemu-persistence-reboot and compile-check all pass. Co-Authored-By: Claude Opus 5 --- kernel/fs/mutable_fs.c | 231 ++++++++++++++++++++++++++++++++++------- 1 file changed, 195 insertions(+), 36 deletions(-) diff --git a/kernel/fs/mutable_fs.c b/kernel/fs/mutable_fs.c index d8f28a46..727e9ec3 100644 --- a/kernel/fs/mutable_fs.c +++ b/kernel/fs/mutable_fs.c @@ -2,6 +2,7 @@ #include #include #include +#include #include #define MFS_MAGIC "XAIOSMFS2" @@ -202,6 +203,7 @@ static uint64_t g_close_count; static uint64_t g_metadata_verified_checksum; static uint8_t g_metadata_buffer[MFS_V5_METADATA_SECTORS * MFS_SECTOR_SIZE]; +static xaios_spinlock_t g_mutable_fs_lock = XAIOS_SPINLOCK_INIT; static uint8_t g_file_buffer[MFS_V5_MAX_FILE_BYTES]; static char g_path_transaction[MFS_V5_MAX_NODES][MFS_PATH_MAX]; @@ -1721,8 +1723,7 @@ static xaios_status_t write_pending_journal_file(const char *path, return write_journal(&journal); } -xaios_status_t mutable_fs_record_service_state(const char *name, - const char *state) { +static xaios_status_t mutable_fs_record_service_state_locked(const char *name, const char *state) { char path[MFS_PATH_MAX]; char record[256]; uint64_t path_offset = 0; @@ -1752,8 +1753,7 @@ xaios_status_t mutable_fs_record_service_state(const char *name, return status; } -xaios_status_t mutable_fs_record_workspace_state(uint32_t workspace_id, - const char *revision) { +static xaios_status_t mutable_fs_record_workspace_state_locked(uint32_t workspace_id, const char *revision) { char path[MFS_PATH_MAX]; char record[256]; uint64_t path_offset = 0; @@ -1785,7 +1785,7 @@ xaios_status_t mutable_fs_record_workspace_state(uint32_t workspace_id, return status; } -xaios_status_t mutable_fs_record_update_state(const char *policy) { +static xaios_status_t mutable_fs_record_update_state_locked(const char *policy) { char record[256]; uint64_t record_offset = 0; bytes_zero(record, sizeof(record)); @@ -1806,10 +1806,7 @@ xaios_status_t mutable_fs_record_update_state(const char *policy) { return status; } -xaios_status_t mutable_fs_record_update_transaction(uint32_t generation, - const char *state, - const char *target, - const char *rollback_label) { +static xaios_status_t mutable_fs_record_update_transaction_locked(uint32_t generation, const char *state, const char *target, const char *rollback_label) { char record[256]; uint64_t record_offset = 0; bytes_zero(record, sizeof(record)); @@ -1842,11 +1839,7 @@ xaios_status_t mutable_fs_record_update_transaction(uint32_t generation, return status; } -xaios_status_t mutable_fs_record_admin_status(const char *service, - const char *state, - uint32_t starts, - uint32_t restarts, - uint32_t logs) { +static xaios_status_t mutable_fs_record_admin_status_locked(const char *service, const char *state, uint32_t starts, uint32_t restarts, uint32_t logs) { char record[256]; uint64_t record_offset = 0; bytes_zero(record, sizeof(record)); @@ -1881,46 +1874,43 @@ xaios_status_t mutable_fs_record_admin_status(const char *service, return status; } -xaios_status_t mutable_fs_commit(const char *label) { +static xaios_status_t mutable_fs_commit_locked(const char *label) { return commit_snapshot(label); } -xaios_status_t mutable_fs_rollback(void) { +static xaios_status_t mutable_fs_rollback_locked(void) { return rollback_snapshot(); } -xaios_status_t mutable_fs_mkdir(const char *path) { +static xaios_status_t mutable_fs_mkdir_locked(const char *path) { return create_dir(path); } -xaios_status_t mutable_fs_write(const char *path, const void *data, - uint64_t size) { +static xaios_status_t mutable_fs_write_locked(const char *path, const void *data, uint64_t size) { return write_file(path, data, size); } -xaios_status_t mutable_fs_read(const char *path, void *buffer, - uint64_t buffer_size, uint64_t *out_size) { +static xaios_status_t mutable_fs_read_locked(const char *path, void *buffer, uint64_t buffer_size, uint64_t *out_size) { return read_file(path, buffer, buffer_size, out_size); } -xaios_status_t mutable_fs_delete(const char *path) { +static xaios_status_t mutable_fs_delete_locked(const char *path) { return delete_node(path); } -xaios_status_t mutable_fs_delete_tree(const char *path) { +static xaios_status_t mutable_fs_delete_tree_locked(const char *path) { return delete_tree(path); } -xaios_status_t mutable_fs_rename(const char *old_path, const char *new_path) { +static xaios_status_t mutable_fs_rename_locked(const char *old_path, const char *new_path) { return rename_node(old_path, new_path); } -xaios_status_t mutable_fs_stat(const char *path, xaios_mfs_stat_t *stat) { +static xaios_status_t mutable_fs_stat_locked(const char *path, xaios_mfs_stat_t *stat) { return stat_node(path, stat); } -xaios_status_t mutable_fs_list(const char *path, char *buffer, - uint64_t buffer_size, uint64_t *out_size) { +static xaios_status_t mutable_fs_list_locked(const char *path, char *buffer, uint64_t buffer_size, uint64_t *out_size) { return list_dir(path, buffer, buffer_size, out_size); } @@ -1932,7 +1922,7 @@ static xaios_mfs_file_handle_t *handle_for_fd(uint32_t fd) { return handle->in_use != 0 ? handle : 0; } -int64_t mutable_fs_open(const char *path, uint32_t flags) { +static int64_t mutable_fs_open_locked(const char *path, uint32_t flags) { char normalized[MFS_PATH_MAX]; if (normalize_path(path, normalized) != XAIOS_OK || (flags & (XAIOS_MFS_OPEN_READ | XAIOS_MFS_OPEN_WRITE)) == 0 || @@ -1987,7 +1977,7 @@ int64_t mutable_fs_open(const char *path, uint32_t flags) { return (int64_t)XAIOS_ERR_NO_MEMORY; } -int64_t mutable_fs_read_fd(uint32_t fd, void *buffer, uint64_t size) { +static int64_t mutable_fs_read_fd_locked(uint32_t fd, void *buffer, uint64_t size) { xaios_mfs_file_handle_t *handle = handle_for_fd(fd); if (handle == 0 || buffer == 0 || size == 0 || (handle->flags & XAIOS_MFS_OPEN_READ) == 0) { @@ -2014,7 +2004,7 @@ int64_t mutable_fs_read_fd(uint32_t fd, void *buffer, uint64_t size) { return (int64_t)copy; } -int64_t mutable_fs_write_fd(uint32_t fd, const void *buffer, uint64_t size) { +static int64_t mutable_fs_write_fd_locked(uint32_t fd, const void *buffer, uint64_t size) { xaios_mfs_file_handle_t *handle = handle_for_fd(fd); if (handle == 0 || buffer == 0 || size == 0 || (handle->flags & XAIOS_MFS_OPEN_WRITE) == 0 || @@ -2054,7 +2044,7 @@ int64_t mutable_fs_write_fd(uint32_t fd, const void *buffer, uint64_t size) { return (int64_t)size; } -xaios_status_t mutable_fs_seek(uint32_t fd, uint64_t offset) { +static xaios_status_t mutable_fs_seek_locked(uint32_t fd, uint64_t offset) { xaios_mfs_file_handle_t *handle = handle_for_fd(fd); if (handle == 0 || offset > g_active_max_file_bytes) { ++g_reject_count; @@ -2064,7 +2054,7 @@ xaios_status_t mutable_fs_seek(uint32_t fd, uint64_t offset) { return XAIOS_OK; } -xaios_status_t mutable_fs_close(uint32_t fd) { +static xaios_status_t mutable_fs_close_locked(uint32_t fd) { xaios_mfs_file_handle_t *handle = handle_for_fd(fd); if (handle == 0) { ++g_reject_count; @@ -2114,7 +2104,7 @@ static xaios_status_t mount_failure(xaios_block_device_t *device, return status; } -xaios_status_t mutable_fs_mount_device(const char *identifier) { +static xaios_status_t mutable_fs_mount_device_locked(const char *identifier) { if (g_persistent_device != 0) return XAIOS_ERR_BUSY; xaios_block_device_t *device = 0; xaios_status_t status = block_device_open(identifier, &device); @@ -2198,7 +2188,7 @@ xaios_status_t mutable_fs_mount_device(const char *identifier) { return XAIOS_OK; } -xaios_status_t mutable_fs_mount_persistent(uint32_t slot) { +static xaios_status_t mutable_fs_mount_persistent_locked(uint32_t slot) { char identifier[16] = "/dev/vblk"; uint32_t value = slot; uint32_t digits = 1U; @@ -2212,7 +2202,7 @@ xaios_status_t mutable_fs_mount_persistent(uint32_t slot) { identifier[9U + digits - 1U - index] = (char)('0' + slot % 10U); slot /= 10U; } - return mutable_fs_mount_device(identifier); + return mutable_fs_mount_device_locked(identifier); } static void fsck_count_file_blocks( @@ -2233,7 +2223,7 @@ static void fsck_count_file_blocks( } } -xaios_mfs_fsck_result_t mutable_fs_fsck(void) { +static xaios_mfs_fsck_result_t mutable_fs_fsck_locked(void) { xaios_mfs_fsck_result_t result; uint8_t references[MFS_V5_DATA_SECTORS]; bytes_zero(&result, sizeof(result)); @@ -2271,6 +2261,175 @@ xaios_mfs_fsck_result_t mutable_fs_fsck(void) { return result; } + +/* Serialised public entry points. + The volume is reachable from every CPU through the filesystem syscalls and + from kernel services, yet its node table, open-file table and block bitmap + were mutated with no mutual exclusion at all. Each entry point now runs + under one lock. The bodies above assume the lock is already held and must + not be called directly. mutable_fs_self_test stays outside deliberately: it + drives these same entry points and runs single threaded during boot. */ +xaios_status_t mutable_fs_record_service_state(const char *name, const char *state) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_record_service_state_locked(name, state); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_record_workspace_state(uint32_t workspace_id, const char *revision) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_record_workspace_state_locked(workspace_id, revision); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_record_update_state(const char *policy) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_record_update_state_locked(policy); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_record_update_transaction(uint32_t generation, const char *state, const char *target, const char *rollback_label) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_record_update_transaction_locked(generation, state, target, rollback_label); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_record_admin_status(const char *service, const char *state, uint32_t starts, uint32_t restarts, uint32_t logs) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_record_admin_status_locked(service, state, starts, restarts, logs); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_commit(const char *label) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_commit_locked(label); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_rollback(void) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_rollback_locked(); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_mkdir(const char *path) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_mkdir_locked(path); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_write(const char *path, const void *data, uint64_t size) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_write_locked(path, data, size); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_read(const char *path, void *buffer, uint64_t buffer_size, uint64_t *out_size) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_read_locked(path, buffer, buffer_size, out_size); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_delete(const char *path) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_delete_locked(path); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_delete_tree(const char *path) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_delete_tree_locked(path); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_rename(const char *old_path, const char *new_path) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_rename_locked(old_path, new_path); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_stat(const char *path, xaios_mfs_stat_t *stat) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_stat_locked(path, stat); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_list(const char *path, char *buffer, uint64_t buffer_size, uint64_t *out_size) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_list_locked(path, buffer, buffer_size, out_size); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +int64_t mutable_fs_open(const char *path, uint32_t flags) { + xaios_spin_lock(&g_mutable_fs_lock); + int64_t result = mutable_fs_open_locked(path, flags); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +int64_t mutable_fs_read_fd(uint32_t fd, void *buffer, uint64_t size) { + xaios_spin_lock(&g_mutable_fs_lock); + int64_t result = mutable_fs_read_fd_locked(fd, buffer, size); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +int64_t mutable_fs_write_fd(uint32_t fd, const void *buffer, uint64_t size) { + xaios_spin_lock(&g_mutable_fs_lock); + int64_t result = mutable_fs_write_fd_locked(fd, buffer, size); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_seek(uint32_t fd, uint64_t offset) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_seek_locked(fd, offset); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_close(uint32_t fd) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_close_locked(fd); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_mount_device(const char *identifier) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_mount_device_locked(identifier); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_status_t mutable_fs_mount_persistent(uint32_t slot) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_status_t result = mutable_fs_mount_persistent_locked(slot); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + +xaios_mfs_fsck_result_t mutable_fs_fsck(void) { + xaios_spin_lock(&g_mutable_fs_lock); + xaios_mfs_fsck_result_t result = mutable_fs_fsck_locked(); + xaios_spin_unlock(&g_mutable_fs_lock); + return result; +} + void mutable_fs_self_test(void) { kassert(sizeof(xaios_mfs_journal_t) == MFS_SECTOR_SIZE); kassert(sizeof(xaios_mfs_disk_t) <= MFS_METADATA_SECTORS * MFS_SECTOR_SIZE); From 92e9000562e0eda41b711fbce78fc056d557cc3f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Borchert?= Date: Sun, 23 Aug 2026 14:51:48 +0700 Subject: [PATCH 16/16] Key process resources on an incarnation, not a reusable pid The FreeBSD gate's intermittent failures were file handles being destroyed under a live session. The guest named it once the gate started capturing kernel diagnostics: mutable-fs: open fd=2 path=/tmp/freebsd-sftp flags=0xe mutable-fs: write-fd fd=2 bytes=35 cursor=35 user: rejected syscall=14 arg0=0x2 reason=fs-close-denied The write landed; the close was refused. Transient processes and SSH child channels all draw from the first free process-table slot, so in one run a single pid -- 32 -- was issued and reaped nineteen times, and every child channel ran as child=32. VFS handles and kernel sockets were keyed on that pid, and syscall_release_process_resources sweeps every resource owned by it when a process is reaped. Any transient exiting while an SFTP or outbound SSH session had a file open therefore released that session's handle, and the session failed on its next operation. Which run it hit depended purely on overlap, which is why it only appeared under slow emulation and never reproduced on a fast host. Give each process incarnation a token that is never issued twice, and key VFS handles and kernel sockets on it instead of the pid. The pid keeps its meaning for scheduling, reporting and diagnostics; only ownership moves. The VFS and socket layers are unchanged: they already compared an opaque owner value, and now that value no longer collides. copy_process enumerates fields by hand, so the token also has to be copied there. It was missed at first and every socket allocation failed with owner zero, taking sshd's listener down with boot error 2301 -- worth knowing about before adding another field to that struct. Verified: the FreeBSD bidirectional suite passes locally end to end, and qemu-smoke, qemu-local-console-gate, qemu-storage-crash, qemu-regression-suite, filesystem milestone 62, qemu-persistence-reboot and compile-check all pass. Co-Authored-By: Claude Opus 5 --- kernel/include/xaios/syscall.h | 5 ++- kernel/include/xaios/user.h | 5 +++ kernel/user/syscall.c | 80 +++++++++++++++++----------------- kernel/user/user.c | 14 +++++- 4 files changed, 62 insertions(+), 42 deletions(-) diff --git a/kernel/include/xaios/syscall.h b/kernel/include/xaios/syscall.h index e0e76e7f..c4e1b288 100644 --- a/kernel/include/xaios/syscall.h +++ b/kernel/include/xaios/syscall.h @@ -251,7 +251,10 @@ typedef struct xaios_syscall_control_query_request { uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, uint64_t arg2); void syscall_self_test(void); -void syscall_release_process_resources(uint32_t owner_pid); +/* Release everything owned by one process incarnation. Takes the owner token, + not the pid: process-table slots are reused immediately, so a pid cannot + distinguish a reaped process from the next one occupying its slot. */ +void syscall_release_process_resources(uint32_t owner_token); uint64_t syscall_control_plane_count(void); uint64_t syscall_control_plane_denial_count(void); uint64_t syscall_service_descriptor_read_count(void); diff --git a/kernel/include/xaios/user.h b/kernel/include/xaios/user.h index e7e87af1..a28603cb 100644 --- a/kernel/include/xaios/user.h +++ b/kernel/include/xaios/user.h @@ -26,6 +26,11 @@ typedef enum xaios_user_process_state { typedef struct xaios_user_process { uint32_t pid; + /* Process-table slots, and therefore pids, are handed straight back to the + next transient process. Resource ownership must not be keyed on a value + that is reused within milliseconds, so every incarnation also carries a + token that is never issued twice. */ + uint32_t owner_token; uint32_t parent_pid; const char *name; xaios_user_process_state_t state; diff --git a/kernel/user/syscall.c b/kernel/user/syscall.c index 3db48ecb..f768500c 100644 --- a/kernel/user/syscall.c +++ b/kernel/user/syscall.c @@ -180,7 +180,7 @@ typedef struct kernel_socket { uint8_t bind_addr[16]; /* bind address (16 bytes for IPv6) */ uint8_t peer_addr[16]; /* peer address (connected sockets) */ uint16_t peer_port; - uint32_t owner_pid; + uint32_t owner_token; uint64_t id; /* unique socket ID from alloc */ } kernel_socket_t; @@ -210,8 +210,8 @@ static void kernel_socket_table_init(void) { } static uint64_t kernel_socket_alloc(uint32_t type, uint16_t port, - uint32_t owner_pid) { - if (g_kernel_sockets == 0 || owner_pid == 0U) return 0U; + uint32_t owner_token) { + if (g_kernel_sockets == 0 || owner_token == 0U) return 0U; xaios_spin_lock(&g_kernel_socket_lock); if (g_total_connections >= g_kernel_socket_capacity) { xaios_spin_unlock(&g_kernel_socket_lock); @@ -241,7 +241,7 @@ static uint64_t kernel_socket_alloc(uint32_t type, uint16_t port, if (g_kernel_sockets[i].state == 0) { g_kernel_sockets[i].state = type; g_kernel_sockets[i].port = port; - g_kernel_sockets[i].owner_pid = owner_pid; + g_kernel_sockets[i].owner_token = owner_token; g_kernel_sockets[i].id = g_socket_next_id; g_total_connections++; uint64_t id = g_socket_next_id++; @@ -255,10 +255,10 @@ static uint64_t kernel_socket_alloc(uint32_t type, uint16_t port, } static kernel_socket_t *kernel_socket_find_owned_locked(uint64_t sockfd, - uint32_t owner_pid) { + uint32_t owner_token) { for (uint32_t i = 0; i < g_kernel_socket_capacity; ++i) { if (g_kernel_sockets[i].state != 0 && g_kernel_sockets[i].id == sockfd && - g_kernel_sockets[i].owner_pid == owner_pid) { + g_kernel_sockets[i].owner_token == owner_token) { return &g_kernel_sockets[i]; } } @@ -266,11 +266,11 @@ static kernel_socket_t *kernel_socket_find_owned_locked(uint64_t sockfd, } static xaios_status_t kernel_socket_snapshot_owned(uint64_t sockfd, - uint32_t owner_pid, + uint32_t owner_token, kernel_socket_t *snapshot) { if (snapshot == 0) return XAIOS_ERR_INVALID; xaios_spin_lock(&g_kernel_socket_lock); - kernel_socket_t *socket = kernel_socket_find_owned_locked(sockfd, owner_pid); + kernel_socket_t *socket = kernel_socket_find_owned_locked(sockfd, owner_token); if (socket == 0) { xaios_spin_unlock(&g_kernel_socket_lock); return XAIOS_ERR_INVALID; @@ -280,16 +280,16 @@ static xaios_status_t kernel_socket_snapshot_owned(uint64_t sockfd, return XAIOS_OK; } -static xaios_status_t kernel_socket_free(uint64_t sockfd, uint32_t owner_pid) { +static xaios_status_t kernel_socket_free(uint64_t sockfd, uint32_t owner_token) { xaios_spin_lock(&g_kernel_socket_lock); - kernel_socket_t *socket = kernel_socket_find_owned_locked(sockfd, owner_pid); + kernel_socket_t *socket = kernel_socket_find_owned_locked(sockfd, owner_token); if (socket != 0) { socket->state = 0; socket->port = 0; socket->family = 0; socket->protocol = 0; socket->peer_port = 0; - socket->owner_pid = 0; + socket->owner_token = 0; socket->id = 0; for (uint32_t j = 0; j < 16; ++j) { socket->bind_addr[j] = 0; @@ -303,9 +303,9 @@ static xaios_status_t kernel_socket_free(uint64_t sockfd, uint32_t owner_pid) { return XAIOS_ERR_INVALID; } -void syscall_release_process_resources(uint32_t owner_pid) { - if (owner_pid == 0U) return; - (void)vfs_release_owner(owner_pid); +void syscall_release_process_resources(uint32_t owner_token) { + if (owner_token == 0U) return; + (void)vfs_release_owner(owner_token); if (g_kernel_sockets == 0) return; for (;;) { kernel_socket_t snapshot; @@ -313,7 +313,7 @@ void syscall_release_process_resources(uint32_t owner_pid) { xaios_spin_lock(&g_kernel_socket_lock); for (uint32_t i = 0; i < g_kernel_socket_capacity; ++i) { if (g_kernel_sockets[i].state != 0U && - g_kernel_sockets[i].owner_pid == owner_pid) { + g_kernel_sockets[i].owner_token == owner_token) { snapshot = g_kernel_sockets[i]; found = 1U; break; @@ -335,7 +335,7 @@ void syscall_release_process_resources(uint32_t owner_pid) { } else if (snapshot.state == KERNEL_SOCK_DATAGRAM) { network_stack_unregister_udp_listener(snapshot.port); } - (void)kernel_socket_free(snapshot.id, owner_pid); + (void)kernel_socket_free(snapshot.id, owner_token); } } @@ -731,7 +731,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, return reject_syscall(syscall, arg0, arg1, "fs-open-read-denied"); } const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; int64_t fd = vfs_open(path, (uint32_t)arg2, owner_id); if (fd < 0) { return reject_syscall(syscall, arg0, arg1, "fs-open-denied"); @@ -746,7 +746,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, return reject_syscall(syscall, arg0, arg1, "bad-fs-read-buffer"); } const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; int64_t bytes = vfs_read((uint32_t)arg0, owner_id, (void *)(uintptr_t)arg1, arg2); if (bytes < 0) { @@ -772,7 +772,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, return reject_syscall(syscall, arg0, arg1, "fs-write-secret-denied"); } const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; int64_t bytes = vfs_write((uint32_t)arg0, owner_id, write_snapshot, arg2); kheap_free(write_snapshot); @@ -799,7 +799,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, return reject_syscall(syscall, arg0, arg1, "bad-fs-positional-buffer"); } const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; int64_t bytes; if (syscall == XAIOS_SYSCALL_FS_PREAD) { bytes = vfs_pread((uint32_t)request.fd, owner_id, @@ -835,7 +835,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, if (syscall == XAIOS_SYSCALL_FS_FSYNC) { const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; if (arg0 > UINT32_MAX || vfs_fsync((uint32_t)arg0, owner_id) != XAIOS_OK) { return reject_syscall(syscall, arg0, arg1, "fs-fsync-denied"); @@ -845,7 +845,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, if (syscall == XAIOS_SYSCALL_FS_SEEK) { const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; if (arg0 > UINT32_MAX || vfs_seek((uint32_t)arg0, owner_id, arg1) != XAIOS_OK) { return reject_syscall(syscall, arg0, arg1, "fs-seek-denied"); @@ -855,7 +855,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, if (syscall == XAIOS_SYSCALL_FS_CLOSE) { const xaios_user_process_t *current = user_current_process(); - uint32_t owner_id = current != 0 ? current->pid : 0U; + uint32_t owner_id = current != 0 ? current->owner_token : 0U; if (arg0 > UINT32_MAX || vfs_close((uint32_t)arg0, owner_id) != XAIOS_OK) { return reject_syscall(syscall, arg0, arg1, "fs-close-denied"); @@ -1510,9 +1510,9 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, "net-connect-handshake-failed"); } const xaios_user_process_t *process = user_current_process(); - uint32_t owner_pid = process != 0 ? process->pid : 0U; + uint32_t owner_token = process != 0 ? process->owner_token : 0U; uint64_t sockfd = kernel_socket_alloc(KERNEL_SOCK_CONNECTED, - (uint16_t)request.port, owner_pid); + (uint16_t)request.port, owner_token); if (sockfd == 0U) { (void)network_stack_tcp_abort_flow(flow_id); return reject_syscall(syscall, arg0, arg1, @@ -1520,7 +1520,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, } xaios_spin_lock(&g_kernel_socket_lock); kernel_socket_t *socket = - kernel_socket_find_owned_locked(sockfd, owner_pid); + kernel_socket_find_owned_locked(sockfd, owner_token); kassert(socket != 0); socket->protocol = XAIOS_NETWORK_PROTOCOL_TCP; socket->family = remote_addr.family; @@ -1568,18 +1568,18 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, } } const xaios_user_process_t *process = user_current_process(); - uint32_t owner_pid = process != 0 ? process->pid : 0U; + uint32_t owner_token = process != 0 ? process->owner_token : 0U; uint32_t socket_type = protocol == XAIOS_NETWORK_PROTOCOL_UDP ? KERNEL_SOCK_DATAGRAM : KERNEL_SOCK_LISTEN; uint64_t sockfd = kernel_socket_alloc(socket_type, (uint16_t)request.port, - owner_pid); + owner_token); if (sockfd == 0) { return reject_syscall(syscall, arg0, arg1, "net-listen-no-memory"); } xaios_spin_lock(&g_kernel_socket_lock); kernel_socket_t *socket = - kernel_socket_find_owned_locked(sockfd, owner_pid); + kernel_socket_find_owned_locked(sockfd, owner_token); kassert(socket != 0); socket->protocol = (uint8_t)protocol; if (request.addr_ptr != 0U) { @@ -1619,9 +1619,9 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, return reject_syscall(syscall, arg0, arg1, "net-accept-denied"); } const xaios_user_process_t *process = user_current_process(); - uint32_t owner_pid = process != 0 ? process->pid : 0U; + uint32_t owner_token = process != 0 ? process->owner_token : 0U; kernel_socket_t listener; - if (kernel_socket_snapshot_owned(request.sockfd, owner_pid, &listener) != + if (kernel_socket_snapshot_owned(request.sockfd, owner_token, &listener) != XAIOS_OK || listener.state != KERNEL_SOCK_LISTEN || listener.protocol != XAIOS_NETWORK_PROTOCOL_TCP) { @@ -1641,7 +1641,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, } /* Allocate connected socket */ uint64_t connfd = - kernel_socket_alloc(KERNEL_SOCK_CONNECTED, listen_port, owner_pid); + kernel_socket_alloc(KERNEL_SOCK_CONNECTED, listen_port, owner_token); if (connfd == 0) { network_stack_tcp_close_flow(flow_id); return reject_syscall(syscall, arg0, arg1, "net-accept-no-memory"); @@ -1649,7 +1649,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, /* Store peer info on the socket */ xaios_spin_lock(&g_kernel_socket_lock); kernel_socket_t *socket = - kernel_socket_find_owned_locked(connfd, owner_pid); + kernel_socket_find_owned_locked(connfd, owner_token); kassert(socket != 0); socket->peer_port = peer_port; socket->family = peer_addr.family; @@ -1700,9 +1700,9 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, } network_poll_tick(); const xaios_user_process_t *process = user_current_process(); - uint32_t owner_pid = process != 0 ? process->pid : 0U; + uint32_t owner_token = process != 0 ? process->owner_token : 0U; kernel_socket_t socket_snapshot; - if (kernel_socket_snapshot_owned(request.sockfd, owner_pid, + if (kernel_socket_snapshot_owned(request.sockfd, owner_token, &socket_snapshot) != XAIOS_OK) { return reject_syscall(syscall, arg0, arg1, "net-recv-no-socket"); } @@ -1766,9 +1766,9 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, } network_poll_tick(); const xaios_user_process_t *process = user_current_process(); - uint32_t owner_pid = process != 0 ? process->pid : 0U; + uint32_t owner_token = process != 0 ? process->owner_token : 0U; kernel_socket_t socket_snapshot; - if (kernel_socket_snapshot_owned(request.sockfd, owner_pid, + if (kernel_socket_snapshot_owned(request.sockfd, owner_token, &socket_snapshot) != XAIOS_OK) { return reject_syscall(syscall, arg0, arg1, "net-send-no-socket"); } @@ -1807,9 +1807,9 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, if (syscall == XAIOS_SYSCALL_NET_CLOSE) { const xaios_user_process_t *process = user_current_process(); - uint32_t owner_pid = process != 0 ? process->pid : 0U; + uint32_t owner_token = process != 0 ? process->owner_token : 0U; kernel_socket_t socket_snapshot; - if (kernel_socket_snapshot_owned(arg0, owner_pid, &socket_snapshot) != + if (kernel_socket_snapshot_owned(arg0, owner_token, &socket_snapshot) != XAIOS_OK) { return reject_syscall(syscall, arg0, arg1, "net-close-no-socket"); } @@ -1827,7 +1827,7 @@ uint64_t syscall_dispatch(uint64_t syscall, uint64_t arg0, uint64_t arg1, } else if (socket_snapshot.state == KERNEL_SOCK_DATAGRAM) { network_stack_unregister_udp_listener(socket_snapshot.port); } - kassert(kernel_socket_free(arg0, owner_pid) == XAIOS_OK); + kassert(kernel_socket_free(arg0, owner_token) == XAIOS_OK); klog("syscall: net_close sockfd=%lu\n", arg0); return XAIOS_OK; } diff --git a/kernel/user/user.c b/kernel/user/user.c index fef2f114..64d72a04 100644 --- a/kernel/user/user.c +++ b/kernel/user/user.c @@ -16,6 +16,16 @@ #define USER_STACK_PAGES UINT64_C(64) #define XAIOS_TRANSIENT_PID_FIRST 32U +/* Monotonic, never reused. Zero is reserved for "no owner", so the counter + skips it on wrap. */ +static uint32_t g_owner_token_next = 1U; + +static uint32_t user_next_owner_token(void) { + uint32_t token = g_owner_token_next++; + if (g_owner_token_next == 0U) g_owner_token_next = 1U; + return token; +} + static void bytes_zero(void *buffer, uint64_t size) { uint8_t *bytes = (uint8_t *)buffer; for (uint64_t i = 0; i < size; ++i) { @@ -85,6 +95,7 @@ extern uint64_t aarch64_enter_user(uint64_t entry, uint64_t stack, static void copy_process(xaios_user_process_t *dst, const xaios_user_process_t *src) { dst->pid = src->pid; + dst->owner_token = src->owner_token; dst->parent_pid = src->parent_pid; dst->name = src->name; dst->state = src->state; @@ -681,6 +692,7 @@ xaios_status_t user_load_process(const xaios_initramfs_file_t *file, reset_process_slot(process); process->pid = pid; + process->owner_token = user_next_owner_token(); process->name = file->path; process->capability_mask = capability_mask; @@ -1085,7 +1097,7 @@ void user_process_reclaim_address_space(const xaios_user_process_t *process) { process->pid); return; } - syscall_release_process_resources(process->pid); + syscall_release_process_resources(process->owner_token); /* Use ELF loader reclaim for processes with per-process address spaces */ if (process->aspace.l3_count > 0) {