From 784f4b423b3ed46ed3a3e817d4ca62783bc3f497 Mon Sep 17 00:00:00 2001 From: Lucius <54578015+lizhicui@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:43:16 +0800 Subject: [PATCH 1/5] feat(pvisor): decouple filesystem isolation from network policy --- crates/persisting-pvisor/src/bundle.rs | 50 ++-- crates/persisting-pvisor/src/cli/mod.rs | 12 +- crates/persisting-pvisor/src/cli/product.rs | 10 +- crates/persisting-pvisor/src/cli/run.rs | 218 +++++++++++++----- crates/persisting-pvisor/src/config.rs | 26 ++- crates/persisting-pvisor/src/lib.rs | 15 +- crates/persisting-pvisor/src/process.rs | 197 ++++++++++++---- crates/persisting-pvisor/src/pvisor.rs | 49 +++- crates/persisting-pvisor/src/sandbox.rs | 146 +++++++++--- .../persisting-pvisor/tests/rootless_local.rs | 24 +- 10 files changed, 580 insertions(+), 167 deletions(-) diff --git a/crates/persisting-pvisor/src/bundle.rs b/crates/persisting-pvisor/src/bundle.rs index 22895762..4c72ec1d 100644 --- a/crates/persisting-pvisor/src/bundle.rs +++ b/crates/persisting-pvisor/src/bundle.rs @@ -210,12 +210,10 @@ impl RunBundle { let network_non_bypassable = !sandbox_setup_failed && enforcement .is_some_and(|evidence| evidence.is_enforced(CapabilityDimension::Network)); - let filesystem_read_non_bypassable = filesystem.is_some() - && !sandbox_setup_failed + let filesystem_read_non_bypassable = !sandbox_setup_failed && enforcement .is_some_and(|evidence| evidence.is_enforced(CapabilityDimension::FilesystemRead)); - let filesystem_write_non_bypassable = filesystem.is_some() - && !sandbox_setup_failed + let filesystem_write_non_bypassable = !sandbox_setup_failed && enforcement .is_some_and(|evidence| evidence.is_enforced(CapabilityDimension::FilesystemWrite)); let filesystem_non_bypassable = @@ -236,16 +234,35 @@ impl RunBundle { ); } if rootless_process && !sandbox_setup_failed { - safety_warnings.push( - "filesystem access and process-tree cleanup are kernel-enforced; the host kernel and syscall surface remain shared" - .into(), - ); + if enforcement.is_some_and(|evidence| { + evidence.is_enforced(CapabilityDimension::FilesystemRead) + || evidence.is_enforced(CapabilityDimension::FilesystemWrite) + }) { + safety_warnings.push( + "filesystem access and process-tree cleanup are kernel-enforced; the host kernel and syscall surface remain shared" + .into(), + ); + } else { + safety_warnings.push( + "process-tree cleanup and selected network boundaries are kernel-enforced; filesystem access remains host-visible" + .into(), + ); + } } if seatbelt_process && !sandbox_setup_failed { - safety_warnings.push( - "filesystem writes are Seatbelt-enforced; reads, the host PID namespace, syscall surface, and resource limits remain shared" - .into(), - ); + if enforcement.is_some_and(|evidence| { + evidence.is_enforced(CapabilityDimension::FilesystemWrite) + }) { + safety_warnings.push( + "filesystem writes are Seatbelt-enforced; reads, the host PID namespace, syscall surface, and resource limits remain shared" + .into(), + ); + } else { + safety_warnings.push( + "selected network boundaries are Seatbelt-enforced; filesystem writes remain host-visible" + .into(), + ); + } } if virtual_machine { safety_warnings.push( @@ -415,9 +432,9 @@ fn resource_summary(record: &RunRecord, result: &RunResult) -> ResourceSummary { #[cfg(unix)] fn effective_native_limits(requested: &ResourceLimits) -> ResourceLimits { - #[cfg(target_os = "linux")] + #[cfg(all(target_os = "linux", not(target_env = "musl")))] type RlimitResource = libc::__rlimit_resource_t; - #[cfg(not(target_os = "linux"))] + #[cfg(any(not(target_os = "linux"), target_env = "musl"))] type RlimitResource = libc::c_int; fn clamp(resource: RlimitResource, requested: Option) -> Option { @@ -642,6 +659,11 @@ mod tests { let intercepted_vm = RunBundle::capture(&record, &result, agentctl.clone(), true).unwrap(); assert!(intercepted_vm.safety.filesystem_non_bypassable); assert!(intercepted_vm.safety.network_non_bypassable); + let mut vm_without_stage = record.clone(); + vm_without_stage.overlay = None; + let vm_without_stage = + RunBundle::capture(&vm_without_stage, &result, agentctl.clone(), true).unwrap(); + assert!(vm_without_stage.safety.filesystem_non_bypassable); record.executor = Some(ExecutorDescriptor { name: "local-rootless-v1".into(), diff --git a/crates/persisting-pvisor/src/cli/mod.rs b/crates/persisting-pvisor/src/cli/mod.rs index 846f0f4f..fe34fcda 100644 --- a/crates/persisting-pvisor/src/cli/mod.rs +++ b/crates/persisting-pvisor/src/cli/mod.rs @@ -13,15 +13,15 @@ use clap::{Parser, Subcommand}; #[cfg(target_os = "linux")] const ROOT_ABOUT: &str = - "Foreground Agent Run manager with rootless Linux sandboxing and reviewable workspaces"; + "Foreground Agent Run manager with independent filesystem, network, and staging policies"; #[cfg(target_os = "linux")] -const ROOT_LONG_ABOUT: &str = "Foreground Agent Run manager: execute, control, Gateway, and OverlayFS.\n\nOn Linux, host runs use safe-best-effort rootless isolation when supported: user and mount namespaces, a minimal synthetic root with chroot, Landlock, no_new_privs, and dropped capabilities. Add `--overlaynet-deny-all` to isolate direct network sockets in a private network namespace."; +const ROOT_LONG_ABOUT: &str = "Foreground Agent Run manager: execute, control, Gateway, and OverlayFS.\n\nHost execution preserves the host filesystem view by default. Use `--filesystem sandbox` for synthetic-root/Landlock restrictions, `--stage` for independent workspace staging, and `--overlaynet` for network policy. On Linux, `--overlaynet-deny-all` uses a private network namespace without enabling filesystem restrictions."; #[cfg(target_os = "macos")] const ROOT_ABOUT: &str = - "Foreground Agent Run manager with Seatbelt isolation and reviewable workspaces"; + "Foreground Agent Run manager with independent filesystem, network, and staging policies"; #[cfg(target_os = "macos")] -const ROOT_LONG_ABOUT: &str = "Foreground Agent Run manager: execute, control, Gateway, and OverlayFS.\n\nOn macOS, host runs use safe-best-effort macFUSE workspace views and Seatbelt confinement when supported. Full-disk reads remain available for local toolchain compatibility. `--overlaynet-deny-all` also blocks non-loopback IP and ambient host Unix sockets while retaining loopback proxy access and Run-local IPC."; +const ROOT_LONG_ABOUT: &str = "Foreground Agent Run manager: execute, control, Gateway, and OverlayFS.\n\nHost execution preserves the host filesystem view by default. Use `--filesystem sandbox` for Seatbelt filesystem restrictions, `--stage` for independent workspace staging (macFUSE may be required), and `--overlaynet` for network policy. Full-disk reads remain available and ambient unless filesystem sandboxing is requested. `--overlaynet-deny-all` blocks non-loopback IP and ambient host Unix sockets while retaining loopback proxy access and Run-local IPC."; #[cfg(not(any(target_os = "linux", target_os = "macos")))] const ROOT_ABOUT: &str = "Foreground Agent Run manager with staged, reviewable workspaces"; @@ -268,8 +268,8 @@ mod tests { #[cfg(target_os = "linux")] { - assert!(help.contains("safe-best-effort")); - assert!(help.contains("rootless isolation")); + assert!(help.contains("independent filesystem, network, and staging policies")); + assert!(help.contains("--filesystem sandbox")); assert!(help.contains("namespace")); assert!(help.contains("Landlock")); } diff --git a/crates/persisting-pvisor/src/cli/product.rs b/crates/persisting-pvisor/src/cli/product.rs index b6d88bb0..2edf68ca 100644 --- a/crates/persisting-pvisor/src/cli/product.rs +++ b/crates/persisting-pvisor/src/cli/product.rs @@ -115,11 +115,17 @@ pub fn review(args: ReviewArgs) -> anyhow::Result<()> { .as_ref() .map(|executor| executor.isolation) { + Some(persisting_control::IsolationKind::RootlessProcess) + if bundle.safety.filesystem_non_bypassable => + "rootless user namespace + Landlock", Some(persisting_control::IsolationKind::RootlessProcess) => { - "rootless user namespace + Landlock" + "rootless user namespace + network namespace" } + Some(persisting_control::IsolationKind::SandboxedProcess) + if bundle.safety.filesystem_write_non_bypassable => + "macOS Seatbelt filesystem policy", Some(persisting_control::IsolationKind::SandboxedProcess) => { - "macOS Seatbelt process sandbox" + "macOS Seatbelt network policy" } Some(persisting_control::IsolationKind::Container) => { "OCI container with injected pVisor" diff --git a/crates/persisting-pvisor/src/cli/run.rs b/crates/persisting-pvisor/src/cli/run.rs index 7cf9925b..e7f43c56 100644 --- a/crates/persisting-pvisor/src/cli/run.rs +++ b/crates/persisting-pvisor/src/cli/run.rs @@ -95,9 +95,9 @@ use persisting_overlaynet::{NetworkAccessRule, NetworkBandwidthLimit}; use serde::Deserialize; use crate::config::{ - ContainerMount, ContainerNetwork, ContainerPlatform, GatewayMode, OverlayFsBackend, - OverlayFsCommit, OverlayFsSettings, OverlayNetMode, OverlayNetPolicy, OverlayNetSettings, - RunConfig, RunExecutorKind, RunPolicy, RunStdio, + ContainerMount, ContainerNetwork, ContainerPlatform, FilesystemMode, GatewayMode, + OverlayFsBackend, OverlayFsCommit, OverlayFsSettings, OverlayNetMode, OverlayNetPolicy, + OverlayNetSettings, RunConfig, RunExecutorKind, RunPolicy, RunStdio, }; use crate::runtime::{RunLineage, default_run_home, resolve_run}; use crate::{ @@ -110,20 +110,20 @@ use super::trajectory::{JsonlEventSink, JsonlWriter, jsonl_capture_sink}; #[cfg(target_os = "linux")] pub(super) const RUN_COMMAND_ABOUT: &str = - "Execute one Agent Run with safe-best-effort host isolation by default"; + "Execute one Agent Run with independent filesystem, staging, and network policies"; #[cfg(target_os = "linux")] -pub(super) const RUN_COMMAND_LONG_ABOUT: &str = "Execute one Agent Run under pVisor management. Host execution uses safe-best-effort isolation when supported by the system."; +pub(super) const RUN_COMMAND_LONG_ABOUT: &str = "Execute one Agent Run under pVisor management. Host execution preserves the host filesystem view by default; use --filesystem sandbox for filesystem access restrictions, --stage for change staging, and --overlaynet for network policy."; #[cfg(target_os = "macos")] pub(super) const RUN_COMMAND_ABOUT: &str = - "Execute one Agent Run with safe-best-effort host isolation by default"; + "Execute one Agent Run with independent filesystem, staging, and network policies"; #[cfg(target_os = "macos")] pub(super) const RUN_COMMAND_LONG_ABOUT: &str = MACOS_RUN_COMMAND_LONG_ABOUT; // Compile the macOS description in tests on every platform so Linux CI also // checks its safety disclosures instead of leaving them to the macOS shard. #[cfg(any(target_os = "macos", test))] -const MACOS_RUN_COMMAND_LONG_ABOUT: &str = "Execute one Agent Run under pVisor management. Host execution uses safe-best-effort isolation when supported by the system.\n\nOn macOS, staged workspace views use macFUSE and Seatbelt confines writes when available. Full-disk reads remain ambient; selective network policies remain cooperative. With --overlaynet-deny-all, Seatbelt blocks non-loopback IP traffic and ambient host Unix sockets while permitting loopback proxy access and Run-scoped Unix IPC.\n\nUnavailable isolation capabilities are reported as warnings in best-effort mode. With --strict, insufficient isolation guarantees cause the Run to fail before Agent execution."; +const MACOS_RUN_COMMAND_LONG_ABOUT: &str = "Execute one Agent Run under pVisor management. Host execution preserves the host filesystem view by default. Use --filesystem sandbox for filesystem access restrictions, --stage for change staging, and --overlaynet for network policy.\n\nOn macOS, staged workspace views use macFUSE when requested, and Seatbelt is used for requested filesystem sandboxing or deny-all network isolation. Full-disk reads remain ambient unless filesystem sandboxing is requested; selective network policies remain cooperative. With --overlaynet-deny-all, Seatbelt blocks non-loopback IP traffic and ambient host Unix sockets while permitting loopback proxy access and Run-scoped Unix IPC.\n\nUnavailable isolation capabilities are reported as warnings in best-effort mode. With --strict, insufficient isolation guarantees cause the Run to fail before Agent execution."; #[cfg(not(any(target_os = "linux", target_os = "macos")))] pub(super) const RUN_COMMAND_ABOUT: &str = "Execute one Agent Run under pVisor management"; @@ -131,9 +131,9 @@ pub(super) const RUN_COMMAND_ABOUT: &str = "Execute one Agent Run under pVisor m pub(super) const RUN_COMMAND_LONG_ABOUT: &str = RUN_COMMAND_ABOUT; #[cfg(target_os = "linux")] -const EXECUTOR_HELP: &str = "Execution provider: host, container, or vm. `vm` uses the statically linked libkrun backend; host uses safe-best-effort isolation"; +const EXECUTOR_HELP: &str = "Execution provider: host, container, or vm. `vm` uses the statically linked libkrun backend; host supports optional filesystem and network isolation"; #[cfg(target_os = "macos")] -const EXECUTOR_HELP: &str = "Execution provider: host, container, or vm. `vm` uses the statically linked libkrun backend; host uses safe-best-effort isolation"; +const EXECUTOR_HELP: &str = "Execution provider: host, container, or vm. `vm` uses the statically linked libkrun backend; host supports optional filesystem and network isolation"; #[cfg(not(any(target_os = "linux", target_os = "macos")))] const EXECUTOR_HELP: &str = "Execution provider for the Agent command"; @@ -199,6 +199,9 @@ struct RunOverrides { name: Option, #[arg(long, value_enum, help = EXECUTOR_HELP)] executor: Option, + /// Filesystem access policy; independent from OverlayNet and OverlayFS staging. + #[arg(long, value_enum)] + filesystem: Option, #[arg(long, value_name = "DURATION")] timeout: Option, #[arg(long, value_enum)] @@ -561,7 +564,8 @@ pub async fn run(args: RunArgs) -> anyhow::Result { { return run_prepared_spec(args).await; } - // Host runs use the safe-best-effort profile by default. + // Host runs keep the best-effort lifecycle/evidence profile by default; + // filesystem restrictions, staging, and network isolation remain opt-in. let safe = true; let run_id = format!("run-{}", uuid::Uuid::new_v4()); let mut config = args @@ -1037,37 +1041,54 @@ async fn execute_config( None }; if config.run.executor == RunExecutorKind::Vm { - let (rootfs, workspace) = resolve_vm_layout(&config)?; - config.vm.rootfs = Some(rootfs.clone()); - config.run.workspace = Some(workspace.clone()); - let has_guest_overlay = config - .overlayfs - .as_ref() - .and_then(|overlay| overlay.target.as_ref()) - .is_some(); - if !has_guest_overlay { - let overlay = config + #[cfg(all(target_os = "macos", target_arch = "x86_64"))] + anyhow::bail!( + "VM execution is unsupported on Intel macOS; use Linux x86_64 or Apple Silicon macOS" + ); + #[cfg(not(all(target_os = "macos", target_arch = "x86_64")))] + { + let (rootfs, workspace) = resolve_vm_layout(&config)?; + config.vm.rootfs = Some(rootfs.clone()); + config.run.workspace = Some(workspace.clone()); + let has_guest_overlay = config .overlayfs - .get_or_insert_with(OverlayFsSettings::default); - // Keep the host workspace path stable inside the guest. The - // workspace is a separate virtio-fs mount; using the rootfs as its - // base would make `cwd` point at a path that does not exist in an - // image guest and would bypass workspace staging. - overlay.base = Some(workspace.clone()); - overlay.target = Some(workspace.clone()); - overlay.commit = OverlayFsCommit::Manual; - } - if config.vm.library_dir.is_none() && crate::vm::bundled_firmware_dir().is_none() { - eprintln!( - "pVisor firmware: resolving libkrunfw {}", - crate::firmware::VERSION + .as_ref() + .and_then(|overlay| overlay.target.as_ref()) + .is_some(); + if !has_guest_overlay { + let overlay = config + .overlayfs + .get_or_insert_with(OverlayFsSettings::default); + // Keep the host workspace path stable inside the guest. The + // workspace is a separate virtio-fs mount; using the rootfs as its + // base would make `cwd` point at a path that does not exist in an + // image guest and would bypass workspace staging. + overlay.base = Some(workspace.clone()); + overlay.target = Some(workspace.clone()); + overlay.commit = OverlayFsCommit::Manual; + } + #[cfg(all(target_os = "linux", target_env = "musl", target_arch = "x86_64"))] + anyhow::ensure!( + config.vm.library_dir.is_none(), + "--vm-library-dir is unavailable in the static musl build; libkrun's kernel bundle is embedded" ); - let directory = - tokio::task::spawn_blocking(|| crate::firmware::FirmwareStore::new()?.prepare()) - .await - .context("libkrunfw preparation task failed")??; - eprintln!("pVisor firmware: {}", directory.display()); - config.vm.library_dir = Some(directory); + #[cfg(not(any( + all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), + all(target_os = "macos", target_arch = "x86_64") + )))] + if config.vm.library_dir.is_none() && crate::vm::bundled_firmware_dir().is_none() { + eprintln!( + "pVisor firmware: resolving libkrunfw {}", + crate::firmware::VERSION + ); + let directory = tokio::task::spawn_blocking(|| { + crate::firmware::FirmwareStore::new()?.prepare() + }) + .await + .context("libkrunfw preparation task failed")??; + eprintln!("pVisor firmware: {}", directory.display()); + config.vm.library_dir = Some(directory); + } } } else { if let Some(base) = config @@ -1125,6 +1146,9 @@ async fn execute_config( overlay.protect_target = true; } let overlay_enabled = overlay.is_some(); + let filesystem_isolated = config.filesystem == FilesystemMode::Sandbox; + let network_namespace_required = config.run.executor == RunExecutorKind::Host + && config.overlaynet.policy == OverlayNetPolicy::Deny; let resolved_stage_for_limit = overlay.as_ref().and_then(|hint| hint.stage_dir.clone()); let proxy = resolve_proxy(&config)?; @@ -1165,8 +1189,10 @@ async fn execute_config( let executor: Arc = match config.run.executor { #[cfg(target_os = "linux")] - RunExecutorKind::Host if safe_profile_requested => { - if crate::process::rootless_runtime_available() { + RunExecutorKind::Host + if safe_profile_requested && (filesystem_isolated || network_namespace_required) => + { + if crate::process::rootless_runtime_available(!filesystem_isolated) { match ProcessExecutor::rootless_with_launcher(std::env::current_exe()?) { Ok(executor) => Arc::new(executor), Err(error) => { @@ -1178,13 +1204,15 @@ async fn execute_config( } } else { eprintln!( - "pVisor safe-best-effort: user/mount/PID namespaces unavailable; falling back to host process" + "pVisor safe-best-effort: required namespaces unavailable; falling back to host process" ); Arc::new(ProcessExecutor::default()) } } #[cfg(target_os = "macos")] - RunExecutorKind::Host if safe_profile_requested => { + RunExecutorKind::Host + if safe_profile_requested && (filesystem_isolated || network_namespace_required) => + { match ProcessExecutor::seatbelt_with_launcher(std::env::current_exe()?) { Ok(executor) => Arc::new(executor), Err(error) => { @@ -1196,7 +1224,9 @@ async fn execute_config( } } #[cfg(not(any(target_os = "linux", target_os = "macos")))] - RunExecutorKind::Host if safe_profile_requested => Arc::new(ProcessExecutor::default()), + RunExecutorKind::Host if safe_profile_requested && filesystem_isolated => { + Arc::new(ProcessExecutor::default()) + } RunExecutorKind::Host => Arc::new(ProcessExecutor::default()), RunExecutorKind::Container => Arc::new(ContainerExecutor::new(config.container.clone())?), RunExecutorKind::Vm => Arc::new(VmExecutor::new(config.vm.clone())?), @@ -1326,6 +1356,16 @@ async fn execute_config( spec.metadata .insert("pvisor.safe".into(), serde_json::Value::Bool(true)); } + spec.metadata.insert( + "pvisor.filesystem.mode".into(), + serde_json::Value::String( + match config.filesystem { + FilesystemMode::Host => "host", + FilesystemMode::Sandbox => "sandbox", + } + .into(), + ), + ); if safe_profile_requested { let network_boundary = if config.run.executor == RunExecutorKind::Vm @@ -1344,29 +1384,50 @@ async fn execute_config( } else { "cooperative network review" }; - if overlay_enabled { - eprintln!("pVisor safe profile: staged workspace + {network_boundary}"); + let filesystem_boundary = if filesystem_isolated { + "restricted filesystem access" } else { - eprintln!( - "pVisor safe profile: best-effort isolation + {network_boundary}; workspace writes are not staged (pass --stage for COW/review)" - ); - } + "host filesystem access" + }; + let staging = if overlay_enabled { + "staged workspace" + } else { + "workspace writes are direct" + }; + eprintln!("pVisor safe profile: {filesystem_boundary} + {staging} + {network_boundary}"); eprintln!("workspace: {}", workspace.display()); eprintln!("Run storage: {}", storage.display()); match config.run.executor { RunExecutorKind::Host => { #[cfg(target_os = "linux")] - eprintln!( - "boundary: rootless user/mount/PID namespaces + PID 1 reaper + synthetic root + Landlock filesystem; network remains cooperative unless explicitly denied" - ); + { + let process_boundary = if filesystem_isolated { + "rootless user/mount/PID namespaces + PID 1 reaper" + } else if network_namespace_required { + "private network namespace + process supervisor" + } else { + "host process" + }; + eprintln!( + "boundary: {process_boundary}; filesystem and network boundaries follow the selected policies" + ); + } #[cfg(target_os = "macos")] - if overlay_enabled { + if filesystem_isolated && overlay_enabled { eprintln!( "boundary: Seatbelt-enforced staged writes; reads and selective network policies remain ambient/cooperative" ); + } else if filesystem_isolated { + eprintln!( + "boundary: Seatbelt-enforced filesystem writes; reads and selective network policies remain ambient/cooperative" + ); + } else if network_namespace_required { + eprintln!( + "boundary: Seatbelt-enforced deny-all network; filesystem access remains host-visible" + ); } else { eprintln!( - "boundary: Seatbelt best-effort when available; workspace writes are not staged; reads and selective network policies remain ambient/cooperative" + "boundary: host process; filesystem access and selective network policies remain host-visible/cooperative" ); } #[cfg(not(any(target_os = "linux", target_os = "macos")))] @@ -1577,6 +1638,9 @@ fn apply_cli(config: &mut RunConfig, args: RunArgs) -> anyhow::Result<()> { if let Some(value) = explicit_executor { config.run.executor = value; } + if let Some(value) = args.run.filesystem { + config.filesystem = value; + } if args.vm.vm { config.run.executor = RunExecutorKind::Vm; } @@ -1944,6 +2008,7 @@ fn resolve_workspace(workspace: &Path) -> anyhow::Result { Ok(workspace) } +#[cfg_attr(all(target_os = "macos", target_arch = "x86_64"), allow(dead_code))] fn resolve_vm_layout(config: &RunConfig) -> anyhow::Result<(PathBuf, PathBuf)> { let rootfs = config .vm @@ -2178,6 +2243,7 @@ mod tests { apply_cli(&mut config, *args).unwrap(); apply_safe_defaults(&mut config).unwrap(); assert!(config.overlayfs.is_none(), "stage is opt-in"); + assert_eq!(config.filesystem, FilesystemMode::Host); assert_eq!(config.overlaynet.mode, OverlayNetMode::Auto); assert_eq!(config.run.agent, "true"); assert_ne!( @@ -2186,6 +2252,40 @@ mod tests { ); } + #[test] + fn filesystem_policy_is_independent_from_overlaynet() { + let crate::cli::Command::Run(args) = Cli::try_parse_from([ + "pvisor", + "run", + "--overlaynet-deny-all", + "--filesystem", + "sandbox", + "--", + "true", + ]) + .unwrap() + .command + else { + unreachable!() + }; + let mut config = RunConfig::default(); + apply_cli(&mut config, *args).unwrap(); + assert_eq!(config.filesystem, FilesystemMode::Sandbox); + assert_eq!(config.overlaynet.policy, OverlayNetPolicy::Deny); + + let crate::cli::Command::Run(args) = + Cli::try_parse_from(["pvisor", "run", "--overlaynet-deny-all", "--", "true"]) + .unwrap() + .command + else { + unreachable!() + }; + let mut config = RunConfig::default(); + apply_cli(&mut config, *args).unwrap(); + assert_eq!(config.filesystem, FilesystemMode::Host); + assert_eq!(config.overlaynet.policy, OverlayNetPolicy::Deny); + } + #[test] fn fork_inherits_or_reidentifies_the_agent_with_its_command() { let source = vec!["/bin/sh".into(), "-c".into(), "work".into()]; @@ -2418,6 +2518,7 @@ mod tests { assert!(error.to_string().contains("requires --executor vm")); } + #[cfg(not(all(target_os = "macos", target_arch = "x86_64")))] #[test] fn vm_rejects_the_host_only_explicit_proxy_mode() { let temporary = tempfile::tempdir().unwrap(); @@ -2479,6 +2580,7 @@ mod tests { assert_eq!(resolved_workspace, project.canonicalize().unwrap()); } + #[cfg(not(all(target_os = "macos", target_arch = "x86_64")))] #[test] fn overlayfs_path_is_valid_for_vm_executor() { let mut config = RunConfig::default(); @@ -2713,7 +2815,7 @@ mod tests { // Terminal wrapping must not affect checks of the safety description. let help = help.split_whitespace().collect::>().join(" "); for disclosure in [ - "safe-best-effort", + "filesystem sandbox", "macFUSE", "Seatbelt", "Full-disk reads remain ambient", @@ -2739,8 +2841,8 @@ mod tests { #[cfg(target_os = "linux")] { - assert!(help.contains("safe-best-effort")); - assert!(help.contains("Host execution uses safe-best-effort isolation")); + assert!(help.contains("host filesystem view by default")); + assert!(help.contains("--filesystem sandbox")); } #[cfg(target_os = "macos")] { diff --git a/crates/persisting-pvisor/src/config.rs b/crates/persisting-pvisor/src/config.rs index bf018546..ca18539a 100644 --- a/crates/persisting-pvisor/src/config.rs +++ b/crates/persisting-pvisor/src/config.rs @@ -20,7 +20,10 @@ pub struct RunConfig { pub container: ContainerSettings, #[serde(alias = "kvm")] pub vm: VmSettings, - /// Transactional filesystem configuration. Absence means host filesystem access. + /// Process filesystem access policy. This is independent from OverlayFS + /// change staging and from OverlayNet network policy. + pub filesystem: FilesystemMode, + /// Transactional OverlayFS configuration. Absence means no staged OverlayFS view. pub overlayfs: Option, pub overlaynet: OverlayNetSettings, pub gateway: GatewaySettings, @@ -162,8 +165,10 @@ pub struct VmSettings { pub image_store: Option, /// Reject apply operations that would mutate the configured rootfs lower. pub rootfs_immutable: bool, - /// Optional directory containing libkrunfw. Packaged builds discover it - /// next to pVisor; source builds use a verified per-user download cache. + /// Optional directory containing libkrunfw. Packaged glibc/macOS builds + /// discover it next to pVisor; source builds use a verified per-user + /// download cache. The x86_64 Linux musl build embeds the kernel bundle + /// and rejects this setting. pub library_dir: Option, pub memory_mib: u32, pub cpus: u16, @@ -219,6 +224,18 @@ pub enum RunPolicy { Enforce, } +/// Whether a host process receives pVisor's synthetic-root/Landlock or +/// Seatbelt filesystem access restrictions. +#[derive(Debug, Clone, Copy, Default, Deserialize, Serialize, PartialEq, Eq, clap::ValueEnum)] +#[serde(rename_all = "kebab-case")] +pub enum FilesystemMode { + /// Preserve the host process filesystem view and permissions. + #[default] + Host, + /// Restrict filesystem access to pVisor-declared roots. + Sandbox, +} + #[derive(Debug, Clone, Deserialize, Serialize)] #[serde(default, deny_unknown_fields)] pub struct OverlayFsSettings { @@ -448,6 +465,8 @@ mod tests { fn run_config_toml_roundtrip() { let config: RunConfig = toml::from_str( r#" +filesystem = "sandbox" + [run] executor = "container" command = ["codex"] @@ -488,6 +507,7 @@ upstream = "https://api.openai.com/v1" "#, ) .unwrap(); + assert_eq!(config.filesystem, FilesystemMode::Sandbox); assert_eq!( config .overlayfs diff --git a/crates/persisting-pvisor/src/lib.rs b/crates/persisting-pvisor/src/lib.rs index 7fe0c5eb..3a0920c7 100644 --- a/crates/persisting-pvisor/src/lib.rs +++ b/crates/persisting-pvisor/src/lib.rs @@ -4,6 +4,8 @@ //! network, filesystem, and the optional internal Gateway driver. Durable //! EventRecord output uses local JSONL when recording is enabled. +#![cfg_attr(all(target_os = "macos", target_arch = "x86_64"), allow(dead_code))] + pub mod cli; pub mod core; mod runtime; @@ -19,6 +21,10 @@ mod control; mod delegated; mod event; mod executor; +#[cfg(not(any( + all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), + all(target_os = "macos", target_arch = "x86_64") +)))] mod firmware; mod oci; mod process; @@ -40,10 +46,11 @@ pub use checkpoint::{ latest_logical_checkpoint, restore_logical_checkpoint, }; pub use config::{ - ContainerMount, ContainerNetwork, ContainerPlatform, ContainerSettings, GatewayDriverConfig, - GatewayMode, GatewaySettings, NetworkDriverConfig, OverlayFsBackend, OverlayFsCommit, - OverlayFsSettings, OverlayNetMode, OverlayNetPolicy, OverlayNetSettings, PVisorConfig, - RecordSettings, RunConfig, RunExecutorKind, RunPolicy, RunSettings, RunStdio, VmSettings, + ContainerMount, ContainerNetwork, ContainerPlatform, ContainerSettings, FilesystemMode, + GatewayDriverConfig, GatewayMode, GatewaySettings, NetworkDriverConfig, OverlayFsBackend, + OverlayFsCommit, OverlayFsSettings, OverlayNetMode, OverlayNetPolicy, OverlayNetSettings, + PVisorConfig, RecordSettings, RunConfig, RunExecutorKind, RunPolicy, RunSettings, RunStdio, + VmSettings, }; pub use container::ContainerExecutor; pub use control::{ diff --git a/crates/persisting-pvisor/src/process.rs b/crates/persisting-pvisor/src/process.rs index ae7da600..afd7afe3 100644 --- a/crates/persisting-pvisor/src/process.rs +++ b/crates/persisting-pvisor/src/process.rs @@ -5,7 +5,7 @@ use crate::sandbox::{INTERNAL_SANDBOX_ARG, NetworkIsolation}; use crate::sandbox::{MACOS_SANDBOX_EXEC, SEATBELT_ATTESTATION, SeatbeltPlan, seatbelt_profile}; #[cfg(target_os = "linux")] use crate::sandbox::{ROOTLESS_ATTESTATION, SandboxPlan, landlock_runtime_available}; -use crate::sandbox::{SANDBOX_PLAN_ENV, SANDBOX_SETUP_FAILED_WARNING}; +use crate::sandbox::{SANDBOX_ARG0_ENV, SANDBOX_PLAN_ENV, SANDBOX_SETUP_FAILED_WARNING}; use async_trait::async_trait; use persisting_control::{ CapabilityDimension, CapabilityEnforcementEvidence, ExecutorDescriptor, ExecutorKind, @@ -369,6 +369,19 @@ fn network_isolation(spec: &RunSpec) -> NetworkIsolation { } } +#[cfg(any(target_os = "linux", target_os = "macos"))] +fn filesystem_isolation(spec: &RunSpec) -> bool { + // Library users that construct a rootless executor directly retain the + // historical restricted default. CLI runs set this marker explicitly so + // filesystem access can be configured independently from networking. + !matches!( + spec.metadata + .get("pvisor.filesystem.mode") + .and_then(serde_json::Value::as_str), + Some("host") + ) +} + fn resolve_host_program(program: &str) -> std::path::PathBuf { if program.contains(std::path::MAIN_SEPARATOR) { return program.into(); @@ -518,6 +531,13 @@ impl ProcessExecutor { command.envs(&invocation.env); // This is a reserved supervisor-to-launcher capability. Apply it last // so an untrusted Run environment cannot remove or replace the policy. + if sandbox_plan.is_some() { + // The launcher canonicalizes the executable so it can project the + // real inode into a synthetic root. Preserve the caller's original + // argv[0] separately: Alpine's /bin commands are often BusyBox + // symlinks, and BusyBox chooses its applet from that basename. + command.env(SANDBOX_ARG0_ENV, &invocation.program); + } if let Some(sandbox_plan) = sandbox_plan { command.env(SANDBOX_PLAN_ENV, sandbox_plan); } @@ -536,15 +556,25 @@ impl ProcessExecutor { /// probe safe in a multithreaded Tokio process and distinguishes an unavailable /// host capability from a later Agent failure. #[cfg(target_os = "linux")] -pub(crate) fn rootless_runtime_available() -> bool { - landlock_runtime_available() - && StdCommand::new("unshare") - .args(["--user", "--mount", "--pid", "--fork", "true"]) +pub(crate) fn rootless_runtime_available(preserve_host_filesystem: bool) -> bool { + let probe = |args: &[&str]| { + StdCommand::new("unshare") + .args(args) .stdin(Stdio::null()) .stdout(Stdio::null()) .stderr(Stdio::null()) .status() .is_ok_and(|status| status.success()) + }; + if preserve_host_filesystem { + // Network-only Runs still use a user namespace so the Agent cannot + // create another namespace and escape deny-all. Landlock is omitted + // at runtime, so it must not be a prerequisite for this probe. + probe(["--user", "--mount", "--net", "--fork", "true"].as_slice()) + } else { + landlock_runtime_available() + && probe(["--user", "--mount", "--pid", "--fork", "true"].as_slice()) + } } #[cfg(unix)] @@ -624,6 +654,7 @@ fn platform_launcher_command( })?; let sandbox_root = SandboxResources::create()?; let network = network_isolation(spec); + let filesystem_isolated = filesystem_isolation(spec); let plan = rootless_plan( spec, invocation, @@ -637,6 +668,7 @@ fn platform_launcher_command( .expect("created rootless attestation") .to_owned(), network, + filesystem_isolated, )?; let encoded = serde_json::to_string(&plan).map_err(std::io::Error::other)?; let mut command = Command::new(launcher); @@ -662,6 +694,7 @@ fn platform_launcher_command( ) })?; let resources = SandboxResources::create()?; + let filesystem_isolated = filesystem_isolation(spec); let cwd = invocation .cwd .as_deref() @@ -682,26 +715,28 @@ fn platform_launcher_command( for path in ["/dev/null", "/dev/zero", "/dev/tty", "/dev/fd"] { push_existing(&mut writable_paths, Path::new(path)); } - for capability in &spec.capabilities.filesystem { - if capability.access != FilesystemAccess::ReadWrite { - continue; - } - let path = PathBuf::from(&capability.path); - let path = if path.is_absolute() { - path - } else { - cwd.join(path) - }; - if !path.exists() { - return Err(std::io::Error::new( - std::io::ErrorKind::NotFound, - format!( - "filesystem capability path does not exist: {}", - path.display() - ), - )); + if filesystem_isolated { + for capability in &spec.capabilities.filesystem { + if capability.access != FilesystemAccess::ReadWrite { + continue; + } + let path = PathBuf::from(&capability.path); + let path = if path.is_absolute() { + path + } else { + cwd.join(path) + }; + if !path.exists() { + return Err(std::io::Error::new( + std::io::ErrorKind::NotFound, + format!( + "filesystem capability path does not exist: {}", + path.display() + ), + )); + } + writable_paths.push(path); } - writable_paths.push(path); } let network = network_isolation(spec); @@ -730,6 +765,7 @@ fn platform_launcher_command( &allowed_unix_sockets, &local_socket_roots, network, + filesystem_isolated, )?; let plan = SeatbeltPlan { attestation: resources @@ -737,6 +773,7 @@ fn platform_launcher_command( .expect("created Seatbelt attestation") .to_owned(), network, + filesystem_isolated, }; let encoded = serde_json::to_string(&plan).map_err(std::io::Error::other)?; @@ -785,6 +822,7 @@ fn rootless_plan( root: PathBuf, attestation: PathBuf, network: NetworkIsolation, + filesystem_isolated: bool, ) -> std::io::Result { let cwd = invocation .cwd @@ -831,25 +869,27 @@ fn rootless_plan( } } - for capability in &spec.capabilities.filesystem { - let path = PathBuf::from(&capability.path); - let path = if path.is_absolute() { - path - } else { - cwd.join(path) - }; - if !path.exists() { - return Err(std::io::Error::new( - std::io::ErrorKind::NotFound, - format!( - "filesystem capability path does not exist: {}", - path.display() - ), - )); - } - match capability.access { - FilesystemAccess::Read => push_existing(&mut read_only, &path), - FilesystemAccess::ReadWrite => push_existing(&mut read_write, &path), + if filesystem_isolated { + for capability in &spec.capabilities.filesystem { + let path = PathBuf::from(&capability.path); + let path = if path.is_absolute() { + path + } else { + cwd.join(path) + }; + if !path.exists() { + return Err(std::io::Error::new( + std::io::ErrorKind::NotFound, + format!( + "filesystem capability path does not exist: {}", + path.display() + ), + )); + } + match capability.access { + FilesystemAccess::Read => push_existing(&mut read_only, &path), + FilesystemAccess::ReadWrite => push_existing(&mut read_write, &path), + } } } @@ -864,6 +904,7 @@ fn rootless_plan( read_only, read_write, network, + filesystem_isolated, process_limit: spec.runtime.resource_limits.processes, }) } @@ -1424,6 +1465,76 @@ mod tests { assert_ne!(encoded, r#"{"read_write":["/"]}"#); } + #[cfg(target_os = "linux")] + #[test] + fn rootless_plan_can_keep_host_filesystem_access_for_network_only_runs() { + let temporary = tempfile::tempdir().unwrap(); + let mut spec = RunSpec::process("run", "agent", "/bin/true"); + spec.metadata.insert( + "pvisor.filesystem.mode".into(), + serde_json::Value::String("host".into()), + ); + { + let RunInvocation::Process(invocation) = &mut spec.invocation; + invocation.cwd = Some(temporary.path().display().to_string()); + } + + let executor = + ProcessExecutor::rootless_with_launcher(std::env::current_exe().unwrap()).unwrap(); + let RunInvocation::Process(invocation) = &spec.invocation; + let command = executor.spawn_command(&spec, invocation).unwrap(); + let encoded = command + .command + .as_std() + .get_envs() + .find_map(|(key, value)| { + (key == SANDBOX_PLAN_ENV).then(|| value.unwrap().to_string_lossy().into_owned()) + }) + .unwrap(); + let plan: SandboxPlan = serde_json::from_str(&encoded).unwrap(); + assert!(!plan.filesystem_isolated); + } + + #[cfg(any(target_os = "linux", target_os = "macos"))] + #[test] + fn sandbox_launcher_preserves_symlink_argv0_with_an_untrusted_environment() { + let temporary = tempfile::tempdir().unwrap(); + let alias = temporary.path().join("sh"); + std::os::unix::fs::symlink("/bin/sh", &alias).unwrap(); + let mut spec = RunSpec::process("run", "agent", alias.to_str().unwrap()); + { + let RunInvocation::Process(invocation) = &mut spec.invocation; + invocation.args = vec!["-c".into(), "exit 0".into()]; + invocation.cwd = Some(temporary.path().display().to_string()); + invocation.inherit_env = false; + invocation + .env + .insert(SANDBOX_ARG0_ENV.into(), "wrong-applet".into()); + } + + #[cfg(target_os = "linux")] + let executor = + ProcessExecutor::rootless_with_launcher(std::env::current_exe().unwrap()).unwrap(); + #[cfg(target_os = "macos")] + let executor = + ProcessExecutor::seatbelt_with_launcher(std::env::current_exe().unwrap()).unwrap(); + let RunInvocation::Process(invocation) = &spec.invocation; + let prepared = executor.spawn_command(&spec, invocation).unwrap(); + let command = prepared.command.as_std(); + let arg0 = command + .get_envs() + .find_map(|(key, value)| (key == SANDBOX_ARG0_ENV).then(|| value.unwrap())) + .expect("trusted launcher argv[0] must survive the run environment"); + assert_eq!(arg0, alias.as_os_str()); + let arguments = command.get_args().collect::>(); + assert_eq!( + arguments[arguments.len() - 3], + alias.canonicalize().unwrap() + ); + assert_eq!(arguments[arguments.len() - 2], "-c"); + assert_eq!(arguments[arguments.len() - 1], "exit 0"); + } + #[cfg(not(target_os = "linux"))] #[test] fn rootless_executor_fails_closed_off_linux() { diff --git a/crates/persisting-pvisor/src/pvisor.rs b/crates/persisting-pvisor/src/pvisor.rs index bc3d9f0b..6f9e4fd1 100644 --- a/crates/persisting-pvisor/src/pvisor.rs +++ b/crates/persisting-pvisor/src/pvisor.rs @@ -590,6 +590,22 @@ fn effective_capability_enforcement( vm_network_enforcing: bool, ) -> CapabilityEnforcementEvidence { let mut evidence = descriptor.capability_enforcement.clone(); + if matches!( + descriptor.isolation, + IsolationKind::RootlessProcess | IsolationKind::SandboxedProcess + ) && spec + .metadata + .get("pvisor.filesystem.mode") + .and_then(serde_json::Value::as_str) + == Some("host") + { + evidence + .dimensions + .remove(&CapabilityDimension::FilesystemRead); + evidence + .dimensions + .remove(&CapabilityDimension::FilesystemWrite); + } if proxy_network_configured { evidence.record( CapabilityDimension::Network, @@ -738,9 +754,40 @@ mod tests { use super::*; use crate::{EventSink, MemoryEventSink}; use async_trait::async_trait; - use persisting_control::{NetworkCapability, RunFailureKind, RunInvocation, StdioMode}; + use persisting_control::{ + ExecutorKind, NetworkCapability, RunFailureKind, RunInvocation, StdioMode, + }; use std::sync::Mutex; + #[test] + fn host_filesystem_mode_only_removes_local_process_filesystem_evidence() { + let mut process = ExecutorDescriptor { + name: "local-rootless-v1".into(), + kind: ExecutorKind::Process, + isolation: IsolationKind::RootlessProcess, + capability_enforcement: CapabilityEnforcementEvidence::default() + .enforced(CapabilityDimension::FilesystemRead, "test-read") + .enforced(CapabilityDimension::FilesystemWrite, "test-write"), + supports_checkpoint: false, + supports_migration: false, + }; + let mut process_spec = RunSpec::process("host-fs", "agent", "/bin/true"); + process_spec.metadata.insert( + "pvisor.filesystem.mode".into(), + serde_json::Value::String("host".into()), + ); + let process_evidence = + effective_capability_enforcement(&process, &process_spec, false, false); + assert!(!process_evidence.is_enforced(CapabilityDimension::FilesystemRead)); + assert!(!process_evidence.is_enforced(CapabilityDimension::FilesystemWrite)); + + process.isolation = IsolationKind::VirtualMachine; + process.kind = ExecutorKind::VirtualMachine; + let vm_evidence = effective_capability_enforcement(&process, &process_spec, false, false); + assert!(vm_evidence.is_enforced(CapabilityDimension::FilesystemRead)); + assert!(vm_evidence.is_enforced(CapabilityDimension::FilesystemWrite)); + } + #[derive(Default)] struct RejectCompletedSink { kinds: Mutex>, diff --git a/crates/persisting-pvisor/src/sandbox.rs b/crates/persisting-pvisor/src/sandbox.rs index d092db7a..636bd087 100644 --- a/crates/persisting-pvisor/src/sandbox.rs +++ b/crates/persisting-pvisor/src/sandbox.rs @@ -15,6 +15,13 @@ use std::path::PathBuf; pub(crate) const INTERNAL_SANDBOX_ARG: &str = "__pvisor-sandbox-exec"; pub(crate) const SANDBOX_PLAN_ENV: &str = "PERSISTING_INTERNAL_SANDBOX_PLAN"; +/// Original Agent argv[0] preserved across the canonicalizing launcher. +/// +/// Linux distributions such as Alpine commonly expose commands as symlinks +/// to BusyBox. The launcher must execute the canonical inode for its +/// filesystem setup, while still presenting the symlink's basename to the +/// child so BusyBox selects the requested applet. +pub(crate) const SANDBOX_ARG0_ENV: &str = "PERSISTING_INTERNAL_SANDBOX_ARG0"; /// Reserved launcher exit status: setup failed before the Agent was executed. #[doc(hidden)] pub const SANDBOX_SETUP_EXIT_CODE: i32 = 125; @@ -70,17 +77,28 @@ pub(crate) struct SandboxPlan { pub read_only: Vec, pub read_write: Vec, pub network: NetworkIsolation, + /// Whether the launcher should construct the synthetic root and install + /// Landlock. Network isolation can run independently of this policy. + #[serde(default = "default_filesystem_isolated")] + pub filesystem_isolated: bool, /// Applied after the private PID namespace is initialized so the trusted /// launcher itself can still create its init/reaper process. #[serde(default)] pub process_limit: Option, } +#[cfg(any(target_os = "linux", target_os = "macos"))] +const fn default_filesystem_isolated() -> bool { + true +} + #[cfg(target_os = "macos")] #[derive(Debug, Clone, Serialize, Deserialize)] pub(crate) struct SeatbeltPlan { pub attestation: PathBuf, pub network: NetworkIsolation, + #[serde(default = "default_filesystem_isolated")] + pub filesystem_isolated: bool, } /// Enter the hidden launcher when the first argument is the internal marker. @@ -100,6 +118,7 @@ pub fn run_internal_if_requested() -> anyhow::Result { #[cfg(target_os = "linux")] fn run_internal() -> anyhow::Result<()> { use anyhow::{Context, bail}; + use std::os::unix::process::CommandExt; let encoded = std::env::var(SANDBOX_PLAN_ENV).context("missing rootless sandbox plan")?; let plan: SandboxPlan = @@ -112,17 +131,19 @@ fn run_internal() -> anyhow::Result<()> { .next() .context("rootless sandbox invocation is missing the Agent executable")?; let arguments = arguments.collect::>(); + let arg0 = std::env::var_os(SANDBOX_ARG0_ENV).unwrap_or_else(|| program.clone()); - enter_rootless_namespaces(plan.network) - .context("initialize rootless user and mount namespaces")?; - enter_child_pid_namespace().context("initialize private PID namespace")?; + enter_rootless_namespaces(plan.network).context("initialize rootless namespaces")?; + if plan.filesystem_isolated { + enter_child_pid_namespace().context("initialize private PID namespace")?; + } if let Some(limit) = plan.process_limit { apply_process_limit(limit).context("apply Agent process limit")?; } // Open the parent-owned inode before chroot/Landlock. The descriptor is // retained only by trusted setup code and closed before Agent execution, // so no attestation pathname needs to be projected into the sandbox. - let attestation = std::fs::OpenOptions::new() + let mut attestation = std::fs::OpenOptions::new() .write(true) .open(&plan.attestation) .with_context(|| { @@ -131,12 +152,15 @@ fn run_internal() -> anyhow::Result<()> { plan.attestation.display() ) })?; - enter_synthetic_root(&plan).context("construct private sandbox root")?; - // The private tmpfs created by `enter_synthetic_root` is writable by the - // Agent, but must also be present in the Landlock allowlist. This uses the - // host-side mount path because rules are installed before chroot. let mut plan = plan; - plan.read_write.push(PathBuf::from("/tmp")); + if plan.filesystem_isolated { + enter_synthetic_root(&plan).context("construct private sandbox root")?; + // The private tmpfs created by `enter_synthetic_root` is writable by + // the Agent, but must also be present in the Landlock allowlist. This + // uses the host-side mount path because rules are installed before + // chroot. + plan.read_write.push(PathBuf::from("/tmp")); + } std::env::set_current_dir(&plan.cwd) .with_context(|| format!("enter sandbox workspace {}", plan.cwd.display()))?; @@ -144,14 +168,35 @@ fn run_internal() -> anyhow::Result<()> { // removes access to the host procfs tree. close_unexpected_file_descriptors(Some(attestation.as_raw_fd())) .context("close inherited file descriptors")?; - let landlock_abi = install_landlock(&plan).context("install Landlock filesystem policy")?; + let landlock_abi = if plan.filesystem_isolated { + let abi = install_landlock(&plan).context("install Landlock filesystem policy")?; + Some(abi) + } else { + None + }; + // Network-only runs still use a user namespace so the Agent cannot create + // another namespace and escape the deny-all network policy. The UID/GID + // mapping preserves ordinary host filesystem permissions; skipping + // Landlock keeps the host filesystem view unrestricted. drop_process_capabilities().context("drop namespace capabilities")?; // The child process is configuring its environment immediately before // exec; no concurrent environment mutation occurs in this scope. unsafe { std::env::remove_var(SANDBOX_PLAN_ENV); - std::env::set_var("PERSISTING_SANDBOX_FILESYSTEM", "landlock"); - std::env::set_var("PERSISTING_SANDBOX_LANDLOCK_ABI", landlock_abi.to_string()); + std::env::remove_var(SANDBOX_ARG0_ENV); + std::env::set_var( + "PERSISTING_SANDBOX_FILESYSTEM", + if plan.filesystem_isolated { + "landlock" + } else { + "host" + }, + ); + if let Some(landlock_abi) = landlock_abi { + std::env::set_var("PERSISTING_SANDBOX_LANDLOCK_ABI", landlock_abi.to_string()); + } else { + std::env::remove_var("PERSISTING_SANDBOX_LANDLOCK_ABI"); + } std::env::set_var("PERSISTING_SANDBOX_USER_NAMESPACE", "1"); std::env::set_var( "PERSISTING_SANDBOX_NETWORK", @@ -163,7 +208,20 @@ fn run_internal() -> anyhow::Result<()> { ); } - supervise_pid_namespace(program, arguments, attestation) + if !plan.filesystem_isolated { + // Network-only runs deliberately do not enter CLONE_NEWPID. Exec the + // Agent in the launcher's existing process group so the outer + // ProcessExecutor can terminate that group without the PID-namespace + // supervisor's kill(-1) semantics reaching unrelated host processes. + write_rootless_attestation(&mut attestation) + .context("record installed rootless network controls")?; + drop(attestation); + let mut command = std::process::Command::new(program); + command.args(arguments).arg0(arg0); + return Err(command.exec().into()); + } + + supervise_pid_namespace(program, arguments, arg0, attestation) } #[cfg(target_os = "macos")] @@ -183,6 +241,7 @@ fn run_internal() -> anyhow::Result<()> { .next() .context("Seatbelt sandbox invocation is missing the Agent executable")?; let arguments = arguments.collect::>(); + let arg0 = std::env::var_os(SANDBOX_ARG0_ENV).unwrap_or_else(|| program.clone()); // The parent keeps the already-open inode and checks these bytes after the // process exits. Unlinking before Agent execution keeps the random path and @@ -214,7 +273,15 @@ fn run_internal() -> anyhow::Result<()> { // exec; no concurrent environment mutation occurs in this scope. unsafe { std::env::remove_var(SANDBOX_PLAN_ENV); - std::env::set_var("PERSISTING_SANDBOX_FILESYSTEM", "seatbelt-write"); + std::env::remove_var(SANDBOX_ARG0_ENV); + std::env::set_var( + "PERSISTING_SANDBOX_FILESYSTEM", + if plan.filesystem_isolated { + "seatbelt-write" + } else { + "host" + }, + ); std::env::set_var( "PERSISTING_SANDBOX_NETWORK", if plan.network.is_loopback_only() { @@ -225,10 +292,9 @@ fn run_internal() -> anyhow::Result<()> { ); } - Err(std::process::Command::new(program) - .args(arguments) - .exec() - .into()) + let mut command = std::process::Command::new(program); + command.args(arguments).arg0(arg0); + Err(command.exec().into()) } #[cfg(target_os = "linux")] @@ -372,6 +438,7 @@ pub(crate) fn restrict_krun_runner( read_only, read_write, network: NetworkIsolation::LoopbackOnly, + filesystem_isolated: true, process_limit: None, }; let abi = install_landlock(&plan).context("install libkrun Landlock policy")?; @@ -499,6 +566,7 @@ pub(crate) fn seatbelt_profile( allowed_unix_sockets: &[PathBuf], local_socket_roots: &[PathBuf], network: NetworkIsolation, + filesystem_isolated: bool, ) -> std::io::Result<(String, Vec<(String, PathBuf)>)> { use std::io::{Error, ErrorKind}; @@ -577,14 +645,18 @@ pub(crate) fn seatbelt_profile( (require-not (remote ip \"localhost:*\"))))\n\ (allow network-outbound (remote ip \"localhost:*\"))\n", ); - profile.push_str("(allow file-write*\n"); - for index in 0..writable_paths.len() { - profile.push_str(&format!( - " (literal (param \"PVISOR_WRITABLE_{index}\"))\n\ - (subpath (param \"PVISOR_WRITABLE_{index}\"))\n" - )); + if filesystem_isolated { + profile.push_str("(allow file-write*\n"); + for index in 0..writable_paths.len() { + profile.push_str(&format!( + " (literal (param \"PVISOR_WRITABLE_{index}\"))\n\ + (subpath (param \"PVISOR_WRITABLE_{index}\"))\n" + )); + } + profile.push_str(")\n"); + } else { + profile.push_str("(allow file-write*)\n"); } - profile.push_str(")\n"); profile.push_str("(deny network-outbound\n (require-all\n (remote unix-socket)\n"); for index in 0..allowed_unix_sockets.len() { profile.push_str(&format!( @@ -728,11 +800,11 @@ fn bring_loopback_up() -> std::io::Result<()> { }; ifreq.name[0] = b'l' as libc::c_char; ifreq.name[1] = b'o' as libc::c_char; - if unsafe { libc::ioctl(fd, libc::SIOCGIFFLAGS, &mut ifreq) } != 0 { + if unsafe { libc::ioctl(fd, libc::SIOCGIFFLAGS as _, &mut ifreq) } != 0 { return Err(std::io::Error::last_os_error()); } ifreq.flags |= libc::IFF_UP as libc::c_short | libc::IFF_RUNNING as libc::c_short; - if unsafe { libc::ioctl(fd, libc::SIOCSIFFLAGS, &ifreq) } != 0 { + if unsafe { libc::ioctl(fd, libc::SIOCSIFFLAGS as _, &ifreq) } != 0 { return Err(std::io::Error::last_os_error()); } Ok(()) @@ -960,6 +1032,10 @@ fn drop_process_capabilities() -> std::io::Result<()> { inheritable: u32, } + if unsafe { libc::prctl(libc::PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) } != 0 { + return Err(Error::last_os_error()); + } + let mut header = CapabilityHeader { version: LINUX_CAPABILITY_VERSION_3, pid: 0, @@ -1048,6 +1124,7 @@ fn exit_with_wait_status(status: libc::c_int) -> ! { fn supervise_pid_namespace( program: std::ffi::OsString, arguments: Vec, + arg0: std::ffi::OsString, mut attestation: std::fs::File, ) -> anyhow::Result<()> { use anyhow::Context; @@ -1193,7 +1270,9 @@ fn supervise_pid_namespace( if release_count != 1 || release != 1 { unsafe { libc::_exit(SANDBOX_SETUP_EXIT_CODE) }; } - let error = std::process::Command::new(program).args(arguments).exec(); + let mut command = std::process::Command::new(program); + command.args(arguments).arg0(arg0); + let error = command.exec(); eprintln!("pvisor: execute sandboxed Agent: {error}"); unsafe { libc::_exit(SANDBOX_SETUP_EXIT_CODE) }; } @@ -1261,6 +1340,7 @@ mod tests { &[], &[], NetworkIsolation::LoopbackOnly, + true, ) .unwrap(); @@ -1270,8 +1350,14 @@ mod tests { assert!(profile.contains("(remote ip \"localhost:*\")")); assert!(profile.contains("(allow network-outbound (remote ip \"localhost:*\"))")); - let error = seatbelt_profile(&[PathBuf::from("/")], &[], &[], NetworkIsolation::Ambient) - .unwrap_err(); + let error = seatbelt_profile( + &[PathBuf::from("/")], + &[], + &[], + NetworkIsolation::Ambient, + true, + ) + .unwrap_err(); assert_eq!(error.kind(), std::io::ErrorKind::InvalidInput); } } diff --git a/crates/persisting-pvisor/tests/rootless_local.rs b/crates/persisting-pvisor/tests/rootless_local.rs index 706702e0..905870f9 100644 --- a/crates/persisting-pvisor/tests/rootless_local.rs +++ b/crates/persisting-pvisor/tests/rootless_local.rs @@ -53,13 +53,18 @@ fn setup_failure(root: &Path) -> Option { } fn user_namespaces_are_unavailable(stderr: &str) -> bool { - const CONTEXT: &str = "initialize rootless user and mount namespaces: "; + const CONTEXTS: &[&str] = &[ + "initialize rootless namespaces: ", + "initialize rootless user and mount namespaces: ", + ]; stderr.lines().any(|line| { - let Some((_, error)) = line.split_once(CONTEXT) else { - return false; - }; - error == "unshare user namespace: Operation not permitted (os error 1)" - || error == "unshare user namespace: Permission denied (os error 13)" + CONTEXTS.iter().any(|context| { + let Some((_, error)) = line.split_once(context) else { + return false; + }; + error == "unshare user namespace: Operation not permitted (os error 1)" + || error == "unshare user namespace: Permission denied (os error 13)" + }) }) } @@ -553,6 +558,7 @@ fn denied_network_uses_a_private_network_namespace() { let temporary = tempfile::tempdir().unwrap(); let workspace = temporary.path().join("workspace"); let run_home = temporary.path().join("runs"); + let host_write = temporary.path().join("host-visible.txt"); fs::create_dir_all(&workspace).unwrap(); let listener = TcpListener::bind("127.0.0.1:0").unwrap(); let host_port = listener.local_addr().unwrap().port().to_string(); @@ -560,6 +566,8 @@ fn denied_network_uses_a_private_network_namespace() { let script = r#" set -eu test "$PERSISTING_SANDBOX_NETWORK" = deny +test "$PERSISTING_SANDBOX_FILESYSTEM" = host +printf 'host-visible' > "$HOST_WRITE" if exec 3<>"/dev/tcp/127.0.0.1/${HOST_PORT}"; then echo 'host listener unexpectedly reachable' >&2 exit 50 @@ -578,8 +586,11 @@ printf 'network:%s\n' "$PERSISTING_SANDBOX_NETWORK" "--overlaynet-deny-all", "--pass-env", "HOST_PORT", + "--pass-env", + "HOST_WRITE", ]) .current_dir(&workspace) + .env("HOST_WRITE", &host_write) .args(["--", "/bin/bash", "-c", script]) .output() .unwrap(); @@ -593,6 +604,7 @@ printf 'network:%s\n' "$PERSISTING_SANDBOX_NETWORK" String::from_utf8_lossy(&output.stdout), String::from_utf8_lossy(&output.stderr) ); + assert_eq!(fs::read_to_string(&host_write).unwrap(), "host-visible"); let run = only_run(&stage_root(&run_home)); let bundle = RunBundle::read(&run).unwrap(); From fe82d1bd0ae9f641dd63312b218ac2dae32db10b Mon Sep 17 00:00:00 2001 From: Lucius <54578015+lizhicui@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:43:17 +0800 Subject: [PATCH 2/5] feat(pvisor): support static musl libkrun builds --- Cargo.lock | 1 + Cargo.toml | 7 + crates/persisting-pvisor/Cargo.toml | 8 +- crates/persisting-pvisor/build.rs | 113 +++++++++++++++ crates/persisting-pvisor/src/vm/mod.rs | 15 ++ .../src/{vm.rs => vm/supported.rs} | 39 ++++++ .../persisting-pvisor/src/vm/unsupported.rs | 77 +++++++++++ scripts/extract-libkrun-kernel.py | 60 ++++++++ vendor/libkrun/Cargo.toml | 6 +- vendor/libkrun/Cargo.toml.orig | 4 +- vendor/libkrun/src/lib.rs | 129 +++++++++++++++++- 11 files changed, 451 insertions(+), 8 deletions(-) create mode 100644 crates/persisting-pvisor/build.rs create mode 100644 crates/persisting-pvisor/src/vm/mod.rs rename crates/persisting-pvisor/src/{vm.rs => vm/supported.rs} (96%) create mode 100644 crates/persisting-pvisor/src/vm/unsupported.rs create mode 100644 scripts/extract-libkrun-kernel.py diff --git a/Cargo.lock b/Cargo.lock index 50304da7..96203ff7 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2951,6 +2951,7 @@ dependencies = [ "globset", "libc", "libkrun", + "libloading", "persisting-control", "persisting-gateway", "persisting-overlay-core", diff --git a/Cargo.toml b/Cargo.toml index b116a4e7..6eced498 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -46,6 +46,7 @@ hyper-util = { version = "0.1", default-features = false } ipnet = "2" jj-lib = "0.43.0" libc = "0.2" +libloading = "0.8" libkrun = "=1.19.3" log = "0.4" persisting-control = { path = "crates/persisting-control" } @@ -116,6 +117,12 @@ strip = "symbols" codegen-units = 8 panic = "abort" +# Build scripts and proc macros do not need release stripping. Keeping their +# symbols avoids invoking a host rust-objcopy that may be unavailable when +# cross-compiling from macOS, while the final application remains stripped. +[profile.release.build-override] +strip = "none" + [patch.crates-io] fuser = { path = "vendor/fuser" } libkrun = { path = "vendor/libkrun" } diff --git a/crates/persisting-pvisor/Cargo.toml b/crates/persisting-pvisor/Cargo.toml index 0ec2ec0f..6989bc2b 100644 --- a/crates/persisting-pvisor/Cargo.toml +++ b/crates/persisting-pvisor/Cargo.toml @@ -6,6 +6,7 @@ authors.workspace = true license.workspace = true description = "pVisor foreground Agent Run manager and portable execution runtime" readme = "README.md" +build = "build.rs" [features] default = [] @@ -21,7 +22,6 @@ flate2.workspace = true fs2.workspace = true globset.workspace = true libc.workspace = true -libkrun = { workspace = true, features = ["net"] } persisting-control.workspace = true persisting-gateway = { workspace = true, default-features = false } persisting-overlaynet.workspace = true @@ -42,9 +42,15 @@ uuid = { workspace = true, features = ["v4"] } zstd.workspace = true reqwest = { workspace = true, features = ["blocking", "json", "rustls-tls"] } +[target.'cfg(not(all(target_os = "macos", target_arch = "x86_64")))'.dependencies] +libkrun = { workspace = true, features = ["net"] } + [[bin]] name = "pvisor" path = "src/bin/pvisor.rs" [dev-dependencies] proptest.workspace = true + +[build-dependencies] +libloading.workspace = true diff --git a/crates/persisting-pvisor/build.rs b/crates/persisting-pvisor/build.rs new file mode 100644 index 00000000..1af1e585 --- /dev/null +++ b/crates/persisting-pvisor/build.rs @@ -0,0 +1,113 @@ +use std::env; +use std::fs; +use std::path::PathBuf; + +fn main() { + println!("cargo:rerun-if-env-changed=PERSISTING_KRUNFW_PATH"); + println!("cargo:rerun-if-env-changed=PERSISTING_KRUNFW_KERNEL_BUNDLE"); + + if env::var("TARGET").as_deref() != Ok("x86_64-unknown-linux-musl") { + return; + } + + let out_dir = PathBuf::from(env::var_os("OUT_DIR").expect("Cargo did not set OUT_DIR")); + let bundle_dir = env::var_os("PERSISTING_KRUNFW_KERNEL_BUNDLE").map(PathBuf::from); + let (kernel, guest_addr, entry_addr) = if let Some(directory) = bundle_dir { + println!( + "cargo:rerun-if-changed={}", + directory.join("kernel.bin").display() + ); + println!( + "cargo:rerun-if-changed={}", + directory.join("kernel.json").display() + ); + let kernel = fs::read(directory.join("kernel.bin")).unwrap_or_else(|error| { + panic!( + "read embedded libkrun kernel bundle {}: {error}", + directory.join("kernel.bin").display() + ) + }); + let metadata = fs::read_to_string(directory.join("kernel.json")).unwrap_or_else(|error| { + panic!( + "read embedded libkrun kernel metadata {}: {error}", + directory.join("kernel.json").display() + ) + }); + let guest_addr = metadata_value(&metadata, "guest_addr"); + let entry_addr = metadata_value(&metadata, "entry_addr"); + (kernel, guest_addr, entry_addr) + } else { + let path = env::var_os("PERSISTING_KRUNFW_PATH").unwrap_or_else(|| { + panic!( + "building x86_64 Linux musl requires \ + PERSISTING_KRUNFW_PATH pointing to libkrunfw.so.5, or \ + PERSISTING_KRUNFW_KERNEL_BUNDLE pointing to a directory containing \ + kernel.bin and kernel.json" + ) + }); + println!("cargo:rerun-if-changed={}", PathBuf::from(&path).display()); + extract_kernel_bundle(&PathBuf::from(path)) + }; + + assert!(!kernel.is_empty(), "libkrun kernel bundle is empty"); + let kernel_path = out_dir.join("embedded-libkrun-kernel.bin"); + fs::write(&kernel_path, &kernel).unwrap_or_else(|error| { + panic!( + "write embedded libkrun kernel {}: {error}", + kernel_path.display() + ) + }); + let generated = out_dir.join("embedded_kernel.rs"); + let kernel_literal = format!("{:?}", kernel_path.to_string_lossy()); + let source = format!( + "pub static KERNEL: &[u8] = include_bytes!({kernel_literal});\n\ + pub const GUEST_ADDR: u64 = {guest_addr};\n\ + pub const ENTRY_ADDR: u64 = {entry_addr};\n" + ); + fs::write(&generated, source) + .unwrap_or_else(|error| panic!("write generated kernel module: {error}")); +} + +fn extract_kernel_bundle(path: &std::path::Path) -> (Vec, u64, u64) { + type GetKernel = unsafe extern "C" fn(*mut u64, *mut u64, *mut usize) -> *mut std::ffi::c_char; + + let library = unsafe { libloading::Library::new(path) }.unwrap_or_else(|error| { + panic!( + "load libkrunfw {} while extracting kernel bundle: {error}", + path.display() + ) + }); + let get_kernel = unsafe { library.get::(b"krunfw_get_kernel\0") } + .unwrap_or_else(|error| panic!("resolve krunfw_get_kernel: {error}")); + let mut guest_addr = 0; + let mut entry_addr = 0; + let mut size = 0; + let pointer = unsafe { get_kernel(&mut guest_addr, &mut entry_addr, &mut size) }; + if pointer.is_null() || size == 0 { + panic!("krunfw_get_kernel returned an empty kernel bundle"); + } + let kernel = unsafe { std::slice::from_raw_parts(pointer.cast::(), size) }.to_vec(); + (kernel, guest_addr, entry_addr) +} + +fn metadata_value(metadata: &str, key: &str) -> u64 { + let marker = format!("\"{key}\""); + let token = metadata + .split_once(&marker) + .and_then(|(_, rest)| rest.split_once(':')) + .and_then(|(_, rest)| { + let value = rest + .chars() + .skip_while(|character| !character.is_ascii_hexdigit() && *character != 'x') + .take_while(|character| character.is_ascii_hexdigit() || *character == 'x') + .collect::(); + (!value.is_empty()).then_some(value) + }) + .unwrap_or_else(|| panic!("kernel metadata is missing numeric {key}")); + if let Some(hex) = token.strip_prefix("0x") { + u64::from_str_radix(hex, 16) + } else { + token.parse() + } + .unwrap_or_else(|error| panic!("kernel metadata {key} is not a valid integer: {error}")) +} diff --git a/crates/persisting-pvisor/src/vm/mod.rs b/crates/persisting-pvisor/src/vm/mod.rs new file mode 100644 index 00000000..ef883344 --- /dev/null +++ b/crates/persisting-pvisor/src/vm/mod.rs @@ -0,0 +1,15 @@ +#[cfg(not(all(target_os = "macos", target_arch = "x86_64")))] +mod supported; +#[cfg(all(target_os = "macos", target_arch = "x86_64"))] +mod unsupported; + +#[cfg(not(all(target_os = "macos", target_arch = "x86_64")))] +pub use supported::{VmExecutor, run_internal_if_requested}; +#[cfg(not(any( + all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), + all(target_os = "macos", target_arch = "x86_64") +)))] +pub(crate) use supported::{bundled_firmware_dir, firmware_name}; + +#[cfg(all(target_os = "macos", target_arch = "x86_64"))] +pub use unsupported::{VmExecutor, run_internal_if_requested}; diff --git a/crates/persisting-pvisor/src/vm.rs b/crates/persisting-pvisor/src/vm/supported.rs similarity index 96% rename from crates/persisting-pvisor/src/vm.rs rename to crates/persisting-pvisor/src/vm/supported.rs index eadebbad..260581d3 100644 --- a/crates/persisting-pvisor/src/vm.rs +++ b/crates/persisting-pvisor/src/vm/supported.rs @@ -25,6 +25,14 @@ const NETWORK_FD_ENV: &str = "PERSISTING_KRUN_NETWORK_FD"; const NETWORK_CHILD_FD: RawFd = 198; const NET_FLAG_DHCP_CLIENT: u32 = 1 << 1; +#[cfg(all(target_os = "linux", target_env = "musl", not(target_arch = "x86_64")))] +compile_error!("static musl VM support currently targets x86_64 only"); + +#[cfg(all(target_os = "linux", target_env = "musl", target_arch = "x86_64"))] +mod embedded_kernel { + include!(concat!(env!("OUT_DIR"), "/embedded_kernel.rs")); +} + #[derive(Debug, Clone)] pub struct VmExecutor { settings: VmSettings, @@ -72,6 +80,11 @@ impl VmExecutor { anyhow::ensure!(settings.memory_mib > 0, "vm.memory_mib must be positive"); anyhow::ensure!(settings.cpus > 0, "vm.cpus must be positive"); anyhow::ensure!(settings.cpus <= 8, "libkrunfw supports at most 8 vCPUs"); + #[cfg(all(target_os = "linux", target_env = "musl", target_arch = "x86_64"))] + anyhow::ensure!( + settings.library_dir.is_none(), + "vm.library_dir is unavailable in the static musl build; libkrun's kernel bundle is embedded" + ); let rootfs = settings .rootfs .as_deref() @@ -173,6 +186,19 @@ impl RunExecutor for VmExecutor { "libkrun execution requires Linux/KVM or Apple Silicon macOS/HVF".into(), ); } + #[cfg(target_os = "linux")] + if let Err(error) = std::fs::OpenOptions::new() + .read(true) + .write(true) + .open("/dev/kvm") + { + return failed_to_start( + &spec, + context.attempt_id(), + started_at, + format!("libkrun VM requires an accessible /dev/kvm: {error}"), + ); + } let RunInvocation::Process(invocation) = &mut spec.invocation; let overlay_target = spec @@ -706,6 +732,19 @@ fn run_linked_krun(spec: RunnerSpec) -> anyhow::Result<()> { krun::krun_set_vm_config(ctx, spec.cpus, spec.memory_mib), "krun_set_vm_config", )?; + #[cfg(all(target_os = "linux", target_env = "musl", target_arch = "x86_64"))] + check_krun( + unsafe { + krun::krun_set_embedded_kernel( + ctx, + embedded_kernel::KERNEL.as_ptr(), + embedded_kernel::KERNEL.len(), + embedded_kernel::GUEST_ADDR, + embedded_kernel::ENTRY_ADDR, + ) + }, + "krun_set_embedded_kernel", + )?; add_krun_overlay(ctx, "/dev/root", &spec.root, 1 << 29)?; if let Some(workspace) = &spec.workspace { add_krun_overlay(ctx, workspace_tag.to_str()?, workspace, 0)?; diff --git a/crates/persisting-pvisor/src/vm/unsupported.rs b/crates/persisting-pvisor/src/vm/unsupported.rs new file mode 100644 index 00000000..762586b1 --- /dev/null +++ b/crates/persisting-pvisor/src/vm/unsupported.rs @@ -0,0 +1,77 @@ +//! Intel macOS VM stub. +//! +//! libkrun's macOS backend is supported only on Apple Silicon. Keeping this +//! target out of the libkrun dependency graph lets the host/container CLI and +//! their tests build on Intel macOS while producing a clear VM error. + +use crate::config::VmSettings; +use crate::executor::{AttemptContext, RunExecutor}; +use async_trait::async_trait; +use persisting_control::{ + CapabilityEnforcementEvidence, ExecutorDescriptor, ExecutorKind, IsolationKind, ProcessOutput, + RunFailure, RunFailureKind, RunInvocation, RunResult, RunState, +}; + +const UNSUPPORTED_MESSAGE: &str = + "VM execution is unsupported on Intel macOS; use Linux x86_64 or Apple Silicon macOS"; + +#[derive(Debug, Clone)] +pub struct VmExecutor { + settings: VmSettings, +} + +impl VmExecutor { + pub fn new(_settings: VmSettings) -> anyhow::Result { + anyhow::bail!(UNSUPPORTED_MESSAGE) + } + + pub fn settings(&self) -> &VmSettings { + &self.settings + } +} + +#[async_trait] +impl RunExecutor for VmExecutor { + fn descriptor(&self) -> ExecutorDescriptor { + ExecutorDescriptor { + name: "libkrun-root-overlay-v1".into(), + kind: ExecutorKind::VirtualMachine, + isolation: IsolationKind::VirtualMachine, + capability_enforcement: CapabilityEnforcementEvidence::default(), + supports_checkpoint: false, + supports_migration: false, + } + } + + fn supports(&self, invocation: &RunInvocation) -> bool { + matches!(invocation, RunInvocation::Process(_)) + } + + async fn execute(&self, context: AttemptContext) -> RunResult { + let spec = context.spec(); + RunResult { + run_id: spec.run_id.clone(), + attempt_id: context.attempt_id().clone(), + lease_epoch: spec.lease_epoch, + state: RunState::Failed, + started_at_unix_ms: crate::util::unix_now_ms(), + finished_at_unix_ms: crate::util::unix_now_ms(), + exit_code: None, + failure: Some(RunFailure { + kind: RunFailureKind::Spawn, + message: UNSUPPORTED_MESSAGE.into(), + retryable: false, + }), + output: ProcessOutput::default(), + value: None, + metrics: Default::default(), + artifacts: Vec::new(), + event_stream_ref: None, + warnings: Vec::new(), + } + } +} + +pub fn run_internal_if_requested() -> anyhow::Result { + Ok(false) +} diff --git a/scripts/extract-libkrun-kernel.py b/scripts/extract-libkrun-kernel.py new file mode 100644 index 00000000..db227f03 --- /dev/null +++ b/scripts/extract-libkrun-kernel.py @@ -0,0 +1,60 @@ +#!/usr/bin/env python3 +"""Extract a libkrunfw kernel bundle on a host that can load the firmware.""" + +from __future__ import annotations + +import argparse +import ctypes +import json +from pathlib import Path + + +def extract(source: Path, destination: Path) -> None: + library = ctypes.CDLL(str(source)) + get_kernel = library.krunfw_get_kernel + get_kernel.argtypes = [ + ctypes.POINTER(ctypes.c_uint64), + ctypes.POINTER(ctypes.c_uint64), + ctypes.POINTER(ctypes.c_size_t), + ] + get_kernel.restype = ctypes.c_void_p + + guest_addr = ctypes.c_uint64() + entry_addr = ctypes.c_uint64() + size = ctypes.c_size_t() + pointer = get_kernel( + ctypes.byref(guest_addr), + ctypes.byref(entry_addr), + ctypes.byref(size), + ) + if not pointer or not size.value: + raise RuntimeError("krunfw_get_kernel returned an empty kernel bundle") + + destination.mkdir(parents=True, exist_ok=True) + (destination / "kernel.bin").write_bytes( + ctypes.string_at(pointer, size.value) + ) + (destination / "kernel.json").write_text( + json.dumps( + { + "guest_addr": guest_addr.value, + "entry_addr": entry_addr.value, + "size": size.value, + }, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("libkrunfw", type=Path) + parser.add_argument("destination", type=Path) + args = parser.parse_args() + extract(args.libkrunfw.resolve(), args.destination.resolve()) + + +if __name__ == "__main__": + main() diff --git a/vendor/libkrun/Cargo.toml b/vendor/libkrun/Cargo.toml index 0a4cc0fa..e24c470c 100644 --- a/vendor/libkrun/Cargo.toml +++ b/vendor/libkrun/Cargo.toml @@ -118,9 +118,6 @@ package = "krun-input" [dependencies.libc] version = ">=0.2.39" -[dependencies.libloading] -version = "0.8" - [dependencies.log] version = "0.4.0" @@ -139,6 +136,9 @@ package = "krun-utils" version = "=0.1.0-1.19.3" package = "krun-vmm" +[target.'cfg(not(target_env = "musl"))'.dependencies.libloading] +version = "0.8" + [target.'cfg(target_os = "linux")'.dependencies.aws-nitro] version = "=0.1.0-1.19.3" optional = true diff --git a/vendor/libkrun/Cargo.toml.orig b/vendor/libkrun/Cargo.toml.orig index e3613a71..1a62edfe 100644 --- a/vendor/libkrun/Cargo.toml.orig +++ b/vendor/libkrun/Cargo.toml.orig @@ -27,7 +27,6 @@ aws-nitro = ["vmm/aws-nitro", "devices/aws-nitro", "dep:aws-nitro", "dep:nitro-e crossbeam-channel = ">=0.5.15" env_logger = "0.11" libc = ">=0.2.39" -libloading = "0.8" log = "0.4.0" once_cell = "1.4.1" krun_display = { package = "krun-display", version = "0.1.0", path = "../display", optional = true, features = ["bindgen_clang_runtime"] } @@ -39,6 +38,9 @@ polly = { package = "krun-polly", version = "=0.1.0-1.19.3", path = "../polly" } utils = { package = "krun-utils", version = "=0.1.0-1.19.3", path = "../utils" } vmm = { package = "krun-vmm", version = "=0.1.0-1.19.3", path = "../vmm" } +[target.'cfg(not(target_env = "musl"))'.dependencies] +libloading = "0.8" + [target.'cfg(target_os = "macos")'.dependencies] hvf = { package = "krun-hvf", version = "=0.1.0-1.19.3", path = "../hvf" } diff --git a/vendor/libkrun/src/lib.rs b/vendor/libkrun/src/lib.rs index ef00e8eb..b2e3f66f 100644 --- a/vendor/libkrun/src/lib.rs +++ b/vendor/libkrun/src/lib.rs @@ -37,6 +37,7 @@ use std::os::fd::{BorrowedFd, FromRawFd, RawFd}; use std::path::PathBuf; use std::slice; use std::sync::atomic::{AtomicI32, Ordering}; +#[cfg(not(target_env = "musl"))] use std::sync::LazyLock; use std::sync::Mutex; use utils::eventfd::EventFd; @@ -75,11 +76,23 @@ const KRUN_SUCCESS: i32 = 0; const MAX_ARGS: usize = 4096; // krunfw library name for each context -#[cfg(all(target_os = "linux", not(feature = "tee")))] +#[cfg(all( + target_os = "linux", + not(target_env = "musl"), + not(feature = "tee") +))] const KRUNFW_NAME: &str = "libkrunfw.so.5"; -#[cfg(all(target_os = "linux", feature = "amd-sev"))] +#[cfg(all( + target_os = "linux", + not(target_env = "musl"), + feature = "amd-sev" +))] const KRUNFW_NAME: &str = "libkrunfw-sev.so.5"; -#[cfg(all(target_os = "linux", feature = "tdx"))] +#[cfg(all( + target_os = "linux", + not(target_env = "musl"), + feature = "tdx" +))] const KRUNFW_NAME: &str = "libkrunfw-tdx.so.5"; #[cfg(target_os = "macos")] const KRUNFW_NAME: &str = "libkrunfw.5.dylib"; @@ -113,9 +126,11 @@ fn init_virtual_entry() -> VirtualDirEntry { } } +#[cfg(not(target_env = "musl"))] static KRUNFW: LazyLock> = LazyLock::new(|| unsafe { libloading::Library::new(KRUNFW_NAME).ok() }); +#[cfg(not(target_env = "musl"))] pub struct KrunfwBindings { get_kernel: libloading::Symbol< 'static, @@ -127,6 +142,7 @@ pub struct KrunfwBindings { get_qboot: libloading::Symbol<'static, unsafe extern "C" fn(*mut size_t) -> *mut c_char>, } +#[cfg(not(target_env = "musl"))] impl KrunfwBindings { fn load_bindings() -> Result { let krunfw = match KRUNFW.as_ref() { @@ -149,6 +165,27 @@ impl KrunfwBindings { } } +/// A page-aligned copy of a kernel bundle supplied by the application. +/// +/// Static musl builds cannot use libloading to discover libkrunfw at runtime. +/// Keeping the mapping in the context makes the pointer stored in +/// `KernelBundle` valid until the VMM has consumed it. +struct EmbeddedKernelMapping { + address: *mut libc::c_void, + mapped_size: usize, +} + +unsafe impl Send for EmbeddedKernelMapping {} + +impl Drop for EmbeddedKernelMapping { + fn drop(&mut self) { + // SAFETY: the mapping was created by mmap with this exact size. + unsafe { + libc::munmap(self.address, self.mapped_size); + } + } +} + #[derive(Clone)] #[cfg(feature = "net")] enum LegacyNetworkConfig { @@ -158,7 +195,9 @@ enum LegacyNetworkConfig { #[derive(Default)] struct ContextConfig { + #[cfg(not(target_env = "musl"))] krunfw: Option, + embedded_kernel: Option, vmr: VmResources, workdir: Option, exec_path: Option, @@ -552,7 +591,9 @@ pub extern "C" fn krun_create_ctx() -> i32 { let ctx_cfg = { ContextConfig { + #[cfg(not(target_env = "musl"))] krunfw: KrunfwBindings::new(), + embedded_kernel: None, shutdown_efd, ..Default::default() } @@ -576,6 +617,76 @@ pub extern "C" fn krun_free_ctx(ctx_id: u32) -> i32 { } } +/// Install a kernel bundle that is already present in the caller's address +/// space. This is used by static Linux builds, where loading libkrunfw with +/// `dlopen` is unavailable. The bytes are copied into a page-aligned mapping +/// owned by the context before the VMM starts. +#[allow(clippy::missing_safety_doc)] +#[no_mangle] +pub unsafe extern "C" fn krun_set_embedded_kernel( + ctx_id: u32, + kernel: *const u8, + kernel_size: usize, + guest_addr: u64, + entry_addr: u64, +) -> i32 { + if kernel.is_null() || kernel_size == 0 { + return -libc::EINVAL; + } + + // SAFETY: the caller promises that `kernel` points to `kernel_size` bytes. + let bytes = unsafe { std::slice::from_raw_parts(kernel, kernel_size) }; + let page_size = unsafe { libc::sysconf(libc::_SC_PAGESIZE) as usize }; + if page_size == 0 || !page_size.is_power_of_two() { + return -libc::EINVAL; + } + let mapped_size = match kernel_size.checked_add(page_size - 1) { + Some(size) => size & !(page_size - 1), + None => return -libc::EOVERFLOW, + }; + // SAFETY: anonymous private memory is owned by this context after this + // call and is released by EmbeddedKernelMapping::drop. + let address = unsafe { + libc::mmap( + std::ptr::null_mut(), + mapped_size, + libc::PROT_READ | libc::PROT_WRITE, + libc::MAP_PRIVATE | libc::MAP_ANONYMOUS, + -1, + 0, + ) + }; + if address == libc::MAP_FAILED { + return -libc::ENOMEM; + } + // Keep the mapping writable: KVM exposes this host memory directly as + // guest RAM, and early kernel boot may write its own data sections. + unsafe { + std::ptr::copy_nonoverlapping(bytes.as_ptr(), address.cast::(), kernel_size); + } + + let bundle = KernelBundle { + host_addr: address as u64, + guest_addr, + entry_addr, + size: kernel_size, + }; + let mut contexts = CTX_MAP.lock().unwrap(); + let Some(context) = contexts.get_mut(&ctx_id) else { + unsafe { libc::munmap(address, mapped_size) }; + return -libc::ENOENT; + }; + if context.vmr.set_kernel_bundle(bundle).is_err() { + unsafe { libc::munmap(address, mapped_size) }; + return -libc::EINVAL; + } + context.embedded_kernel = Some(EmbeddedKernelMapping { + address, + mapped_size, + }); + KRUN_SUCCESS +} + #[no_mangle] pub extern "C" fn krun_set_vm_config(ctx_id: u32, num_vcpus: u8, ram_mib: u32) -> i32 { let mem_size_mib: usize = match ram_mib.try_into() { @@ -2375,6 +2486,7 @@ pub unsafe extern "C" fn krun_set_firmware(ctx_id: u32, c_firmware_path: *const KRUN_SUCCESS } +#[cfg(not(target_env = "musl"))] unsafe fn load_krunfw_payload( krunfw: &KrunfwBindings, vmr: &mut VmResources, @@ -2986,6 +3098,7 @@ pub extern "C" fn krun_start_enter(ctx_id: u32) -> i32 { None => return -libc::ENOENT, }; + #[cfg(not(target_env = "musl"))] if ctx_cfg.vmr.external_kernel.is_none() && ctx_cfg.vmr.kernel_bundle.is_none() && ctx_cfg.vmr.firmware_config.is_none() @@ -3002,6 +3115,16 @@ pub extern "C" fn krun_start_enter(ctx_id: u32) -> i32 { } } + #[cfg(target_env = "musl")] + if ctx_cfg.vmr.external_kernel.is_none() + && ctx_cfg.vmr.kernel_bundle.is_none() + && ctx_cfg.vmr.firmware_config.is_none() + && cfg!(not(feature = "efi")) + { + eprintln!("No embedded kernel bundle was configured for this static build"); + return -libc::ENOENT; + } + #[cfg(feature = "blk")] for block_cfg in ctx_cfg.get_block_cfg() { if ctx_cfg.vmr.add_block_device(block_cfg).is_err() { From f43c98e262370fd2f0e9d36f38965ba2d242342b Mon Sep 17 00:00:00 2001 From: Lucius <54578015+lizhicui@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:43:17 +0800 Subject: [PATCH 3/5] fix(musl): make vendored Linux dependencies portable --- crates/persisting-overlay-core/src/sys.rs | 4 +- vendor/fuser/build.rs | 11 +++- .../src/virtio/fs/linux/passthrough.rs | 53 +++++++++++++++++-- 3 files changed, 60 insertions(+), 8 deletions(-) diff --git a/crates/persisting-overlay-core/src/sys.rs b/crates/persisting-overlay-core/src/sys.rs index 5828d8e6..d4e71c07 100644 --- a/crates/persisting-overlay-core/src/sys.rs +++ b/crates/persisting-overlay-core/src/sys.rs @@ -30,13 +30,13 @@ fn cvt(rc: libc::c_int) -> io::Result<()> { fn timespec(time: SystemTime) -> libc::timespec { match time.duration_since(UNIX_EPOCH) { Ok(value) => libc::timespec { - tv_sec: value.as_secs() as libc::time_t, + tv_sec: value.as_secs() as i64 as _, tv_nsec: value.subsec_nanos() as libc::c_long, }, Err(value) => { let value = value.duration(); libc::timespec { - tv_sec: -(value.as_secs() as libc::time_t) - 1, + tv_sec: (-(value.as_secs() as i64) - 1) as _, tv_nsec: 1_000_000_000 - value.subsec_nanos() as libc::c_long, } } diff --git a/vendor/fuser/build.rs b/vendor/fuser/build.rs index c519d97c..b2c9f54a 100644 --- a/vendor/fuser/build.rs +++ b/vendor/fuser/build.rs @@ -3,8 +3,15 @@ fn main() { // When fuser MSRV is updated to v1.77 or above, we should switch from 'cargo:' to 'cargo::' syntax. println!("cargo:rustc-check-cfg=cfg(fuser_mount_impl, values(\"pure-rust\", \"libfuse2\", \"libfuse3\"))"); - #[cfg(all(not(feature = "libfuse"), not(target_os = "linux")))] - unimplemented!("Building without libfuse is only supported on Linux"); + // Build scripts are compiled for the host, so `cfg(target_os)` describes + // the host rather than the target being compiled. Use Cargo's target + // environment variable here so cross-compiling the pure-Rust Linux + // implementation from macOS (or another host) works correctly. + if cfg!(not(feature = "libfuse")) + && std::env::var("CARGO_CFG_TARGET_OS").as_deref() != Ok("linux") + { + unimplemented!("Building without libfuse is only supported on Linux"); + } #[cfg(not(feature = "libfuse"))] { diff --git a/vendor/krun-devices/src/virtio/fs/linux/passthrough.rs b/vendor/krun-devices/src/virtio/fs/linux/passthrough.rs index 009121cf..6cec887a 100644 --- a/vendor/krun-devices/src/virtio/fs/linux/passthrough.rs +++ b/vendor/krun-devices/src/virtio/fs/linux/passthrough.rs @@ -195,20 +195,65 @@ fn stat(f: &File) -> io::Result { } } +// Linux exposes the statx ABI through libc on glibc targets, but musl's libc +// bindings do not consistently expose it for every supported version. Keep a +// local definition of the ABI and invoke the syscall directly so this code is +// independent of the C library's headers. +#[repr(C)] +#[derive(Clone, Copy, Debug, Default)] +struct LinuxStatxTimestamp { + tv_sec: i64, + tv_nsec: u32, + _reserved: i32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Default)] +struct LinuxStatx { + stx_mask: u32, + stx_blksize: u32, + stx_attributes: u64, + stx_nlink: u32, + stx_uid: u32, + stx_gid: u32, + stx_mode: u16, + _spare0: u16, + stx_ino: u64, + stx_size: u64, + stx_blocks: u64, + stx_attributes_mask: u64, + stx_atime: LinuxStatxTimestamp, + stx_btime: LinuxStatxTimestamp, + stx_ctime: LinuxStatxTimestamp, + stx_mtime: LinuxStatxTimestamp, + stx_rdev_major: u32, + stx_rdev_minor: u32, + stx_dev_major: u32, + stx_dev_minor: u32, + stx_mnt_id: u64, + stx_dio_mem_align: u32, + stx_dio_offset_align: u32, + _spare3: [u64; 12], +} + +const STATX_BASIC_STATS: u32 = 0x0000_07ff; +const STATX_MNT_ID: u32 = 0x0000_1000; + fn statx(f: &File) -> io::Result<(libc::stat64, u64)> { - let mut stx = MaybeUninit::::zeroed(); + let mut stx = MaybeUninit::::zeroed(); // Safe because this is a constant value and a valid C string. let pathname = unsafe { CStr::from_bytes_with_nul_unchecked(EMPTY_CSTR) }; - // Safe because the kernel will only write data in `st` and we check the return + // Safe because the kernel will only write data in `stx` and we check the return // value. let res = unsafe { - libc::statx( + libc::syscall( + libc::SYS_statx, f.as_raw_fd(), pathname.as_ptr(), libc::AT_EMPTY_PATH | libc::AT_SYMLINK_NOFOLLOW, - libc::STATX_BASIC_STATS | libc::STATX_MNT_ID, + STATX_BASIC_STATS | STATX_MNT_ID, stx.as_mut_ptr(), ) }; From 62fbf052c7c0c5e9d164e71dd559c5d74c1546c0 Mon Sep 17 00:00:00 2001 From: Lucius <54578015+lizhicui@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:43:17 +0800 Subject: [PATCH 4/5] docs(pvisor): document independent isolation policies --- docs/src/en/design/isolation.md | 7 ++++--- docs/src/en/guides/execution.md | 2 +- docs/src/en/guides/troubleshooting.md | 8 ++++---- docs/src/en/reference/cli.md | 27 +++++++++++++++------------ docs/src/zh/design/isolation.md | 7 ++++--- docs/src/zh/guides/execution.md | 2 +- docs/src/zh/guides/troubleshooting.md | 7 ++++--- docs/src/zh/reference/cli.md | 19 ++++++++++--------- 8 files changed, 43 insertions(+), 36 deletions(-) diff --git a/docs/src/en/design/isolation.md b/docs/src/en/design/isolation.md index b03c4c73..0dd82aa6 100644 --- a/docs/src/en/design/isolation.md +++ b/docs/src/en/design/isolation.md @@ -6,8 +6,8 @@ A copy-on-write workspace and a security boundary solve different problems. Over | Executor | Mechanisms | Limits to keep explicit | | --- | --- | --- | -| Linux host | Rootless launcher, user/mount/PID namespaces, projected root, negotiated Landlock, descriptor cleanup and capability dropping | Kernel and host configuration matter; selective proxy networking remains cooperative | -| macOS host | Generated Seatbelt profile with staged write controls; deny-all socket policy when requested | Reads and selective network access remain ambient/cooperative; staged mounts require macFUSE | +| Linux host | `--filesystem sandbox` enables the rootless launcher, user/mount/PID namespaces, projected root and negotiated Landlock; `--overlaynet-deny-all` independently enables a private network namespace | Kernel and host configuration matter; selective proxy networking remains cooperative; the default host filesystem view is unrestricted | +| macOS host | `--filesystem sandbox` enables Seatbelt filesystem controls; `--overlaynet-deny-all` independently enables the deny-all socket policy; `--stage` independently selects a staged workspace | Reads and selective network access remain ambient/cooperative unless the corresponding policy is requested; staged mounts require macFUSE | | Native OCI container | Linux OCI runtime, image userland and configured mounts/network | Does not claim complete enforcement of every capability dimension | | libkrun VM | Separate Linux guest kernel, virtio-fs workspace and smoltcp network path | Requires KVM or HVF; host connectors and shared files still form part of the boundary | @@ -15,10 +15,11 @@ Check the actual Run Bundle. Configuration expresses a request; installed contro ## Workspace and lifecycle -Enable an explicit stage for a reviewable workspace: +Filesystem access, network isolation, and change staging are separate settings. Host runs preserve the host filesystem view by default. Use `--filesystem sandbox` when path access must be restricted, and enable an explicit stage when changes should be reviewable: ```bash pvisor run --stage ../stage-001 -- codex +pvisor run --filesystem sandbox --overlaynet-deny-all -- codex ``` Without an OverlayFS option, the host command may write the project directly. The stage does not roll back remote API calls or writes outside its covered workspace. diff --git a/docs/src/en/guides/execution.md b/docs/src/en/guides/execution.md index b1a58bb5..a96b96ef 100644 --- a/docs/src/en/guides/execution.md +++ b/docs/src/en/guides/execution.md @@ -16,7 +16,7 @@ The command must exist in the selected environment. An Ubuntu image does not inc pvisor run --executor host --stage ../stage-host -- /bin/sh ``` -The working directory is the managed copy-on-write view. Without an OverlayFS option, a host command can write directly to the project. Safe-best-effort isolation reports unsupported controls; it is not a uniform guarantee across platforms. +Host execution preserves the host filesystem view by default. Use `--filesystem sandbox` for path access restrictions, `--stage PATH` for a reviewable copy-on-write workspace, and `--overlaynet-deny-all` for deny-all network policy. These settings are independent; a network option does not enable filesystem restrictions. ## Linux container diff --git a/docs/src/en/guides/troubleshooting.md b/docs/src/en/guides/troubleshooting.md index 175516db..45e41460 100644 --- a/docs/src/en/guides/troubleshooting.md +++ b/docs/src/en/guides/troubleshooting.md @@ -28,10 +28,10 @@ Check whether the command used a stage: pvisor run --stage ./runs/task-001 -- AGENT_COMMAND ``` -Without a stage, the host executor may still provide safe-best-effort controls, -but there is no staged filesystem Effect to review and selectively apply. If a -staged Run was expected, inspect the recorded stage path and the executor -warnings before rerunning it. +Without a stage, there is no staged filesystem Effect to review and selectively +apply. Host filesystem access remains direct unless `--filesystem sandbox` was +requested. If a staged Run was expected, inspect the recorded stage path and the +executor warnings before rerunning it. ## A requested capability was not enforced diff --git a/docs/src/en/reference/cli.md b/docs/src/en/reference/cli.md index fbb92d7f..daa4b58d 100644 --- a/docs/src/en/reference/cli.md +++ b/docs/src/en/reference/cli.md @@ -53,8 +53,9 @@ pvisor run --stage ../stage-001 -- codex pvisor review last ``` -Host execution uses safe-best-effort isolation by default. `--stage ` opts -into an OverlayFS stage for the current workspace, creates an independent Run +Host execution preserves the host filesystem view by default. `--filesystem sandbox` +opts into pVisor's synthetic-root/Landlock or Seatbelt filesystem access policy. +`--stage ` independently opts into an OverlayFS stage for the current workspace, creates an independent Run and writable stage at the supplied path, retains changes for manual review, and writes `run-bundle.json` with mode `0600`. @@ -64,16 +65,15 @@ VM executors all request Network and Subprocess enforcement, and none claim Subprocess — so `--strict` currently exits with `UnsupportedPolicy` on those paths. Use it to verify fail-closed behavior, not as a “stronger sandbox is ready” switch. -On Linux, the default host executor self-executes through pVisor's rootless -launcher before the async runtime reaches the Agent. User/mount/PID namespaces, -an in-namespace PID 1 descendant reaper, -minimal bind-projected root plus `chroot`, a kernel-negotiated Landlock ABI v1-v3 policy, closed -inherited descriptors, `no_new_privs`, and an empty capability set make -workspace containment non-bypassable for the Agent process tree. -`--overlaynet-deny-all` adds a private network namespace; the -public/allowlist proxy modes remain cooperative. On macOS the default safe -host executor installs a generated Seatbelt policy that makes staged writes -non-bypassable. For deny-all Runs it blocks IP and ambient host Unix sockets, +On Linux, `--filesystem sandbox` uses pVisor's rootless launcher with +user/mount/PID namespaces, a minimal bind-projected root, `chroot`, and a +kernel-negotiated Landlock policy. `--overlaynet-deny-all` independently adds a +private network namespace; the +public/allowlist proxy modes remain cooperative. On macOS the host executor +installs a generated Seatbelt policy only when filesystem sandboxing or network +isolation is requested; filesystem policy remains independent from network +policy. Staged writes are non-bypassable. For deny-all Runs it blocks IP and +ambient host Unix sockets, while retaining the exact AgentCtl and Run-local IPC. Reads and selective network policy remain ambient/cooperative and are labeled separately in the Bundle. Native OCI and libkrun executors retain the same outer Run, OverlayFS, @@ -249,6 +249,9 @@ source of truth. The equivalent TOML is: ```toml +# host (default) or sandbox; independent from OverlayNet and OverlayFS staging +filesystem = "host" + [run] agent = "codex" executor = "container" diff --git a/docs/src/zh/design/isolation.md b/docs/src/zh/design/isolation.md index 440b0c74..a7807a10 100644 --- a/docs/src/zh/design/isolation.md +++ b/docs/src/zh/design/isolation.md @@ -6,8 +6,8 @@ | Executor | 机制 | 必须明确的限制 | | --- | --- | --- | -| Linux host | rootless launcher、user/mount/PID namespace、投影根目录、协商后的 Landlock、描述符清理和 capability 清除 | 依赖内核及宿主配置;选择性代理网络仍是协作式 | -| macOS host | 生成的 Seatbelt 配置约束暂存写入;请求 deny-all 时安装 socket 策略 | 读取和选择性网络访问仍为 ambient/协作式;暂存挂载需要 macFUSE | +| Linux host | `--filesystem sandbox` 启用 rootless launcher、user/mount/PID namespace、投影根目录和协商后的 Landlock;`--overlaynet-deny-all` 独立启用私有网络 namespace | 依赖内核及宿主配置;选择性代理网络仍是协作式;默认 host 文件系统视图不受限制 | +| macOS host | `--filesystem sandbox` 启用 Seatbelt 文件系统控制;`--overlaynet-deny-all` 独立启用 deny-all socket 策略;`--stage` 独立选择暂存工作区 | 未请求对应策略时,读取和选择性网络访问仍为 ambient/协作式;暂存挂载需要 macFUSE | | 原生 OCI 容器 | Linux OCI runtime、镜像用户空间和配置的挂载及网络 | 不声明所有 capability 维度均已完整强制执行 | | libkrun VM | 独立 Linux 客户机内核、virtio-fs 工作区和 smoltcp 网络路径 | 需要 KVM 或 HVF;宿主连接器和共享文件仍是边界的一部分 | @@ -15,10 +15,11 @@ ## 工作区与生命周期 -显式启用暂存,才能得到可审查工作区: +文件系统访问、网络隔离和改动暂存是独立设置。Host 默认保留宿主文件系统视图。需要限制路径访问时使用 `--filesystem sandbox`,需要审查改动时显式启用暂存: ```bash pvisor run --stage ../stage-001 -- codex +pvisor run --filesystem sandbox --overlaynet-deny-all -- codex ``` 没有 OverlayFS 选项时,host 命令可能直接写入项目。暂存不能回滚远程 API 调用或覆盖工作区之外的写入。 diff --git a/docs/src/zh/guides/execution.md b/docs/src/zh/guides/execution.md index a129631c..285fc7b7 100644 --- a/docs/src/zh/guides/execution.md +++ b/docs/src/zh/guides/execution.md @@ -16,7 +16,7 @@ pvisor run --executor host --stage ../stage-host -- /bin/sh ``` -工作目录是受管理的写时复制视图。没有 OverlayFS 选项时,host 命令可以直接写入项目。safe-best-effort 隔离会报告不支持的控制,不代表各平台具有相同保证。 +Host 默认保留宿主文件系统视图。需要限制路径访问时使用 `--filesystem sandbox`,需要可审查的写时复制工作区时使用 `--stage PATH`,需要拒绝所有网络时使用 `--overlaynet-deny-all`。这些设置相互独立,网络参数不会启用文件系统限制。 ## Linux 容器 diff --git a/docs/src/zh/guides/troubleshooting.md b/docs/src/zh/guides/troubleshooting.md index 35a6982f..dbcea9fc 100644 --- a/docs/src/zh/guides/troubleshooting.md +++ b/docs/src/zh/guides/troubleshooting.md @@ -25,9 +25,10 @@ pvisor review last pvisor run --stage ./runs/task-001 -- AGENT_COMMAND ``` -不使用 stage 时,host executor 仍可能提供 safe-best-effort 控制,但不会产生可以审查和 -选择性 apply 的 staged filesystem Effect。如果本来就需要 stage,请检查记录的 stage 路径和 -executor warning,再重新执行。 +不使用 stage 时,host executor 不会产生可以审查和选择性 apply 的 staged filesystem Effect; +默认也保留宿主机文件系统视图。需要文件系统访问限制时使用 +`--filesystem sandbox`,需要审查改动时再使用 `--stage`。如果本来就需要 stage,请检查 +记录的 stage 路径和 executor warning,再重新执行。 ## 请求的 capability 没有被强制执行 diff --git a/docs/src/zh/reference/cli.md b/docs/src/zh/reference/cli.md index 46c79c1c..fdf623fb 100644 --- a/docs/src/zh/reference/cli.md +++ b/docs/src/zh/reference/cli.md @@ -44,7 +44,8 @@ pvisor run --stage ../stage-001 -- codex pvisor review last ``` -默认 host 执行使用 safe-best-effort 隔离;`--stage ` 才启用当前目录的 +默认 host 执行保留宿主机文件系统视图;`--filesystem sandbox` 才启用 pVisor 的 +synthetic-root/Landlock 或 Seatbelt 文件系统访问策略。`--stage ` 独立启用当前目录的 OverlayFS stage,在显式 `--stage` 路径创建独立 Run 和可写 stage,保留改动供人工审查,并以 `0600` 写入 `run-bundle.json`。 @@ -52,14 +53,11 @@ Run 和可写 stage,保留改动供人工审查,并以 `0600` 写入 `run-bu 否则在 Agent 启动前失败关闭。当前 host / container / VM 都会请求 Network 与 Subprocess,且无一 claim Subprocess,因此 `--strict` 在这些路径上会以 `UnsupportedPolicy` 退出。该旗标用于验证 fail-closed,不表示「更强沙箱已就绪」。 -在 Linux 上,默认 host executor 会在异步 runtime 到达 Agent 之前,通过 -pVisor 的 rootless launcher 自执行。User/mount/PID namespace、namespace 内 -PID 1 后代回收器、最小 bind-projected root 加 `chroot`、按内核协商的 Landlock ABI v1-v3 -策略、关闭继承描述符、`no_new_privs` 以及空 capability 集,使工作区约束对 -Agent 进程树不可绕过。 -`--overlaynet-deny-all` 再加一个私有 network namespace;public/allowlist -代理模式仍是协作式。在 macOS 上,默认 safe host executor 安装生成的 -Seatbelt 策略,使 staged 写入不可绕过。对 deny-all Run,它拦截 IP 和 +在 Linux 上,`--filesystem sandbox` 会使用 pVisor 的 rootless launcher,启用 +User/mount/PID namespace、最小 bind-projected root、`chroot` 和按内核协商的 +Landlock 策略。`--overlaynet-deny-all` 独立增加私有 network namespace;public/allowlist +代理模式仍是协作式。在 macOS 上,host executor 只在请求文件系统 sandbox 或网络隔离时安装生成的 +Seatbelt 策略;文件系统策略与网络策略相互独立。对 deny-all Run,它拦截 IP 和 ambient host Unix socket,同时保留精确的 AgentCtl 与 Run 本地 IPC。读取和 选择性网络策略仍是 ambient/协作式,并在 Bundle 中单独标注。原生 OCI 和 libkrun executor保留同样的外层 Run、OverlayFS 和 AgentCtl 状态观察。 @@ -217,6 +215,9 @@ pvisor run \ 等价 TOML 是: ```toml +# host(默认)或 sandbox;与 OverlayNet 和 OverlayFS 暂存相互独立 +filesystem = "host" + [run] agent = "codex" executor = "container" From e77d5cc477729bd2b791a03100ccaa854fdc4a73 Mon Sep 17 00:00:00 2001 From: Lucius <54578015+lizhicui@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:28:44 +0800 Subject: [PATCH 5/5] fix(ci): align isolation tests with independent policies --- .github/workflows/ci.yml | 7 ++-- crates/persisting-pvisor/src/cli/mod.rs | 9 ++--- .../persisting-pvisor/tests/rootless_local.rs | 33 +++++++++++++++++-- crates/persisting-replay/src/process.rs | 6 ++-- .../pvisor/01-filesystem-isolation/run.sh | 2 +- scripts/install-nightly.sh | 15 ++++++--- 6 files changed, 55 insertions(+), 17 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 05553ade..8da2dd0d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,9 +31,10 @@ jobs: - uses: Swatinem/rust-cache@v2 with: shared-key: ci-clippy - - uses: taiki-e/install-action@v2 - with: - tool: actionlint@1.7.7 + - name: Install actionlint + run: | + go install github.com/rhysd/actionlint/cmd/actionlint@v1.7.7 + echo "$(go env GOPATH)/bin" >> "$GITHUB_PATH" - run: actionlint - run: just fmt-rust --check - run: just lint-rust diff --git a/crates/persisting-pvisor/src/cli/mod.rs b/crates/persisting-pvisor/src/cli/mod.rs index fe34fcda..0f2666ac 100644 --- a/crates/persisting-pvisor/src/cli/mod.rs +++ b/crates/persisting-pvisor/src/cli/mod.rs @@ -268,10 +268,11 @@ mod tests { #[cfg(target_os = "linux")] { - assert!(help.contains("independent filesystem, network, and staging policies")); - assert!(help.contains("--filesystem sandbox")); - assert!(help.contains("namespace")); - assert!(help.contains("Landlock")); + let normalized = help.split_whitespace().collect::>().join(" "); + assert!(normalized.contains("host filesystem view by default")); + assert!(normalized.contains("--filesystem sandbox")); + assert!(normalized.contains("namespace")); + assert!(normalized.contains("Landlock")); } #[cfg(target_os = "macos")] { diff --git a/crates/persisting-pvisor/tests/rootless_local.rs b/crates/persisting-pvisor/tests/rootless_local.rs index 905870f9..524d01a4 100644 --- a/crates/persisting-pvisor/tests/rootless_local.rs +++ b/crates/persisting-pvisor/tests/rootless_local.rs @@ -159,6 +159,8 @@ printf '%s:%s:%s\n' "$PERSISTING_SANDBOX_FILESYSTEM" "$PERSISTING_SANDBOX_LANDLO "capture", "--stage", temporary.path().join("stage").to_str().unwrap(), + "--filesystem", + "sandbox", "--pass-env", "OUTSIDE_SECRET", "--pass-env", @@ -259,6 +261,8 @@ printf metadata-denied "capture", "--stage", temporary.path().join("stage").to_str().unwrap(), + "--filesystem", + "sandbox", "--pass-env", "OUTSIDE_FILE", "--pass-env", @@ -414,7 +418,14 @@ fn safe_apply_refuses_to_overwrite_a_concurrently_changed_target() { let output = Command::new(env!("CARGO_BIN_EXE_pvisor")) .env("PERSISTING_RUN_HOME", &run_home) - .args(["run", "--stdio", "capture", "--stage"]) + .args([ + "run", + "--stdio", + "capture", + "--filesystem", + "sandbox", + "--stage", + ]) .arg(temporary.path().join("stage")) .current_dir(&workspace) .args(["--", "/bin/sh", "-c", "printf staged > value.txt"]) @@ -488,6 +499,8 @@ fn safe_launcher_closes_inherited_host_file_descriptors() { "capture", "--stage", temporary.path().join("stage").to_str().unwrap(), + "--filesystem", + "sandbox", "--overlaynet-deny-all", "--pass-env", "PERSISTING_LEAKED_FD", @@ -643,7 +656,14 @@ fn synthetic_root_hides_ungranted_host_unix_sockets() { .env("PERSISTING_RUN_HOME", &run_home) .env("PERSISTING_SOCKET_PROBE", &host_socket) .env("SSH_AUTH_SOCK", &host_socket) - .args(["run", "--stdio", "capture", "--stage"]) + .args([ + "run", + "--stdio", + "capture", + "--filesystem", + "sandbox", + "--stage", + ]) .arg(temporary.path().join("stage")) .current_dir(&workspace) .arg("--") @@ -683,7 +703,14 @@ fn safe_run_reaps_setsid_double_fork_descendants_after_success() { let output = Command::new(env!("CARGO_BIN_EXE_pvisor")) .env("PERSISTING_RUN_HOME", &run_home) - .args(["run", "--stdio", "capture", "--stage"]) + .args([ + "run", + "--stdio", + "capture", + "--filesystem", + "sandbox", + "--stage", + ]) .arg(temporary.path().join("stage")) .current_dir(&workspace) .arg("--") diff --git a/crates/persisting-replay/src/process.rs b/crates/persisting-replay/src/process.rs index 3f92b24d..302dfaa5 100644 --- a/crates/persisting-replay/src/process.rs +++ b/crates/persisting-replay/src/process.rs @@ -628,11 +628,13 @@ mod tests { let log_path = temporary.path().join("redirect-idle.log"); let events_path = temporary.path().join("events.jsonl"); // stderr stays silent; only the redirect file grows. Without polling - // the redirect, a 300ms idle watchdog would kill this mid-loop. + // the redirect, a 700ms idle watchdog would kill this mid-loop; the + // 500ms slack over the 200ms write cadence absorbs scheduler jitter + // on loaded CI runners while still proving the refresh works. let script = "for i in 1 2 3 4 5 6; do echo event-$i; sleep 0.2; done"; let mut spec = shell_spec(script, &log_path); spec.stdout_redirect = Some(events_path.clone()); - spec.idle_timeout = Some(Duration::from_millis(300)); + spec.idle_timeout = Some(Duration::from_millis(700)); spec.timeout = Duration::from_secs(10); let output = run_process(spec).unwrap(); diff --git a/examples/pvisor/01-filesystem-isolation/run.sh b/examples/pvisor/01-filesystem-isolation/run.sh index 1de90259..d48e70b3 100755 --- a/examples/pvisor/01-filesystem-isolation/run.sh +++ b/examples/pvisor/01-filesystem-isolation/run.sh @@ -15,7 +15,7 @@ base="$work_dir/base" # The project workspace is reusable; pVisor creates an independent stage for this Run. ( cd "$base" - "$pvisor_bin" run --overlayfs-commit manual --stdio capture -- \ + "$pvisor_bin" run --filesystem sandbox --overlayfs-commit manual --stdio capture -- \ /bin/sh -c 'printf "changed\n" > existing.txt; printf "new\n" > new.txt' ) run_dir="$(find "$PERSISTING_RUN_HOME" -mindepth 1 -maxdepth 1 -type d -name 'run-*' -print -quit)" diff --git a/scripts/install-nightly.sh b/scripts/install-nightly.sh index a9ee30fb..eb4156fa 100755 --- a/scripts/install-nightly.sh +++ b/scripts/install-nightly.sh @@ -39,7 +39,13 @@ api="https://api.github.com/repos/${REPO}/releases/tags/${TAG}" echo "Fetching nightly release assets from ${REPO} (tag=${TAG})..." >&2 -url="$("$PYTHON" - "$api" "$platform_re" "$REPO" "$TAG" <<'PY' +# Write the asset picker to a helper file instead of embedding a here-document +# inside a command substitution: bash 5.2 mis-parses long heredocs whose body +# contains a line beginning with the delimiter (for example `PY3_RE`) in that +# position. +asset_helper="$(mktemp)" +trap 'rm -f "$asset_helper"' EXIT +cat > "$asset_helper" <<'PY' import json import re import sys @@ -51,7 +57,7 @@ platform_re = re.compile(platform_pattern) # Wheels contain native CLIs but only pure Python modules, so one py3-none wheel # is published per supported OS/architecture. -PY3_RE = re.compile(r"-py3-none-") +WHEEL_RE = re.compile(r"-py3-none-") req = urllib.request.Request(api, headers={"Accept": "application/vnd.github+json"}) try: @@ -69,7 +75,7 @@ for asset in data.get("assets", []): name = asset.get("name", "") if not name.endswith(".whl") or not name.startswith("pvisor-"): continue - if not PY3_RE.search(name): + if not WHEEL_RE.search(name): continue if platform_re.search(name): print(asset["browser_download_url"]) @@ -80,7 +86,8 @@ else: f"check https://github.com/{repo}/releases/tag/{tag}" ) PY -)" + +url="$("$PYTHON" - "$api" "$platform_re" "$REPO" "$TAG" < "$asset_helper")" echo "Installing ${url}" >&2 "$PYTHON" -m pip install --upgrade pip