diff --git a/crates/openshell-core/src/driver_mounts.rs b/crates/openshell-core/src/driver_mounts.rs index b1a3049882..25429e7a91 100644 --- a/crates/openshell-core/src/driver_mounts.rs +++ b/crates/openshell-core/src/driver_mounts.rs @@ -220,7 +220,8 @@ pub fn path_is_or_under(path: &Path, parent: &Path) -> bool { path == parent || path.starts_with(parent) } -fn paths_overlap(left: &Path, right: &Path) -> bool { +/// Return true when either path is the other or contains it. +pub fn paths_overlap(left: &Path, right: &Path) -> bool { path_is_or_under(left, right) || path_is_or_under(right, left) } diff --git a/crates/openshell-driver-vm/README.md b/crates/openshell-driver-vm/README.md index 334d968dee..bb1dc82a8d 100644 --- a/crates/openshell-driver-vm/README.md +++ b/crates/openshell-driver-vm/README.md @@ -174,6 +174,7 @@ Select the VM driver with `--compute-driver vm`, `OPENSHELL_COMPUTE_DRIVER=vm`, | `proxy_connect_by_hostname` | unset | Send hostnames rather than validated IPs in CONNECT. Last resort for proxies whose ACLs reject IP CONNECT targets. | | `proxy_ca_bundle` | unset | Gateway-host PEM CA bundle trusted for the corporate proxy and TLS-intercepted server certificates. The driver validates it and stages it at a fixed non-secret guest path in the protected overlay. Requires `https_proxy`. | | `provider_spiffe_workload_api_tcp_endpoint` | unset | Explicit guest-reachable `tcp:IP:port` SPIFFE Workload API listener for provider token exchange. It requires `provider_spiffe_allow_guest_tcp = true`; a host UNIX socket is never silently exposed to a VM guest. | +| `enable_bind_mounts` | `false` | Allow per-sandbox `driver_config.vm.mounts` host directory shares (libkrun virtiofs only). The driver writes a `tag\ttarget\tmode` manifest to `/.openshell/mounts.manifest` in the overlay, and guest init mounts each share after `/sandbox` ownership fixups. Rejected while resource admission is enabled. | The proxy settings are operator-owned and deployment-level: they are not accepted through `template.driver_config.vm`, and the driver passes them only to native host control. Every present-but-invalid value is fatal at gateway or sandbox startup rather than degrading to a direct dial. diff --git a/crates/openshell-driver-vm/runtime/kernel/openshell.kconfig b/crates/openshell-driver-vm/runtime/kernel/openshell.kconfig index d4493abd7b..701a7e6db9 100644 --- a/crates/openshell-driver-vm/runtime/kernel/openshell.kconfig +++ b/crates/openshell-driver-vm/runtime/kernel/openshell.kconfig @@ -15,6 +15,10 @@ CONFIG_VIRTIO_BLK=y CONFIG_EXT4_FS=y CONFIG_EXT4_USE_FOR_EXT2=y +# Host-directory sharing via virtiofs (used for VM driver bind mounts). +CONFIG_FUSE_FS=y +CONFIG_VIRTIO_FS=y + # Cgroups used for process supervision and resource limits. CONFIG_CGROUPS=y CONFIG_CGROUP_DEVICE=y diff --git a/crates/openshell-driver-vm/scripts/openshell-vm-sandbox-init.sh b/crates/openshell-driver-vm/scripts/openshell-vm-sandbox-init.sh index 7975f60cdb..5daf438fac 100644 --- a/crates/openshell-driver-vm/scripts/openshell-vm-sandbox-init.sh +++ b/crates/openshell-driver-vm/scripts/openshell-vm-sandbox-init.sh @@ -544,6 +544,73 @@ run_openshell_init_dropins() { done < <(LC_ALL=C sort -u "$manifest") } +# Create a virtiofs mount point under ROOT_PREFIX without following symlinks, +# so an image cannot redirect a validated target onto another guest path. +# Directories created under /sandbox are handed to the sandbox user. +prepare_virtiofs_target() { + local target="$1" owner="$2" + local path="${ROOT_PREFIX:-}" component + local -a components + IFS=/ read -r -a components <<<"${target#/}" + for component in "${components[@]}"; do + [ -n "$component" ] || continue + path="${path}/${component}" + if [ -L "$path" ]; then + ts >&2 "FATAL: virtiofs mount target ${target} traverses symlink ${path#"${ROOT_PREFIX:-}"}" + exit 1 + fi + if [ ! -e "$path" ]; then + if ! mkdir "$path"; then + ts >&2 "FATAL: failed to create virtiofs mount point ${target}" + exit 1 + fi + case "${path#"${ROOT_PREFIX:-}"}" in + /sandbox/*) + if ! chown "$owner" "$path"; then + ts >&2 "FATAL: failed to hand virtiofs mount point parent ${path#"${ROOT_PREFIX:-}"} to ${owner}" + exit 1 + fi + ;; + esac + elif [ ! -d "$path" ]; then + ts >&2 "FATAL: virtiofs mount target ${target} is not a directory" + exit 1 + fi + done + printf '%s\n' "$path" +} + +# Runs after every /sandbox ownership fixup and the root-run init drop-ins so +# nothing in init walks into host-backed shares. +mount_virtiofs_shares() { + local manifest + manifest="$(root_path /.openshell/mounts.manifest)" + [ -f "$manifest" ] || return 0 + + ts "mounting virtiofs shares" + local tag target mode mount_opts guest_target owner + owner="$(sandbox_owner)" + while IFS=$'\t' read -r tag target mode; do + [ -n "$tag" ] || continue + case "$mode" in + ro) mount_opts="-o ro" ;; + rw) mount_opts="" ;; + *) + ts "FATAL: unknown virtiofs mount mode '${mode}' for tag ${tag}" + exit 1 + ;; + esac + guest_target="$(prepare_virtiofs_target "$target" "$owner")" || exit 1 + # shellcheck disable=SC2086 + if mount -t virtiofs $mount_opts "$tag" "$guest_target"; then + ts " mounted virtiofs ${tag} -> ${target} (${mode})" + else + ts "FATAL: failed to mount virtiofs ${tag} at ${target}" + exit 1 + fi + done < "$manifest" +} + run_post_overlay_setup() { # Source QEMU-injected environment variables if present. The file lives in # the overlay upperdir so the cached bootstrap rootfs remains immutable. @@ -617,6 +684,8 @@ fi run_openshell_init_dropins +mount_virtiofs_shares + if [ -n "${OPENSHELL_SANDBOX_ID:-}" ]; then ts "OPENSHELL_SANDBOX_ID=${OPENSHELL_SANDBOX_ID}" fi diff --git a/crates/openshell-driver-vm/src/driver.rs b/crates/openshell-driver-vm/src/driver.rs index 799cb0b6e8..c9decdda74 100644 --- a/crates/openshell-driver-vm/src/driver.rs +++ b/crates/openshell-driver-vm/src/driver.rs @@ -40,6 +40,7 @@ use oci_client::manifest::{ use oci_client::secrets::RegistryAuth; use oci_client::{Reference, RegistryOperation}; use openshell_core::UpstreamProxyConfig; +use openshell_core::driver_mounts; use openshell_core::gpu::{ driver_gpu_requirements, effective_driver_gpu_count, validate_specific_gpu_device_request, }; @@ -116,6 +117,30 @@ const DEFAULT_ROOTFS_TAR_MAX_BYTES: u64 = 10 * 1024 * 1024 * 1024; const ROOTFS_TAR_STAGING_DIR: &str = "rootfs-tar-staging"; const VM_CONSOLE_DIAGNOSTIC_BYTES: u64 = 8 * 1024; +/// Mirrors the Docker/Podman `type` tag so their bind entries parse +/// unchanged; virtiofs can only share host directories. +#[derive(Debug, Clone, Copy, serde::Deserialize)] +#[serde(rename_all = "lowercase")] +enum VmMountType { + Bind, +} + +#[derive(Debug, Clone, serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct VmMountConfig { + #[serde(default, rename = "type")] + #[allow(dead_code)] + mount_type: Option, + source: String, + target: String, + #[serde(default = "default_read_only")] + read_only: bool, +} + +fn default_read_only() -> bool { + true +} + #[derive(Debug, Clone, Default, serde::Deserialize)] #[serde(default, deny_unknown_fields)] struct VmSandboxDriverConfig { @@ -125,6 +150,8 @@ struct VmSandboxDriverConfig { )] gpu_device_ids: Option>, rootfs_tar_path: Option, + #[serde(default)] + mounts: Vec, } impl VmSandboxDriverConfig { @@ -246,6 +273,7 @@ enum GuestImagePayloadSource { #[derive(Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] +#[allow(clippy::struct_excessive_bools)] pub struct VmDriverConfig { /// Permit caller-supplied driver JSON. Does not waive resource admission. #[serde(default)] @@ -297,6 +325,9 @@ pub struct VmDriverConfig { /// Maximum rootfs tar file size in bytes. Defaults to 10 GiB. #[serde(default, skip_serializing_if = "Option::is_none")] pub rootfs_tar_max_bytes: Option, + + #[serde(default)] + pub enable_bind_mounts: bool, } /// Redacting `Debug` so a proxy URL or credential path never reaches a log. @@ -358,6 +389,7 @@ impl std::fmt::Debug for VmDriverConfig { ) .field("rootfs_tar_staging_dir", &self.rootfs_tar_staging_dir) .field("rootfs_tar_max_bytes", &self.rootfs_tar_max_bytes) + .field("enable_bind_mounts", &self.enable_bind_mounts) .finish() } } @@ -396,6 +428,7 @@ impl Default for VmDriverConfig { sandbox_gid: None, rootfs_tar_staging_dir: None, rootfs_tar_max_bytes: None, + enable_bind_mounts: false, } } } @@ -1086,8 +1119,14 @@ impl VmDriver { sandbox, )?; validate_vm_sandbox(sandbox, self.config.gpu_enabled)?; - let has_rootfs_tar = - VmSandboxDriverConfig::from_sandbox(sandbox).is_ok_and(|c| c.rootfs_tar_path.is_some()); + let driver_config = + VmSandboxDriverConfig::from_sandbox(sandbox).map_err(Status::invalid_argument)?; + validate_vm_driver_mounts( + &driver_config.mounts, + &self.config, + sandbox_requests_gpu(sandbox), + )?; + let has_rootfs_tar = driver_config.rootfs_tar_path.is_some(); if self.resolved_sandbox_image(sandbox).is_none() && !has_rootfs_tar { return Err(Status::failed_precondition( "vm sandboxes require template.image, rootfs_tar_path in driver_config, or a configured default sandbox image", @@ -1433,6 +1472,7 @@ impl VmDriver { }; let needs_qemu = is_gpu; + validate_vm_mount_sources(&driver_config.mounts)?; let mut plan = match self.build_vm_launch_plan(&sandbox.id, needs_qemu, is_gpu, gpu_bdf.clone()) { @@ -1605,6 +1645,10 @@ impl VmDriver { &channel_tls, ) .map_err(|error| Status::internal(format!("inject VM boundary configuration: {error}")))?; + // Always written, even when empty, so an image-supplied manifest in + // the lowerdir can never drive guest mounts. + inject_guest_mount_manifest(&overlay_disk, &driver_config.mounts) + .map_err(|error| Status::internal(format!("inject VM mount manifest: {error}")))?; write_private_file( &state_dir.join(HOST_BOUNDARY_GENERATION_FILE), boundary_generation.as_bytes().to_vec(), @@ -1667,6 +1711,14 @@ impl VmDriver { for env in sandbox_owner_state.guest_environment() { command.arg("--vm-env").arg(env); } + for (i, mount) in driver_config.mounts.iter().enumerate() { + command.arg("--vm-mount").arg(format!( + "{}\t{}\t{}", + mount.source, + virtiofs_tag(i), + vm_mount_mode(mount.read_only) + )); + } info!( sandbox_id = %sandbox.id, @@ -4956,6 +5008,146 @@ fn validate_vm_sandbox_template(template: &SandboxTemplate) -> Result<(), Status Ok(()) } +const VM_RESERVED_GUEST_PATHS: &[&str] = &[ + "/.openshell", + "/.openshell-bootstrap", + "/overlay", + "/lower", + "/newroot", + "/image-cache", + "/srv", + "/proc", + "/sys", + "/dev", + "/tmp", +]; + +const VIRTIOFS_TAG_PREFIX: &str = "osfs"; + +fn virtiofs_tag(index: usize) -> String { + format!("{VIRTIOFS_TAG_PREFIX}{index}") +} + +fn vm_mount_mode(read_only: bool) -> &'static str { + if read_only { "ro" } else { "rw" } +} + +/// Guest manifest consumed by `mount_virtiofs_shares`: `tag\ttarget\tmode`. +fn render_mount_manifest(mounts: &[VmMountConfig]) -> String { + let mut manifest = String::new(); + for (i, mount) in mounts.iter().enumerate() { + writeln!( + manifest, + "{}\t{}\t{}", + virtiofs_tag(i), + driver_mounts::normalize_mount_target(&mount.target), + vm_mount_mode(mount.read_only) + ) + .expect("write to String cannot fail"); + } + manifest +} + +fn sandbox_requests_gpu(sandbox: &Sandbox) -> bool { + sandbox + .spec + .as_ref() + .and_then(|spec| spec.resource_requirements.as_ref()) + .and_then(|requirements| driver_gpu_requirements(Some(requirements))) + .is_some() +} + +fn validate_no_control_chars(value: &str, field: &str) -> Result<(), String> { + if value.chars().any(char::is_control) { + return Err(format!("{field} must not contain control characters")); + } + Ok(()) +} + +#[allow(clippy::result_large_err)] +fn validate_vm_driver_mounts( + mounts: &[VmMountConfig], + config: &VmDriverConfig, + is_qemu: bool, +) -> Result<(), Status> { + if mounts.is_empty() { + return Ok(()); + } + if is_qemu { + return Err(Status::failed_precondition( + "virtiofs mounts are not supported with the QEMU backend", + )); + } + if !config.enable_bind_mounts { + return Err(Status::failed_precondition( + "vm bind mounts require enable_bind_mounts = true in the VM driver configuration", + )); + } + config + .resource_admission + .reject_unlabelable("host bind mount")?; + let mut targets: Vec = Vec::with_capacity(mounts.len()); + for mount in mounts { + driver_mounts::validate_absolute_mount_source(&mount.source, "vm mount source") + .map_err(Status::invalid_argument)?; + validate_no_control_chars(&mount.source, "vm mount source") + .map_err(Status::invalid_argument)?; + driver_mounts::validate_container_mount_target(&mount.target) + .map_err(Status::invalid_argument)?; + let normalized = driver_mounts::normalize_mount_target(&mount.target); + driver_mounts::validate_workspace_mount_target( + &normalized, + driver_mounts::DEFAULT_WORKSPACE_ROOT, + ) + .map_err(Status::invalid_argument)?; + for reserved in VM_RESERVED_GUEST_PATHS { + if driver_mounts::paths_overlap(Path::new(&normalized), Path::new(reserved)) { + return Err(Status::invalid_argument(format!( + "vm mount target '{}' conflicts with VM-internal path '{reserved}'", + mount.target + ))); + } + } + // Nested shares would make guest init create directories inside an + // already-mounted host share, so targets must be disjoint. + if let Some(existing) = targets.iter().find(|existing| { + driver_mounts::paths_overlap(Path::new(&normalized), Path::new(existing)) + }) { + return Err(Status::invalid_argument(if *existing == normalized { + format!("duplicate vm driver_config mount target '{normalized}'") + } else { + format!("vm mount target '{normalized}' overlaps mount target '{existing}'") + })); + } + targets.push(normalized); + } + Ok(()) +} + +/// Host filesystem checks, run at provisioning rather than in +/// `validate_sandbox` so a briefly missing source (for example an NFS share +/// not yet mounted after a reboot) fails provisioning visibly instead of +/// denying recovery of a persisted sandbox. +#[allow(clippy::result_large_err)] +fn validate_vm_mount_sources(mounts: &[VmMountConfig]) -> Result<(), Status> { + for mount in mounts { + let source_path = Path::new(&mount.source); + if !source_path.exists() { + return Err(Status::failed_precondition(format!( + "vm mount source path does not exist: {}", + mount.source + ))); + } + if !source_path.is_dir() { + return Err(Status::invalid_argument(format!( + "vm mount source must be a directory, not a file: {}", + mount.source + ))); + } + } + Ok(()) +} + #[allow(clippy::result_large_err)] fn validate_gpu_request(sandbox: &Sandbox, gpu_enabled: bool) -> Result<(), Status> { let spec = sandbox @@ -6809,6 +7001,17 @@ fn inject_guest_boundary_bundle( Ok(()) } +fn inject_guest_mount_manifest( + overlay_disk: &Path, + mounts: &[VmMountConfig], +) -> Result<(), String> { + let manifest = render_mount_manifest(mounts); + let guest_path = overlay_upper_path("/.openshell/mounts.manifest"); + write_rootfs_image_file(overlay_disk, &guest_path, manifest.as_bytes())?; + set_rootfs_image_file_mode(overlay_disk, &guest_path, 0o644)?; + Ok(()) +} + fn guest_boundary_config_path(generation: &str) -> String { format!("{GUEST_BOUNDARY_CONFIG_DIR}/bootstrap-{generation}.json") } @@ -12840,4 +13043,293 @@ mod tests { assert!(alpha.join(SANDBOX_STOPPED_FILE).exists()); } + + // ── VM mount config tests ────────────────────────────────────────── + + fn mount_test_config(enable_bind_mounts: bool) -> VmDriverConfig { + let mut config = VmDriverConfig { + enable_bind_mounts, + ..VmDriverConfig::default() + }; + config.resource_admission.enabled = false; + config + } + + #[test] + fn vm_mount_validation_rejects_when_resource_admission_enabled() { + let dir = std::env::temp_dir(); + let mounts = vec![VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + let mut config = mount_test_config(true); + config.resource_admission.enabled = true; + let err = validate_vm_driver_mounts(&mounts, &config, false).unwrap_err(); + assert_eq!(err.code(), Code::FailedPrecondition); + assert!(err.message().contains("resource admission")); + } + + #[test] + fn vm_mount_config_deserializes_with_default_readonly() { + let json = serde_json::json!({ + "mounts": [{"source": "/host/data", "target": "/sandbox/data"}] + }); + let config: VmSandboxDriverConfig = serde_json::from_value(json).unwrap(); + assert_eq!(config.mounts.len(), 1); + assert_eq!(config.mounts[0].source, "/host/data"); + assert_eq!(config.mounts[0].target, "/sandbox/data"); + assert!(config.mounts[0].read_only); + } + + #[test] + fn vm_mount_config_accepts_docker_style_bind_type() { + let json = serde_json::json!({ + "mounts": [{"type": "bind", "source": "/host/data", "target": "/sandbox/data"}] + }); + let config: VmSandboxDriverConfig = serde_json::from_value(json).unwrap(); + assert!(matches!( + config.mounts[0].mount_type, + Some(VmMountType::Bind) + )); + assert!(config.mounts[0].read_only); + } + + #[test] + fn vm_mount_config_rejects_non_bind_types() { + for mount_type in ["volume", "tmpfs", "image"] { + let json = serde_json::json!({ + "mounts": [{"type": mount_type, "source": "x", "target": "/sandbox/data"}] + }); + let err = serde_json::from_value::(json).unwrap_err(); + assert!( + err.to_string().contains("expected `bind`"), + "{mount_type}: {err}" + ); + } + } + + #[test] + fn vm_mount_config_deserializes_explicit_readwrite() { + let json = serde_json::json!({ + "mounts": [{"source": "/host/data", "target": "/sandbox/data", "read_only": false}] + }); + let config: VmSandboxDriverConfig = serde_json::from_value(json).unwrap(); + assert!(!config.mounts[0].read_only); + } + + #[test] + fn vm_mount_validation_requires_enable_bind_mounts() { + let dir = std::env::temp_dir(); + let mounts = vec![VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + let err = validate_vm_driver_mounts(&mounts, &mount_test_config(false), false).unwrap_err(); + assert_eq!(err.code(), Code::FailedPrecondition); + assert!(err.message().contains("enable_bind_mounts")); + } + + #[test] + fn vm_mount_validation_allows_when_enabled() { + let dir = std::env::temp_dir(); + let mounts = vec![VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + assert!(validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).is_ok()); + } + + #[test] + fn vm_mount_validation_rejects_relative_source() { + let mounts = vec![VmMountConfig { + mount_type: None, + source: "relative/path".to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + let err = validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).unwrap_err(); + assert_eq!(err.code(), Code::InvalidArgument); + } + + #[test] + fn vm_mount_validation_rejects_reserved_openshell_target() { + let dir = std::env::temp_dir(); + let mounts = vec![VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/opt/openshell/data".to_string(), + read_only: true, + }]; + let err = validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).unwrap_err(); + assert_eq!(err.code(), Code::InvalidArgument); + } + + #[test] + fn vm_mount_validation_rejects_vm_internal_paths() { + let dir = std::env::temp_dir(); + for reserved in VM_RESERVED_GUEST_PATHS { + let mounts = vec![VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: reserved.to_string(), + read_only: true, + }]; + let err = + validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).unwrap_err(); + assert_eq!( + err.code(), + Code::InvalidArgument, + "expected rejection for target {reserved}" + ); + assert!( + err.message().contains("VM-internal path"), + "expected VM-internal path message for {reserved}, got: {}", + err.message() + ); + } + } + + #[test] + fn vm_mount_validation_rejects_duplicate_targets() { + let dir = std::env::temp_dir(); + let mounts = vec![ + VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }, + VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/sandbox/data".to_string(), + read_only: false, + }, + ]; + let err = validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).unwrap_err(); + assert_eq!(err.code(), Code::InvalidArgument); + assert!(err.message().contains("duplicate")); + } + + #[test] + fn vm_mount_validation_defers_missing_source_to_provisioning() { + let mounts = vec![VmMountConfig { + mount_type: None, + source: "/nonexistent/openshell-vm-mount-source".to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + assert!(validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).is_ok()); + let err = validate_vm_mount_sources(&mounts).unwrap_err(); + assert_eq!(err.code(), Code::FailedPrecondition); + } + + #[test] + fn vm_mount_validation_rejects_nested_targets() { + let dir = std::env::temp_dir().display().to_string(); + for (first, second) in [ + ("/sandbox/a", "/sandbox/a/b"), + ("/sandbox/a/b", "/sandbox/a"), + ] { + let mounts = vec![ + VmMountConfig { + mount_type: None, + source: dir.clone(), + target: first.to_string(), + read_only: true, + }, + VmMountConfig { + mount_type: None, + source: dir.clone(), + target: second.to_string(), + read_only: true, + }, + ]; + let err = + validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).unwrap_err(); + assert_eq!(err.code(), Code::InvalidArgument); + assert!(err.message().contains("overlaps"), "{first} vs {second}"); + } + } + + #[test] + fn vm_mount_validation_allows_sibling_targets() { + let dir = std::env::temp_dir().display().to_string(); + let mounts = vec![ + VmMountConfig { + mount_type: None, + source: dir.clone(), + target: "/sandbox/a".to_string(), + read_only: true, + }, + VmMountConfig { + mount_type: None, + source: dir, + target: "/sandbox/ab".to_string(), + read_only: true, + }, + ]; + assert!(validate_vm_driver_mounts(&mounts, &mount_test_config(true), false).is_ok()); + } + + #[test] + fn vm_mount_validation_empty_passes() { + assert!(validate_vm_driver_mounts(&[], &mount_test_config(false), false).is_ok()); + } + + #[test] + fn vm_mount_validation_rejects_qemu_backend() { + let dir = std::env::temp_dir(); + let mounts = vec![VmMountConfig { + mount_type: None, + source: dir.display().to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + let err = validate_vm_driver_mounts(&mounts, &mount_test_config(true), true).unwrap_err(); + assert_eq!(err.code(), Code::FailedPrecondition); + assert!(err.message().contains("QEMU")); + } + + #[test] + fn vm_mount_validation_rejects_file_source() { + let file = tempfile::NamedTempFile::new().unwrap(); + let mounts = vec![VmMountConfig { + mount_type: None, + source: file.path().display().to_string(), + target: "/sandbox/data".to_string(), + read_only: true, + }]; + let err = validate_vm_mount_sources(&mounts).unwrap_err(); + assert_eq!(err.code(), Code::InvalidArgument); + assert!(err.message().contains("must be a directory")); + } + + #[test] + fn vm_mount_manifest_renders_tab_separated_entries() { + let mounts = [ + VmMountConfig { + mount_type: None, + source: "/host/a".to_string(), + target: "/sandbox/a".to_string(), + read_only: true, + }, + VmMountConfig { + mount_type: None, + source: "/host/b".to_string(), + target: "/sandbox/b".to_string(), + read_only: false, + }, + ]; + assert_eq!( + render_mount_manifest(&mounts), + "osfs0\t/sandbox/a\tro\nosfs1\t/sandbox/b\trw\n" + ); + } } diff --git a/crates/openshell-driver-vm/src/ffi.rs b/crates/openshell-driver-vm/src/ffi.rs index f84ea35743..db1deb619c 100644 --- a/crates/openshell-driver-vm/src/ffi.rs +++ b/crates/openshell-driver-vm/src/ffi.rs @@ -54,6 +54,14 @@ type KrunDisableImplicitVsock = unsafe extern "C" fn(ctx_id: u32) -> i32; type KrunAddVsock = unsafe extern "C" fn(ctx_id: u32, tsi_features: u32) -> i32; type KrunAddVsockPort2 = unsafe extern "C" fn(ctx_id: u32, port: u32, filepath: *const c_char, listen: bool) -> i32; +type KrunAddVirtiofs4 = unsafe extern "C" fn( + ctx_id: u32, + tag: *const c_char, + path: *const c_char, + shm_size: u64, + read_only: bool, + semantics: u32, +) -> i32; // Field names mirror the libkrun C API symbol names (`krun_*`); preserving // the prefix keeps the FFI binding 1:1 with the upstream library. @@ -72,6 +80,9 @@ pub struct LibKrun { pub krun_disable_implicit_vsock: KrunDisableImplicitVsock, pub krun_add_vsock: KrunAddVsock, pub krun_add_vsock_port2: KrunAddVsockPort2, + /// Load error kept so runtimes without virtiofs still boot sandboxes + /// that request no bind mounts, and mount requests report why. + pub krun_add_virtiofs4: Result, } static LIBKRUN: OnceLock = OnceLock::new(); @@ -134,6 +145,7 @@ impl LibKrun { )?, krun_add_vsock: load_symbol(library, b"krun_add_vsock\0", &libkrun_path)?, krun_add_vsock_port2: load_symbol(library, b"krun_add_vsock_port2\0", &libkrun_path)?, + krun_add_virtiofs4: load_symbol(library, b"krun_add_virtiofs4\0", &libkrun_path), }) } } diff --git a/crates/openshell-driver-vm/src/lib.rs b/crates/openshell-driver-vm/src/lib.rs index 5464dd6449..658bf53b23 100644 --- a/crates/openshell-driver-vm/src/lib.rs +++ b/crates/openshell-driver-vm/src/lib.rs @@ -44,5 +44,6 @@ pub use lifecycle::{ }; #[cfg(feature = "compute-driver")] pub use runtime::{ - VM_RUNTIME_DIR_ENV, VmBackend, VmLaunchConfig, VsockPortMap, configured_runtime_dir, run_vm, + VM_RUNTIME_DIR_ENV, VmBackend, VmLaunchConfig, VmMount, VsockPortMap, configured_runtime_dir, + run_vm, }; diff --git a/crates/openshell-driver-vm/src/main.rs b/crates/openshell-driver-vm/src/main.rs index 4b4bcf3f6a..81e00e3d7c 100644 --- a/crates/openshell-driver-vm/src/main.rs +++ b/crates/openshell-driver-vm/src/main.rs @@ -9,7 +9,7 @@ use openshell_core::proto::compute::v1::compute_driver_server::ComputeDriverServ #[cfg(target_os = "macos")] use openshell_driver_vm::{VM_RUNTIME_DIR_ENV, configured_runtime_dir}; use openshell_driver_vm::{ - VmBackend, VmDriver, VmDriverConfig, VmLaunchConfig, VsockPortMap, procguard, run_vm, + VmBackend, VmDriver, VmDriverConfig, VmLaunchConfig, VmMount, VsockPortMap, procguard, run_vm, }; use std::io; use std::net::SocketAddr; @@ -227,6 +227,9 @@ struct Args { #[arg(long, env = "OPENSHELL_VM_PROXY_CA_BUNDLE")] proxy_ca_bundle: Option, + #[arg(long, env = "OPENSHELL_VM_ENABLE_BIND_MOUNTS", default_value_t = false)] + enable_bind_mounts: bool, + #[arg(long, env = "OPENSHELL_VM_ROOTFS_TAR_STAGING_DIR")] rootfs_tar_staging_dir: Option, @@ -247,6 +250,9 @@ struct Args { #[arg(long, hide = true)] vm_vsock_control_socket: Option, + + #[arg(long, hide = true)] + vm_mount: Vec, } #[tokio::main] @@ -328,6 +334,7 @@ async fn main() -> Result<()> { sandbox_gid: args.sandbox_gid, rootfs_tar_staging_dir: args.rootfs_tar_staging_dir.clone(), rootfs_tar_max_bytes: args.rootfs_tar_max_bytes, + enable_bind_mounts: args.enable_bind_mounts, }) .await .map_err(|err| miette::miette!("{err}"))?; @@ -617,6 +624,12 @@ fn build_vm_launch_config(args: &Args) -> std::result::Result return Err(format!("unknown VM backend: {other}")), }; + let mounts = args + .vm_mount + .iter() + .map(|m| parse_vm_mount_arg(m)) + .collect::, _>>()?; + Ok(VmLaunchConfig { root_disk, overlay_disk, @@ -650,6 +663,34 @@ fn build_vm_launch_config(args: &Args) -> std::result::Result std::result::Result { + let mut parts = arg.splitn(3, '\t'); + let source = parts + .next() + .ok_or_else(|| format!("invalid --vm-mount format: {arg}"))?; + let tag = parts + .next() + .ok_or_else(|| format!("invalid --vm-mount format (missing tag): {arg}"))?; + let mode = parts + .next() + .ok_or_else(|| format!("invalid --vm-mount format (missing mode): {arg}"))?; + let read_only = match mode { + "ro" => true, + "rw" => false, + _ => { + return Err(format!( + "invalid --vm-mount mode '{mode}': expected 'ro' or 'rw'" + )); + } + }; + Ok(VmMount { + host_path: PathBuf::from(source), + tag: tag.to_string(), + read_only, }) } @@ -709,7 +750,7 @@ fn maybe_reexec_internal_vm_with_runtime_env() -> Result<()> { mod tests { use super::{ Args, ComputeDriverListenMode, PeerCredentials, authorize_peer_credentials, - compute_driver_listen_mode, + compute_driver_listen_mode, parse_vm_mount_arg, }; use clap::Parser; use std::path::PathBuf; @@ -949,4 +990,30 @@ mod tests { } ); } + + #[test] + fn parse_vm_mount_arg_parses_readonly() { + let m = parse_vm_mount_arg("/host/src\tosfs0\tro").unwrap(); + assert_eq!(m.host_path, PathBuf::from("/host/src")); + assert_eq!(m.tag, "osfs0"); + assert!(m.read_only); + } + + #[test] + fn parse_vm_mount_arg_parses_readwrite() { + let m = parse_vm_mount_arg("/host/src\tosfs1\trw").unwrap(); + assert!(!m.read_only); + } + + #[test] + fn parse_vm_mount_arg_rejects_unknown_mode() { + let err = parse_vm_mount_arg("/host/src\tosfs0\treadonly").unwrap_err(); + assert!(err.contains("expected 'ro' or 'rw'")); + } + + #[test] + fn parse_vm_mount_arg_rejects_missing_fields() { + assert!(parse_vm_mount_arg("/host/src\tosfs0").is_err()); + assert!(parse_vm_mount_arg("/host/src").is_err()); + } } diff --git a/crates/openshell-driver-vm/src/runtime.rs b/crates/openshell-driver-vm/src/runtime.rs index 4d0680bffe..247c2b6907 100644 --- a/crates/openshell-driver-vm/src/runtime.rs +++ b/crates/openshell-driver-vm/src/runtime.rs @@ -32,6 +32,14 @@ pub struct VsockPortMap { pub host_initiated: bool, } +/// A host directory shared into the VM guest via virtiofs. +#[derive(Debug, Clone)] +pub struct VmMount { + pub tag: String, + pub host_path: PathBuf, + pub read_only: bool, +} + pub struct VmLaunchConfig { pub root_disk: PathBuf, pub overlay_disk: PathBuf, @@ -49,6 +57,7 @@ pub struct VmLaunchConfig { pub gpu_bdf: Option, pub vsock_cid: Option, pub vsock_port_map: Option, + pub mounts: Vec, } pub fn run_vm(config: &VmLaunchConfig) -> Result<(), String> { @@ -59,6 +68,9 @@ pub fn run_vm(config: &VmLaunchConfig) -> Result<(), String> { } fn run_qemu_vm(config: &VmLaunchConfig) -> Result<(), String> { + if !config.mounts.is_empty() { + return Err("virtiofs mounts are not yet supported with the QEMU backend".to_string()); + } let gpu_bdf = config .gpu_bdf .as_deref() @@ -336,6 +348,10 @@ fn run_libkrun_vm(config: &VmLaunchConfig) -> Result<(), String> { vm.add_vsock_port(port_map)?; } + for mount in &config.mounts { + vm.add_virtiofs(&mount.tag, &mount.host_path, mount.read_only)?; + } + vm.set_console_output(&config.console_output)?; let env = libkrun_guest_env(config); @@ -661,6 +677,28 @@ impl VmContext { ) } + fn add_virtiofs(&self, tag: &str, host_path: &Path, read_only: bool) -> Result<(), String> { + const SEMANTICS_SIMPLIFIED: u32 = 1; + let tag_c = CString::new(tag).map_err(|e| format!("invalid virtiofs tag: {e}"))?; + let path_c = path_to_cstring(host_path)?; + let add_virtiofs4 = *self.krun.krun_add_virtiofs4.as_ref().map_err(|err| { + format!("VM bind mounts need a libkrun runtime with virtiofs support: {err}") + })?; + check( + unsafe { + add_virtiofs4( + self.ctx_id, + tag_c.as_ptr(), + path_c.as_ptr(), + 0, + read_only, + SEMANTICS_SIMPLIFIED, + ) + }, + "krun_add_virtiofs4", + ) + } + fn start_enter(&self) -> i32 { unsafe { (self.krun.krun_start_enter)(self.ctx_id) } } @@ -777,6 +815,7 @@ mod tests { gpu_bdf: Some("0000:01:00.0".to_string()), vsock_cid: Some(4), vsock_port_map: None, + mounts: Vec::new(), } } diff --git a/crates/openshell-gateway/src/vm.rs b/crates/openshell-gateway/src/vm.rs index d0c00cd786..c5f4ca7f8b 100644 --- a/crates/openshell-gateway/src/vm.rs +++ b/crates/openshell-gateway/src/vm.rs @@ -122,6 +122,10 @@ pub struct VmComputeConfig { pub provider_spiffe_workload_api_tcp_endpoint: Option, #[serde(default)] pub provider_spiffe_allow_guest_tcp: bool, + + /// Allow bind-mounting host directories into VM sandboxes. + #[serde(default)] + pub enable_bind_mounts: bool, } impl VmComputeConfig { @@ -254,6 +258,7 @@ impl Default for VmComputeConfig { proxy_ca_bundle: None, provider_spiffe_workload_api_tcp_endpoint: None, provider_spiffe_allow_guest_tcp: false, + enable_bind_mounts: false, } } } @@ -575,6 +580,9 @@ pub async fn spawn( command.arg("--guest-tls-ca").arg(tls.ca); } append_vm_proxy_and_spiffe_args(&mut command, vm_config); + if vm_config.enable_bind_mounts { + command.arg("--enable-bind-mounts"); + } let mut child = command.spawn().map_err(|e| { Error::execution(format!( diff --git a/docs/how-it-works/gateways/configuration.mdx b/docs/how-it-works/gateways/configuration.mdx index 4d247f476b..154e7e4923 100644 --- a/docs/how-it-works/gateways/configuration.mdx +++ b/docs/how-it-works/gateways/configuration.mdx @@ -1212,6 +1212,11 @@ overlay_disk_mib = 4096 # the exposure; host-only sockets are never exposed automatically. # provider_spiffe_workload_api_tcp_endpoint = "tcp:192.0.2.10:8081" # provider_spiffe_allow_guest_tcp = true +# Unsafe operator override. Host bind mounts expose gateway-host paths inside +# VM sandboxes and can negate OpenShell isolation and filesystem controls. +# Requires allow_driver_config = true and +# [openshell.drivers.vm.resource_admission] enabled = false; libkrun only. +# enable_bind_mounts = false # Where the gateway stages rootfs tar archives for `--from ./rootfs.tar`. # Defaults to /rootfs-tar-staging. The gateway creates one # request-scoped subdirectory per staging slot and removes it after use. diff --git a/docs/how-it-works/sandboxes/runtimes.mdx b/docs/how-it-works/sandboxes/runtimes.mdx index 50e980e857..20db806f1d 100644 --- a/docs/how-it-works/sandboxes/runtimes.mdx +++ b/docs/how-it-works/sandboxes/runtimes.mdx @@ -210,6 +210,29 @@ VM sandboxes have no network interface. All traffic flows through the OpenShell If the gateway is briefly unavailable while a VM starts, its supervisor retries the session connection. The sandbox stays in provisioning until the gateway accepts the session. +### MicroVM Mounts + +The VM driver shares gateway-host directories into the guest with virtiofs. Each entry in `mounts` takes `source` (absolute host directory), `target`, optional `read_only` (default `true`), and optional `type`, which must be `bind` so Docker and Podman bind entries work unchanged. Single-file sources, `selinux_label`, and other mount types are not supported: + +```shell +openshell sandbox create \ + --driver-config-json '{"vm":{"mounts":[{"type":"bind","source":"/srv/datasets","target":"/sandbox/datasets"}]}}' \ + -- claude +``` + +VM mounts work only with the libkrun backend, so GPU sandboxes cannot use them. They carry the same risk as [Docker bind mounts](#docker-mounts) and require the same opt-in: + +```toml +[openshell.drivers.vm] +allow_driver_config = true +enable_bind_mounts = true + +[openshell.drivers.vm.resource_admission] +enabled = false +``` + +Targets cannot replace the workspace root, OpenShell paths, or VM-internal paths such as `/tmp`, `/proc`, `/dev`, and `/srv`, and cannot be nested inside one another. The guest refuses a target that passes through a symlink in the image. + ## Kubernetes Driver The Kubernetes driver runs sandboxes as pods in a sandbox namespace. It requires the [Agent Sandbox](https://github.com/kubernetes-sigs/agent-sandbox) controller. Install the gateway with the Helm chart; see [Kubernetes setup](/kubernetes/setup). diff --git a/e2e/rust/Cargo.toml b/e2e/rust/Cargo.toml index 6b1fc73723..df59d61e35 100644 --- a/e2e/rust/Cargo.toml +++ b/e2e/rust/Cargo.toml @@ -73,6 +73,11 @@ name = "rootfs_tar" path = "tests/rootfs_tar.rs" required-features = ["e2e-vm"] +[[test]] +name = "vm_bind_mount" +path = "tests/vm_bind_mount.rs" +required-features = ["e2e-vm"] + [[test]] name = "docker_preflight" path = "tests/docker_preflight.rs" diff --git a/e2e/rust/e2e-vm.sh b/e2e/rust/e2e-vm.sh index ebb160c987..e92d629ddf 100755 --- a/e2e/rust/e2e-vm.sh +++ b/e2e/rust/e2e-vm.sh @@ -288,8 +288,17 @@ else grpc_endpoint = "https://host.openshell.internal:${HOST_PORT}" driver_dir = "${DRIVER_DIR}" state_dir = "${RUN_STATE_DIR}" +enable_bind_mounts = true EOF fi +# Host bind mounts are exercised by vm_bind_mount, mirroring the Docker and +# Podman e2e gateways. An external driver must acknowledge the same policy. +cat >>"${GATEWAY_CONFIG}" <"${DRIVER_LOG}" 2>&1 & DRIVER_PID=$! e2e_wait_for_socket \ @@ -410,6 +421,7 @@ else run_e2e_test ephemeral_cleanup run_e2e_test host_gateway_alias run_e2e_test vm_overlay + run_e2e_test vm_bind_mount run_e2e_test vm_gateway_start run_e2e_test vm_corporate_proxy fi diff --git a/e2e/rust/tests/vm_bind_mount.rs b/e2e/rust/tests/vm_bind_mount.rs new file mode 100644 index 0000000000..896b5410fc --- /dev/null +++ b/e2e/rust/tests/vm_bind_mount.rs @@ -0,0 +1,90 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! VM driver host bind mounts over virtiofs: a read-write share round-trips +//! data with the host and a read-only share rejects writes. +//! +//! Needs the libkrun backend (macOS HVF or Linux KVM); QEMU rejects virtiofs +//! mounts. The e2e VM gateway configures no GPU or QEMU-only lifecycle +//! extension, so this non-GPU sandbox always launches on libkrun. + +use std::fs; +use std::os::unix::fs::PermissionsExt; + +use openshell_e2e::harness::sandbox::SandboxGuard; + +const RW_TARGET: &str = "/sandbox/e2e-bind"; +const RO_TARGET: &str = "/sandbox/e2e-bind-ro"; + +fn shared_host_dir(prefix: &str) -> tempfile::TempDir { + let dir = tempfile::Builder::new() + .prefix(prefix) + .tempdir() + .expect("create bind mount host dir"); + fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o777)) + .expect("make bind mount host dir writable by sandbox user"); + let input = dir.path().join("input.txt"); + fs::write(&input, "host-bind-ok").expect("seed bind mount host dir"); + fs::set_permissions(&input, fs::Permissions::from_mode(0o666)) + .expect("make bind mount input readable by sandbox user"); + dir +} + +#[tokio::test] +async fn vm_bind_mount() { + let rw_dir = shared_host_dir("openshell-e2e-vm-bind-rw-"); + let ro_dir = shared_host_dir("openshell-e2e-vm-bind-ro-"); + + // `type: bind` keeps the entry identical to Docker/Podman bind mounts. + let driver_config = serde_json::json!({ + "vm": { + "mounts": [ + { + "type": "bind", + "source": rw_dir.path(), + "target": RW_TARGET, + "read_only": false + }, + { + "source": ro_dir.path(), + "target": RO_TARGET + } + ] + } + }) + .to_string(); + + let script = format!( + "set -eu; \ + test \"$(cat {RW_TARGET}/input.txt)\" = host-bind-ok; \ + test \"$(cat {RO_TARGET}/input.txt)\" = host-bind-ok; \ + printf sandbox-bind-ok > {RW_TARGET}/output.txt; \ + if printf x > {RO_TARGET}/output.txt 2>/dev/null; then echo ro-writable; exit 1; fi; \ + echo vm-bind-ok" + ); + let mut sandbox = SandboxGuard::create(&[ + "--driver-config-json", + &driver_config, + "--", + "sh", + "-lc", + &script, + ]) + .await + .expect("sandbox create with VM bind mounts"); + + assert!( + sandbox.create_output.contains("vm-bind-ok"), + "sandbox should read both shares, write the rw share, and be denied on the ro share:\n{}", + sandbox.create_output + ); + + sandbox.cleanup().await; + let output = fs::read_to_string(rw_dir.path().join("output.txt")) + .expect("read sandbox output from rw host dir"); + assert_eq!(output, "sandbox-bind-ok"); + assert!( + !ro_dir.path().join("output.txt").exists(), + "read-only share must not receive writes" + ); +}