From 5b5428d69378a08dda2889ba2f84bf4dec5e7e2d Mon Sep 17 00:00:00 2001 From: Enrique Saurez Date: Mon, 5 Oct 2026 10:51:18 -0700 Subject: [PATCH 1/5] Add opt-in image-backed native-host edge lifecycle Load a separately supplied, caller-approved native host library for scratchless image-backed provision/start/exec/stop/deprovision. Keep the direct backend as default, isolate persisted state, fence unsupported capabilities, and pin the RAM-overlay OpenVMM change. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- aci_edge_sandboxes/Cargo.lock | 71 +- aci_edge_sandboxes/Cargo.toml | 9 + aci_edge_sandboxes/README.md | 52 +- .../examples/nvxhost_lifecycle.rs | 163 +++ aci_edge_sandboxes/src/client.rs | 6 + aci_edge_sandboxes/src/lib.rs | 2 + aci_edge_sandboxes/src/nvxhost.rs | 886 ++++++++++++++ aci_edge_sandboxes/src/openvmm/launch.rs | 1 + aci_edge_sandboxes/src/openvmm/mod.rs | 5 + aci_edge_sandboxes/src/openvmm/native.rs | 1074 +++++++++++++++++ aci_edge_sandboxes/src/openvmm/state.rs | 127 +- aci_edge_sandboxes/tests/nvxhost_guest.rs | 150 +++ openvmm | 2 +- 13 files changed, 2538 insertions(+), 10 deletions(-) create mode 100644 aci_edge_sandboxes/examples/nvxhost_lifecycle.rs create mode 100644 aci_edge_sandboxes/src/nvxhost.rs create mode 100644 aci_edge_sandboxes/src/openvmm/native.rs create mode 100644 aci_edge_sandboxes/tests/nvxhost_guest.rs diff --git a/aci_edge_sandboxes/Cargo.lock b/aci_edge_sandboxes/Cargo.lock index c8d486c48..486ede1b1 100644 --- a/aci_edge_sandboxes/Cargo.lock +++ b/aci_edge_sandboxes/Cargo.lock @@ -8,6 +8,8 @@ version = "0.1.0" dependencies = [ "getrandom", "libc", + "libloading", + "prost", "serde", "serde_json", "sha2", @@ -17,6 +19,12 @@ dependencies = [ "windows-sys", ] +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + [[package]] name = "bitflags" version = "2.13.2" @@ -73,6 +81,12 @@ dependencies = [ "crypto-common", ] +[[package]] +name = "either" +version = "1.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" + [[package]] name = "errno" version = "0.3.14" @@ -110,6 +124,15 @@ dependencies = [ "r-efi", ] +[[package]] +name = "itertools" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" +dependencies = [ + "either", +] + [[package]] name = "itoa" version = "1.0.18" @@ -122,6 +145,16 @@ version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" +[[package]] +name = "libloading" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d7c4b02199fee7c5d21a5ae7d8cfa79a6ef5bb2fc834d6e9058e89c825efdc55" +dependencies = [ + "cfg-if", + "windows-link", +] + [[package]] name = "linux-raw-sys" version = "0.12.1" @@ -155,6 +188,29 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "prost" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5" +dependencies = [ + "bytes", + "prost-derive", +] + +[[package]] +name = "prost-derive" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d" +dependencies = [ + "anyhow", + "itertools", + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "quote" version = "1.0.47" @@ -210,7 +266,7 @@ checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.6", ] [[package]] @@ -237,6 +293,17 @@ dependencies = [ "digest", ] +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "syn" version = "3.0.6" @@ -278,7 +345,7 @@ checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.6", ] [[package]] diff --git a/aci_edge_sandboxes/Cargo.toml b/aci_edge_sandboxes/Cargo.toml index 6cc84402a..14e661ae3 100644 --- a/aci_edge_sandboxes/Cargo.toml +++ b/aci_edge_sandboxes/Cargo.toml @@ -23,11 +23,16 @@ openvmm = [] bundled = ["openvmm", "dep:sha2"] # Tokio-based asynchronous wrappers around the synchronous core. async = ["dep:tokio"] +# Loads an explicitly supplied native host library and boots an image-backed ttrpc guest. +nvxhost = ["openvmm", "dep:libloading", "dep:prost", "dep:sha2-runtime"] # Test doubles: an in-memory `MockBackend` and the `aci-edge-sandboxes-fake-openvmm` binary. testing = [] [dependencies] getrandom = "0.4" +libloading = { version = "0.8", optional = true } +prost = { version = "0.13", optional = true } +sha2-runtime = { package = "sha2", version = "0.10", optional = true } serde = { version = "1.0.200", features = ["derive"] } serde_json = "1.0.120" thiserror = "2" @@ -65,6 +70,10 @@ doc = false name = "lifecycle" required-features = ["openvmm"] +[[example]] +name = "nvxhost_lifecycle" +required-features = ["nvxhost"] + [lints.rust] missing_docs = "warn" unsafe_op_in_unsafe_fn = "deny" diff --git a/aci_edge_sandboxes/README.md b/aci_edge_sandboxes/README.md index c2f69fc57..88a80fbac 100644 --- a/aci_edge_sandboxes/README.md +++ b/aci_edge_sandboxes/README.md @@ -3,12 +3,15 @@ `aci_edge_sandboxes` is the Rust interface to the ACI Edge Sandboxes lifecycle. It exposes provision, start, exec, stop, and deprovision through a pluggable backend. The default backend drives the `openvmm` binary directly; no daemon or -Python tooling is involved at runtime. +Python tooling is involved at runtime. The optional `nvxhost` backend uses a +separately supplied native host library and a different, image-backed guest. -Each sandbox is a microVM that runs the NVX guest's Alpine Linux userland -directly from its initramfs. There are no image layers, scratch disks, or -container namespaces; workloads run as a non-root user in the guest itself, and -guest state lives in memory until the sandbox stops. +By default, each sandbox is a microVM that runs the NVX guest's Alpine Linux +userland directly from its initramfs. That backend has no image layers, scratch +disks, or container namespaces; workloads run as a non-root user in the guest +itself, and guest state lives in memory until the sandbox stops. The opt-in +native backend instead runs commands against a caller-supplied, read-only GPT +image with a RAM overlay. The Cargo package, library, and directory are named `aci_edge_sandboxes`. @@ -95,6 +98,8 @@ envelope (`version`, `phase`, and `containment`) belongs to the caller. - Backends in this crate: - `openvmm::OpenVmmBackend` (feature `openvmm`, default) drives the `openvmm` executable. + - `openvmm::NvxHostBackend` (feature `nvxhost`) drives OpenVMM using a + separately supplied native host library and image-backed guest. - `testing::MockBackend` (feature `testing`) implements the state machine in memory for consumers' unit tests. - `AsyncAciEdgeSandbox` (feature `async`) wraps `AciEdgeSandbox` for Tokio. Lifecycle calls run on @@ -104,6 +109,43 @@ To add a backend, implement `Backend`. Declare only the capabilities it can enforce, and report state-machine violations with the codes listed in the [error mapping](#error-mapping). +## Image-backed native host backend (opt-in) + +Enable `nvxhost` to use `AciEdgeSandbox::nvxhost(NvxHostConfig)`. This leaves the default +direct-OpenVMM backend and its Alpine guest unchanged. The caller supplies OpenVMM, a +compatible kernel and static edge-agent initramfs, a prepared GPT image, the absolute path +to `nvxhost.dll` (`libnvxhost.so` on Linux), and the independently approved SHA-256 of that +library. The Rust crate does **not** build, download, or publish the private library. +`NvxHostBackend::new` verifies the file digest and ABI version and requires the additive +`nvx_build_ramfs_launch_arguments` and `nvx_session_connect_verified` exports. A missing, +wrong-version, or wrong-digest library fails rather than falling back. + +The image is attached read-only in OpenVMM's distro block slot. The edge guest validates +its GPT and p2+ ext4 layers and uses a RAM-backed tmpfs upper/work for its overlay; it +does not interpret p1 OCI configuration or require a scratch disk. OpenVMM must support +the explicit `nvx_overlay_upper=ramfs` scratchless topology. On Windows, the backend +assigns `//./pipe/openvmm-microvm-` control and boot endpoints. It retains the +OpenVMM process/state, sends the 32-byte capability through stdin, checks the serving +process on the connected pipe/socket, and owns graceful-or-forced stop and deprovision. +State is kept separately under `/nvxhost/`; a running VM is never silently +converted to the direct backend. `NvxHostBackend::guest_logs` reads a bounded non-follow +guest-log snapshot before stop. + +This first backend supports provision/start/exec/stop/deprovision with shell commands or +argv and a fixed caller-provided image. It explicitly rejects host file mappings, network +configuration beyond deny-all, piped stdin, execution cancellation, custom working +directories and environments. Snapshot/restore, image selection, and richer guest +operations are not part of this profile. See +[`examples/nvxhost_lifecycle.rs`](examples/nvxhost_lifecycle.rs) for a run requiring +`--openvmm`, `--kernel`, `--initrd`, `--image`, `--host-library`, `--host-sha256`, +`--state-root`, `--hypervisor`, and a command after `--`. The ignored +`tests/nvxhost_guest.rs` exercises an actual WHP guest when the corresponding +`NVXHOST_TEST_*` paths and approved DLL digest are set. + +The caller must independently approve and protect the native asset. Checking a caller-supplied +digest does not make a writable path or a self-declared digest trustworthy; use this profile +only with an immutable, externally authorized library installation. + ## OpenVMM backend ### Configuration diff --git a/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs b/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs new file mode 100644 index 000000000..dddd56b51 --- /dev/null +++ b/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs @@ -0,0 +1,163 @@ +//! Runs the image-backed guest lifecycle with a separately supplied native host library. + +use std::io::Write; +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::Arc; + +use aci_edge_sandboxes::openvmm::{Hypervisor, NvxHostBackend, NvxHostConfig, OpenVmmConfig}; +use aci_edge_sandboxes::{AciEdgeSandbox, ExecOutcome, ExecRequest, ProvisionRequest}; + +fn main() -> ExitCode { + match run() { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::FAILURE + } + } +} + +fn run() -> Result<(), String> { + let mut openvmm = None; + let mut kernel = None; + let mut initrd = None; + let mut image = None; + let mut library = None; + let mut digest = None; + let mut root = None; + let mut hypervisor = None; + let mut command = None; + let mut args = std::env::args().skip(1); + while let Some(option) = args.next() { + let mut value = || { + args.next() + .ok_or_else(|| format!("{option} requires a value")) + }; + match option.as_str() { + "--openvmm" => openvmm = Some(value()?), + "--kernel" => kernel = Some(value()?), + "--initrd" => initrd = Some(value()?), + "--image" => image = Some(value()?), + "--host-library" => library = Some(value()?), + "--host-sha256" => digest = Some(parse_digest(&value()?)?), + "--state-root" => root = Some(value()?), + "--hypervisor" => hypervisor = Some(value()?.parse::().map_err(describe)?), + "--" => { + command = Some(args.by_ref().collect::>().join(" ")); + break; + } + other => return Err(format!("unexpected option {other}")), + } + } + let required = + |name: &str, value: Option| value.ok_or_else(|| format!("missing required {name}")); + let openvmm = required("--openvmm", openvmm)?; + let kernel = required("--kernel", kernel)?; + let initrd = required("--initrd", initrd)?; + let image = required("--image", image)?; + let library = required("--host-library", library)?; + let root = required("--state-root", root)?; + let digest = digest.ok_or("missing required --host-sha256")?; + let hypervisor = hypervisor.ok_or("missing required --hypervisor")?; + let command = command + .filter(|command| !command.is_empty()) + .ok_or("pass the guest command after --")?; + + let config = OpenVmmConfig::new(openvmm, kernel, initrd, hypervisor, PathBuf::from(root)); + let backend = Arc::new( + NvxHostBackend::new(NvxHostConfig::new(config, image, library, digest)) + .map_err(describe)?, + ); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let id = client + .provision(&ProvisionRequest::new()) + .map_err(describe)? + .sandbox_id; + let started = client.start(&id); + let succeeded = started.is_ok(); + let executed = started.and_then(|_| { + let output = client + .exec(&id, &ExecRequest::command_line(command))? + .wait_with_output()?; + let logs = backend.guest_logs(&id)?; + if logs.is_empty() || !logs.windows(8).any(|window| window == b"execute:") { + return Err(aci_edge_sandboxes::Error::new( + aci_edge_sandboxes::ErrorCode::BackendError, + "the guest did not return its execution log", + )); + } + Ok(output) + }); + let stopped = succeeded.then(|| client.stop(&id)); + let deprovisioned = if stopped.as_ref().is_none_or(Result::is_ok) { + Some(client.deprovision(&id)) + } else { + None + }; + let mut failures = Vec::new(); + if let Err(error) = &executed { + failures.push(format!("guest lifecycle: {error}")); + } + if let Some(Err(error)) = &stopped { + failures.push(format!("stop: {error}")); + } + if let Some(Err(error)) = &deprovisioned { + failures.push(format!("deprovision: {error}")); + } + if !failures.is_empty() { + return Err(failures.join("; ")); + } + let stopped = stopped + .expect("started guests always have a stop result") + .map_err(describe)?; + if stopped + .metadata + .as_ref() + .and_then(|metadata| metadata.get("forced")) + .and_then(serde_json::Value::as_bool) + != Some(false) + { + return Err("the guest did not shut down gracefully".to_owned()); + } + let output = executed.map_err(describe)?; + std::io::stdout() + .write_all(&output.stdout) + .map_err(|error| error.to_string())?; + std::io::stderr() + .write_all(&output.stderr) + .map_err(|error| error.to_string())?; + if output.outcome != ExecOutcome::Exited(0) { + return Err(format!("guest command {}", output.outcome)); + } + Ok(()) +} + +fn parse_digest(hex: &str) -> Result<[u8; 32], String> { + if hex.len() != 64 || !hex.is_ascii() { + return Err("--host-sha256 must contain exactly 64 hexadecimal digits".to_owned()); + } + let mut digest = [0u8; 32]; + for (index, byte) in digest.iter_mut().enumerate() { + *byte = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16) + .map_err(|_| "--host-sha256 must be hexadecimal")?; + } + Ok(digest) +} + +fn describe(error: aci_edge_sandboxes::Error) -> String { + error.to_string() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn digest_requires_exact_hex_bytes() { + assert_eq!(parse_digest(&"ab".repeat(32)).unwrap(), [0xab; 32]); + assert!(parse_digest("ab").is_err()); + assert!(parse_digest(&"é".repeat(32)).is_err()); + assert!(parse_digest(&"gg".repeat(32)).is_err()); + } +} diff --git a/aci_edge_sandboxes/src/client.rs b/aci_edge_sandboxes/src/client.rs index 8a7c31d47..0d631ee7b 100644 --- a/aci_edge_sandboxes/src/client.rs +++ b/aci_edge_sandboxes/src/client.rs @@ -43,6 +43,12 @@ impl AciEdgeSandbox { Ok(Self::new(crate::openvmm::OpenVmmBackend::new(config)?)) } + /// Creates an image-backed sandbox through an explicitly supplied nvxhost library. + #[cfg(feature = "nvxhost")] + pub fn nvxhost(config: crate::openvmm::NvxHostConfig) -> Result { + Ok(Self::new(crate::openvmm::NvxHostBackend::new(config)?)) + } + /// Returns the backend. pub fn backend(&self) -> &dyn Backend { self.backend.as_ref() diff --git a/aci_edge_sandboxes/src/lib.rs b/aci_edge_sandboxes/src/lib.rs index 1a2dc6954..c423d6196 100644 --- a/aci_edge_sandboxes/src/lib.rs +++ b/aci_edge_sandboxes/src/lib.rs @@ -58,6 +58,8 @@ mod exec; mod id; mod input; mod model; +#[cfg(feature = "nvxhost")] +mod nvxhost; mod stream; mod validate; diff --git a/aci_edge_sandboxes/src/nvxhost.rs b/aci_edge_sandboxes/src/nvxhost.rs new file mode 100644 index 000000000..c052f10fa --- /dev/null +++ b/aci_edge_sandboxes/src/nvxhost.rs @@ -0,0 +1,886 @@ +//! Checked dynamic binding for the privately supplied nvxhost C ABI. + +#[cfg(not(target_pointer_width = "64"))] +compile_error!("the nvxhost ABI is supported only on 64-bit hosts"); + +use std::ffi::{OsString, c_void}; +use std::fs; +use std::mem::size_of; +use std::path::Path; +use std::ptr; +use std::slice; +use std::sync::Arc; +use std::thread; +use std::time::{Duration, Instant}; + +use libloading::Library; +use prost::Message; +use sha2_runtime::{Digest, Sha256}; + +use crate::error::{Error, Result}; + +const ABI_VERSION: u32 = 1; +const STATUS_OK: i32 = 0; +const STATUS_TIMEOUT: i32 = 2; +const RESULT_OK: i32 = 0; +const RESULT_ERROR: i32 = 1; +const COMPLETION_CONNECT: u32 = 1; +const COMPLETION_SEND: u32 = 2; +const COMPLETION_RECV: u32 = 3; +const COMPLETION_DISPOSE: u32 = 6; +const ERROR_CONNECT: u32 = 12; +const CALL_UNARY: u32 = 0; +const CALL_SERVER_STREAM: u32 = 1; +const FRAME_RESPONSE: u8 = 2; +const FRAME_DATA: u8 = 3; +const FLAG_REMOTE_CLOSED: u8 = 1; +const FLAG_NO_DATA: u8 = 4; +const LAUNCH_GUEST_DEBUG: u32 = 1; +const LAUNCH_HAS_MEMORY: u32 = 4; + +#[repr(C)] +#[derive(Clone, Copy, Default)] +struct NvxStr { + data: *const u8, + length: usize, + present: u32, + reserved: u32, +} + +impl NvxStr { + fn present(text: &str) -> Self { + Self { + data: text.as_ptr(), + length: text.len(), + present: 1, + reserved: 0, + } + } +} + +#[repr(C)] +#[derive(Default)] +struct NvxLaunchRequest { + struct_size: u32, + flags: u32, + cpus: i32, + memory_mb: i32, + kernel_path: NvxStr, + initrd_path: NvxStr, + control_socket_path: NvxStr, + boot_console_socket_path: NvxStr, + hypervisor: NvxStr, + vhd_path: NvxStr, + scratch_vhd_path: NvxStr, + isolation_profile: NvxStr, + startup_mode: NvxStr, +} + +#[repr(C)] +struct NvxCompletion { + struct_size: u32, + kind: u32, + token: u64, + result: i32, + frame_type: u8, + frame_flags: u8, + reserved: u16, + handle: u64, + value: u64, + data: *const u8, + data_length: usize, + error: *mut c_void, +} + +#[repr(C)] +#[derive(Default)] +struct NvxErrorEntry { + struct_size: u32, + kind: u32, + reason: u32, + has_os_error: u32, + os_error: i32, + reserved: u32, + identity: u64, + external_token: u64, + message: *const u8, + message_length: usize, + param: *const u8, + param_length: usize, + data: *const u8, + data_length: usize, +} + +const _: () = { + assert!(size_of::() == 24); + assert!(size_of::() == 232); + assert!(size_of::() == 64); + assert!(size_of::() == 88); + assert!(std::mem::offset_of!(NvxCompletion, error) == 56); + assert!(std::mem::offset_of!(NvxLaunchRequest, hypervisor) == 112); +}; + +type GetVersion = unsafe extern "C" fn() -> u32; +type RuntimeCreate = unsafe extern "C" fn(u32, *mut u64, *mut *mut c_void) -> i32; +type RuntimeNext = unsafe extern "C" fn(u64, u32, *mut *mut NvxCompletion) -> i32; +type CompletionRelease = unsafe extern "C" fn(*mut NvxCompletion); +type OperationCancel = unsafe extern "C" fn(u64, u64) -> i32; +type RuntimeFree = unsafe extern "C" fn(u64) -> i32; +type ErrorCount = unsafe extern "C" fn(*const c_void) -> u32; +type ErrorGet = unsafe extern "C" fn(*const c_void, u32, *mut NvxErrorEntry) -> i32; +type ErrorFree = unsafe extern "C" fn(*mut c_void); +type SessionConnectVerified = + unsafe extern "C" fn(u64, *const u8, usize, *const u8, usize, u32, u64) -> i32; +type SessionDispose = unsafe extern "C" fn(u64, u64) -> i32; +type SessionRelease = unsafe extern "C" fn(u64) -> i32; +type CallOpen = + unsafe extern "C" fn(u64, *const u8, usize, u32, *mut u64, *mut u32, *mut *mut c_void) -> i32; +type CallSendRequest = unsafe extern "C" fn(u64, *const u8, usize, u8, i64, u32, u64) -> i32; +type CallRecv = unsafe extern "C" fn(u64, u64) -> i32; +type CallTryComplete = unsafe extern "C" fn(u64) -> i32; +type CallRelease = unsafe extern "C" fn(u64) -> i32; +type BuildRamfsArguments = unsafe extern "C" fn( + *const NvxLaunchRequest, + *mut *mut u8, + *mut usize, + *mut *mut c_void, +) -> i32; +type BufferFree = unsafe extern "C" fn(*mut u8, usize); + +struct Exports { + runtime_create: RuntimeCreate, + runtime_next: RuntimeNext, + completion_release: CompletionRelease, + operation_cancel: OperationCancel, + runtime_free: RuntimeFree, + error_count: ErrorCount, + error_get: ErrorGet, + error_free: ErrorFree, + session_connect_verified: SessionConnectVerified, + session_dispose: SessionDispose, + session_release: SessionRelease, + call_open: CallOpen, + call_send_request: CallSendRequest, + call_recv: CallRecv, + call_try_complete: CallTryComplete, + call_release: CallRelease, + build_ramfs_arguments: BuildRamfsArguments, + buffer_free: BufferFree, +} + +pub(crate) struct HostLibrary { + _library: Library, + exports: Exports, +} + +impl std::fmt::Debug for HostLibrary { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter + .debug_struct("HostLibrary") + .finish_non_exhaustive() + } +} + +fn resolve(library: &Library, name: &[u8]) -> Result { + // Function pointers remain valid because HostLibrary retains the loaded library. + unsafe { library.get::(name) } + .map(|symbol| *symbol) + .map_err(|error| { + Error::backend_unavailable(format!( + "nvxhost is missing required export {}", + String::from_utf8_lossy(name).trim_end_matches('\0') + )) + .with_source(error) + }) +} + +impl HostLibrary { + pub(crate) fn load(path: &Path, expected_sha256: &[u8; 32]) -> Result> { + if !path.is_absolute() { + return Err(Error::backend_unavailable( + "the nvxhost library path must be absolute", + )); + } + let path = fs::canonicalize(path).map_err(|error| { + Error::backend_unavailable(format!("cannot locate nvxhost at {}", path.display())) + .with_source(error) + })?; + let bytes = fs::read(&path).map_err(|error| { + Error::backend_unavailable(format!("cannot verify nvxhost at {}", path.display())) + .with_source(error) + })?; + let actual: [u8; 32] = Sha256::digest(&bytes).into(); + if &actual != expected_sha256 { + return Err(Error::backend_unavailable(format!( + "nvxhost at {} does not match the approved SHA-256", + path.display() + ))); + } + let library = unsafe { Library::new(&path) }.map_err(|error| { + Error::backend_unavailable(format!("cannot load nvxhost from {}", path.display())) + .with_source(error) + })?; + let version: GetVersion = resolve(&library, b"nvx_get_api_version\0")?; + let actual_version = unsafe { version() }; + if actual_version != ABI_VERSION { + return Err(Error::backend_unavailable(format!( + "nvxhost ABI is {actual_version}, expected {ABI_VERSION}" + ))); + } + let exports = Exports { + runtime_create: resolve(&library, b"nvx_runtime_create\0")?, + runtime_next: resolve(&library, b"nvx_runtime_next_completion\0")?, + completion_release: resolve(&library, b"nvx_completion_release\0")?, + operation_cancel: resolve(&library, b"nvx_operation_cancel\0")?, + runtime_free: resolve(&library, b"nvx_runtime_free\0")?, + error_count: resolve(&library, b"nvx_error_count\0")?, + error_get: resolve(&library, b"nvx_error_get\0")?, + error_free: resolve(&library, b"nvx_error_free\0")?, + session_connect_verified: resolve(&library, b"nvx_session_connect_verified\0")?, + session_dispose: resolve(&library, b"nvx_session_dispose\0")?, + session_release: resolve(&library, b"nvx_session_release\0")?, + call_open: resolve(&library, b"nvx_call_open\0")?, + call_send_request: resolve(&library, b"nvx_call_send_request\0")?, + call_recv: resolve(&library, b"nvx_call_recv\0")?, + call_try_complete: resolve(&library, b"nvx_call_try_complete\0")?, + call_release: resolve(&library, b"nvx_call_release\0")?, + build_ramfs_arguments: resolve(&library, b"nvx_build_ramfs_launch_arguments\0")?, + buffer_free: resolve(&library, b"nvx_buffer_free\0")?, + }; + Ok(Arc::new(Self { + _library: library, + exports, + })) + } + + pub(crate) fn launch_arguments(&self, inputs: &LaunchInputs<'_>) -> Result> { + fn path_text(path: &Path) -> Result<&str> { + path.to_str().ok_or_else(|| { + Error::backend_unavailable(format!( + "the nvxhost ABI requires a UTF-8 artifact path: {}", + path.display() + )) + }) + } + let kernel = path_text(inputs.kernel)?; + let initrd = path_text(inputs.initrd)?; + let image = path_text(inputs.image)?; + let memory_mb = i32::try_from(inputs.memory_mb) + .map_err(|_| Error::policy_validation("guest memory exceeds the nvxhost ABI range"))?; + let request = NvxLaunchRequest { + struct_size: size_of::() as u32, + flags: LAUNCH_HAS_MEMORY + | if inputs.guest_debug { + LAUNCH_GUEST_DEBUG + } else { + 0 + }, + memory_mb, + kernel_path: NvxStr::present(kernel), + initrd_path: NvxStr::present(initrd), + control_socket_path: NvxStr::present(inputs.control), + boot_console_socket_path: NvxStr::present(inputs.boot), + hypervisor: NvxStr::present(inputs.hypervisor), + vhd_path: NvxStr::present(image), + ..NvxLaunchRequest::default() + }; + let mut buffer = ptr::null_mut(); + let mut length = 0; + let mut error = ptr::null_mut(); + let status = unsafe { + (self.exports.build_ramfs_arguments)(&request, &mut buffer, &mut length, &mut error) + }; + self.check_status("building OpenVMM arguments", status, error)?; + if buffer.is_null() || length == 0 { + return Err(Error::backend_error( + "nvxhost returned no OpenVMM launch arguments", + )); + } + let encoded = unsafe { slice::from_raw_parts(buffer, length).to_vec() }; + unsafe { (self.exports.buffer_free)(buffer, length) }; + if encoded.last() != Some(&0) { + return Err(Error::backend_error( + "nvxhost returned an unterminated argument vector", + )); + } + encoded[..encoded.len() - 1] + .split(|&byte| byte == 0) + .map(|bytes| { + if bytes.is_empty() { + return Err(Error::backend_error( + "nvxhost returned an empty OpenVMM argument", + )); + } + String::from_utf8(bytes.to_vec()) + .map(OsString::from) + .map_err(|error| { + Error::backend_error("nvxhost returned a non-UTF-8 OpenVMM argument") + .with_source(error) + }) + }) + .collect() + } + + fn failure(&self, error: *mut c_void) -> NativeFailure { + if error.is_null() || unsafe { (self.exports.error_count)(error) } == 0 { + return NativeFailure { + kind: 0, + os_error: None, + message: "nvxhost returned an error without details".to_owned(), + }; + } + let mut entry = NvxErrorEntry::default(); + if unsafe { (self.exports.error_get)(error, 0, &mut entry) } != STATUS_OK + || entry.struct_size as usize != size_of::() + { + return NativeFailure { + kind: 0, + os_error: None, + message: "nvxhost returned an incompatible error layout".to_owned(), + }; + } + let message = if entry.message_length == 0 { + "nvxhost returned an empty error message".to_owned() + } else if entry.message.is_null() { + "nvxhost returned an invalid error message pointer".to_owned() + } else { + String::from_utf8_lossy(unsafe { + slice::from_raw_parts(entry.message, entry.message_length) + }) + .into_owned() + }; + NativeFailure { + kind: entry.kind, + os_error: (entry.has_os_error != 0).then_some(entry.os_error), + message, + } + } + + fn check_status(&self, action: &str, status: i32, error: *mut c_void) -> Result<()> { + if status == STATUS_OK && error.is_null() { + return Ok(()); + } + let failure = self.failure(error); + if !error.is_null() { + unsafe { (self.exports.error_free)(error) }; + } + Err(failure.as_error(action, status)) + } +} + +pub(crate) struct LaunchInputs<'a> { + pub(crate) kernel: &'a Path, + pub(crate) initrd: &'a Path, + pub(crate) image: &'a Path, + pub(crate) control: &'a str, + pub(crate) boot: &'a str, + pub(crate) hypervisor: &'a str, + pub(crate) memory_mb: u32, + pub(crate) guest_debug: bool, +} + +#[derive(Debug)] +struct NativeFailure { + kind: u32, + os_error: Option, + message: String, +} + +impl NativeFailure { + fn as_error(&self, action: &str, status: i32) -> Error { + Error::backend_error(format!( + "nvxhost {action} failed (status {status}, kind {}): {}", + self.kind, self.message + )) + } + + fn endpoint_unavailable(&self) -> bool { + self.kind == ERROR_CONNECT + && match self.os_error { + #[cfg(windows)] + Some(2 | 231) => true, + #[cfg(target_os = "linux")] + Some(2 | 111) => true, + _ => false, + } + } +} + +struct Completed { + kind: u32, + token: u64, + result: i32, + frame_type: u8, + frame_flags: u8, + handle: u64, + data: Vec, + error: Option, +} + +#[derive(Clone, PartialEq, Message)] +struct TtrpcStatus { + #[prost(int32, tag = "1")] + code: i32, + #[prost(string, tag = "2")] + message: String, +} + +#[derive(Clone, PartialEq, Message)] +struct TtrpcResponse { + #[prost(message, optional, tag = "1")] + status: Option, + #[prost(bytes, tag = "2")] + payload: Vec, +} + +fn response_payload(method: &str, bytes: &[u8]) -> Result> { + let response = TtrpcResponse::decode(bytes).map_err(|error| { + Error::backend_error(format!("{method} returned an invalid ttrpc response")) + .with_source(error) + })?; + if let Some(status) = response.status + && status.code != 0 + { + return Err(Error::backend_error(format!( + "{method} failed with guest status {}: {}", + status.code, status.message + ))); + } + Ok(response.payload) +} + +pub(crate) struct Session { + host: Arc, + runtime: u64, + session: u64, + next_token: u64, +} + +impl Session { + pub(crate) fn connect_verified( + host: Arc, + endpoint: &str, + capability: &[u8; 32], + expected_pid: u32, + deadline: Instant, + is_running: impl Fn() -> Result, + ) -> Result { + let mut runtime = 0; + let mut error = ptr::null_mut(); + let status = unsafe { (host.exports.runtime_create)(2, &mut runtime, &mut error) }; + host.check_status("starting a native runtime", status, error)?; + if runtime == 0 { + return Err(Error::backend_error( + "nvxhost returned an invalid runtime handle", + )); + } + let mut client = Self { + host, + runtime, + session: 0, + next_token: 1, + }; + loop { + let token = client.token()?; + let status = unsafe { + (client.host.exports.session_connect_verified)( + client.runtime, + endpoint.as_ptr(), + endpoint.len(), + capability.as_ptr(), + capability.len(), + expected_pid, + token, + ) + }; + client + .host + .check_status("connecting to OpenVMM", status, ptr::null_mut())?; + let completed = client.next(COMPLETION_CONNECT, token, Some(deadline))?; + if completed.result == RESULT_OK { + if completed.handle == 0 || completed.data.len() != 16 { + if completed.handle != 0 { + unsafe { (client.host.exports.session_release)(completed.handle) }; + } + return Err(Error::backend_error( + "nvxhost returned an invalid broker session identity", + )); + } + client.session = completed.handle; + return Ok(client); + } + if completed.result != RESULT_ERROR { + return Err(Error::backend_error(format!( + "nvxhost returned unexpected connection result {}", + completed.result + ))); + } + let failure = completed.error.ok_or_else(|| { + Error::backend_error("nvxhost connection failed without error details") + })?; + if failure.endpoint_unavailable() && Instant::now() < deadline && is_running()? { + thread::sleep(Duration::from_millis(25)); + continue; + } + return Err(failure.as_error("connecting to OpenVMM", completed.result)); + } + } + + fn token(&mut self) -> Result { + let token = self.next_token; + self.next_token = token + .checked_add(1) + .ok_or_else(|| Error::backend_error("nvxhost completion tokens have been exhausted"))?; + Ok(token) + } + + fn next(&self, kind: u32, token: u64, deadline: Option) -> Result { + let mut completion = ptr::null_mut(); + let status = loop { + let timeout_ms = match deadline { + Some(deadline) => { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + unsafe { (self.host.exports.operation_cancel)(self.runtime, token) }; + return Err(Error::backend_error("nvxhost operation timed out")); + } + u32::try_from(remaining.as_millis()) + .unwrap_or(u32::MAX) + .max(1) + } + None => 1_000, + }; + let status = unsafe { + (self.host.exports.runtime_next)(self.runtime, timeout_ms, &mut completion) + }; + if status == STATUS_TIMEOUT && deadline.is_none() { + continue; + } + break status; + }; + if status == STATUS_TIMEOUT { + unsafe { (self.host.exports.operation_cancel)(self.runtime, token) }; + return Err(Error::backend_error("nvxhost operation timed out")); + } + self.host + .check_status("waiting for a native completion", status, ptr::null_mut())?; + if completion.is_null() { + return Err(Error::backend_error( + "nvxhost returned a null completion pointer", + )); + } + let result = unsafe { + let frame = &*completion; + if frame.struct_size as usize != size_of::() { + Err(Error::backend_error( + "nvxhost returned an incompatible completion layout", + )) + } else if frame.data_length != 0 && frame.data.is_null() { + Err(Error::backend_error( + "nvxhost returned an invalid completion payload pointer", + )) + } else { + Ok(Completed { + kind: frame.kind, + token: frame.token, + result: frame.result, + frame_type: frame.frame_type, + frame_flags: frame.frame_flags, + handle: frame.handle, + data: if frame.data_length == 0 { + Vec::new() + } else { + slice::from_raw_parts(frame.data, frame.data_length).to_vec() + }, + error: (!frame.error.is_null()).then(|| self.host.failure(frame.error)), + }) + } + }; + unsafe { (self.host.exports.completion_release)(completion) }; + let result = result?; + if result.kind != kind || result.token != token { + if result.kind == COMPLETION_CONNECT && result.handle != 0 { + unsafe { (self.host.exports.session_release)(result.handle) }; + } + return Err(Error::backend_error( + "nvxhost returned a completion for a different operation", + )); + } + Ok(result) + } + + fn open_call(&self, method: &str, kind: u32) -> Result { + let mut call = 0; + let mut stream_id = 0; + let mut error = ptr::null_mut(); + let status = unsafe { + (self.host.exports.call_open)( + self.session, + method.as_ptr(), + method.len(), + kind, + &mut call, + &mut stream_id, + &mut error, + ) + }; + self.host + .check_status("opening a guest RPC", status, error)?; + if call == 0 || stream_id == 0 || stream_id & 1 == 0 { + if call != 0 { + unsafe { (self.host.exports.call_release)(call) }; + } + return Err(Error::backend_error( + "nvxhost returned an invalid RPC stream", + )); + } + Ok(Call { + host: Arc::clone(&self.host), + handle: call, + }) + } + + fn send(&mut self, call: &Call, payload: &[u8], deadline: Option) -> Result<()> { + let timeout_nanos = match deadline { + Some(deadline) => { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + return Err(Error::backend_error( + "guest RPC deadline expired before dispatch", + )); + } + i64::try_from(remaining.as_nanos()) + .map_err(|_| Error::backend_error("guest RPC deadline is too long"))? + } + None => 0, + }; + let token = self.token()?; + let status = unsafe { + (self.host.exports.call_send_request)( + call.handle, + payload.as_ptr(), + payload.len(), + FLAG_REMOTE_CLOSED, + timeout_nanos, + 0, + token, + ) + }; + self.host + .check_status("sending a guest RPC", status, ptr::null_mut())?; + let sent = self.next(COMPLETION_SEND, token, deadline)?; + self.completed_ok("sending a guest RPC", &sent) + } + + fn recv(&mut self, call: &Call, deadline: Option) -> Result { + let token = self.token()?; + let status = unsafe { (self.host.exports.call_recv)(call.handle, token) }; + self.host + .check_status("receiving a guest RPC", status, ptr::null_mut())?; + let frame = self.next(COMPLETION_RECV, token, deadline)?; + self.completed_ok("receiving a guest RPC", &frame)?; + Ok(frame) + } + + fn completed_ok(&self, action: &str, completed: &Completed) -> Result<()> { + if completed.result == RESULT_OK { + return Ok(()); + } + if completed.result != RESULT_ERROR { + return Err(Error::backend_error(format!( + "nvxhost {action} returned unexpected result {}", + completed.result + ))); + } + Err(completed.error.as_ref().map_or_else( + || Error::backend_error(format!("nvxhost {action} returned {}", completed.result)), + |failure| failure.as_error(action, completed.result), + )) + } + + pub(crate) fn unary( + &mut self, + method: &str, + payload: &[u8], + timeout: Option, + ) -> Result> { + let deadline = timeout + .map(|timeout| { + Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("guest RPC timeout is too long")) + }) + .transpose()?; + let call = self.open_call(method, CALL_UNARY)?; + self.send(&call, payload, deadline)?; + let response = self.recv(&call, deadline)?; + if response.frame_type != FRAME_RESPONSE { + return Err(Error::backend_error(format!( + "{method} returned unexpected ttrpc frame type {}", + response.frame_type + ))); + } + if unsafe { (self.host.exports.call_try_complete)(call.handle) } != 1 { + return Err(Error::backend_error(format!( + "{method} did not complete its unary stream" + ))); + } + response_payload(method, &response.data) + } + + pub(crate) fn server_stream( + &mut self, + method: &str, + payload: &[u8], + timeout: Duration, + mut on_message: impl FnMut(&[u8]) -> Result<()>, + ) -> Result<()> { + let deadline = Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("guest RPC timeout is too long"))?; + let call = self.open_call(method, CALL_SERVER_STREAM)?; + self.send(&call, payload, Some(deadline))?; + loop { + let frame = self.recv(&call, Some(deadline))?; + match frame.frame_type { + FRAME_DATA => { + if frame.frame_flags & FLAG_NO_DATA == 0 { + on_message(&frame.data)?; + } + if frame.frame_flags & FLAG_REMOTE_CLOSED != 0 { + break; + } + } + FRAME_RESPONSE => { + let final_payload = response_payload(method, &frame.data)?; + if !final_payload.is_empty() { + on_message(&final_payload)?; + } + break; + } + other => { + return Err(Error::backend_error(format!( + "{method} returned unexpected ttrpc frame type {other}" + ))); + } + } + } + if unsafe { (self.host.exports.call_try_complete)(call.handle) } != 1 { + return Err(Error::backend_error(format!( + "{method} did not complete its server stream" + ))); + } + Ok(()) + } + + pub(crate) fn close(mut self) -> Result<()> { + let token = self.token()?; + let status = unsafe { (self.host.exports.session_dispose)(self.session, token) }; + self.host + .check_status("closing a guest session", status, ptr::null_mut())?; + let disposed = self.next( + COMPLETION_DISPOSE, + token, + Some(Instant::now() + Duration::from_secs(5)), + )?; + self.completed_ok("closing a guest session", &disposed)?; + let status = unsafe { (self.host.exports.session_release)(self.session) }; + self.host + .check_status("releasing a guest session", status, ptr::null_mut())?; + self.session = 0; + Ok(()) + } +} + +impl Drop for Session { + fn drop(&mut self) { + if self.session != 0 { + let status = unsafe { (self.host.exports.session_release)(self.session) }; + if status != STATUS_OK { + eprintln!("nvxhost failed to release guest session: {status}"); + } + } + if self.runtime != 0 { + let status = unsafe { (self.host.exports.runtime_free)(self.runtime) }; + if status != STATUS_OK { + eprintln!("nvxhost failed to release native runtime: {status}"); + } + } + } +} + +struct Call { + host: Arc, + handle: u64, +} + +impl Drop for Call { + fn drop(&mut self) { + let status = unsafe { (self.host.exports.call_release)(self.handle) }; + if status != STATUS_OK { + eprintln!("nvxhost failed to release guest RPC: {status}"); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn ttrpc_response_decodes_the_existing_wire_vector() { + let bytes = [0x0a, 0x00, 0x12, 0x06, 0x7a, 0x04, b'r', b'u', b's', b't']; + assert_eq!( + response_payload("GetGuestInfo", &bytes).unwrap(), + b"z\x04rust" + ); + } + + #[test] + fn missing_library_fails_explicitly() { + let path = std::env::temp_dir().join("nvxhost-missing-file.dll"); + let error = HostLibrary::load(&path, &[0; 32]).unwrap_err(); + assert!(error.message().contains("cannot locate nvxhost")); + } + + #[test] + #[ignore = "set NVXHOST_TEST_LIBRARY to a separately built nvxhost DLL or shared library"] + fn privately_built_library_exposes_the_ramfs_launch_abi() { + let path = std::env::var_os("NVXHOST_TEST_LIBRARY") + .map(std::path::PathBuf::from) + .expect("NVXHOST_TEST_LIBRARY must name a private nvxhost build"); + let bytes = fs::read(&path).unwrap(); + let digest: [u8; 32] = Sha256::digest(bytes).into(); + let host = HostLibrary::load(&path, &digest).unwrap(); + let error = HostLibrary::load(&path, &[0; 32]).unwrap_err(); + assert!( + error + .message() + .contains("does not match the approved SHA-256") + ); + let args = host + .launch_arguments(&LaunchInputs { + kernel: Path::new(r"C:\test\vmlinux"), + initrd: Path::new(r"C:\test\edge-initramfs.cpio.gz"), + image: Path::new(r"C:\test\image.gpt"), + control: "//./pipe/openvmm-microvm-edge-control", + boot: "//./pipe/openvmm-microvm-edge-boot", + hypervisor: "whp", + memory_mb: 256, + guest_debug: false, + }) + .unwrap(); + let args: Vec<_> = args + .iter() + .map(|argument| argument.to_str().unwrap()) + .collect(); + assert!(args.contains(&"--microvm-control-auth-stdin")); + assert!(args.contains(&"distro:file:C:\\test\\image.gpt,ro")); + assert_eq!( + args.iter() + .filter(|&&argument| argument == "--microvm-sandbox-block") + .count(), + 1 + ); + } +} diff --git a/aci_edge_sandboxes/src/openvmm/launch.rs b/aci_edge_sandboxes/src/openvmm/launch.rs index db3a67ee5..2f0441564 100644 --- a/aci_edge_sandboxes/src/openvmm/launch.rs +++ b/aci_edge_sandboxes/src/openvmm/launch.rs @@ -107,6 +107,7 @@ mod tests { backend: BACKEND_KEY.to_owned(), network: None, filesystem: None, + native: None, memory_mib: 256, workload_uid: 65534, workload_gid: 65534, diff --git a/aci_edge_sandboxes/src/openvmm/mod.rs b/aci_edge_sandboxes/src/openvmm/mod.rs index e0d07f864..3da4ffd37 100644 --- a/aci_edge_sandboxes/src/openvmm/mod.rs +++ b/aci_edge_sandboxes/src/openvmm/mod.rs @@ -92,6 +92,8 @@ mod config; mod contract; mod filesystem; mod launch; +#[cfg(feature = "nvxhost")] +mod native; mod network; mod platform; mod process; @@ -111,6 +113,8 @@ use serde_json::Value; pub use self::artifacts::Artifacts; pub use self::config::{Hypervisor, OpenVmmConfig}; pub use self::filesystem::{guest_path, resolve_guest_path}; +#[cfg(feature = "nvxhost")] +pub use self::native::{NvxHostBackend, NvxHostConfig}; use self::protocol::{ CAPABILITY_LEN, ExitCategory, GuestFeatures, MAX_ARGUMENT_BYTES, MAX_CWD_BYTES, MAX_OUTPUT_BYTES, MAX_TIMEOUT_MS, Workload, WorkloadEnvironment, @@ -529,6 +533,7 @@ impl Backend for OpenVmmBackend { backend: BACKEND_KEY.to_owned(), network: request.network.clone(), filesystem, + native: None, memory_mib, workload_uid, workload_gid, diff --git a/aci_edge_sandboxes/src/openvmm/native.rs b/aci_edge_sandboxes/src/openvmm/native.rs new file mode 100644 index 000000000..4dcd37729 --- /dev/null +++ b/aci_edge_sandboxes/src/openvmm/native.rs @@ -0,0 +1,1074 @@ +//! Image-backed guest lifecycle over the separately supplied nvxhost library. + +use std::collections::HashMap; +use std::fs::File; +use std::io::{self, Write}; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, Mutex}; +use std::thread; +use std::time::{Duration, Instant}; + +use prost::Message; +use sha2_runtime::{Digest, Sha256}; + +use super::artifacts::absolute; +use super::config::OpenVmmConfig; +use super::platform; +use super::process; +use super::state::{ + LaunchRecord, NativeArtifactRecord, ProcessIdentity, RuntimeRecord, STATE_FORMAT, + SandboxRecord, StateStore, remove_if_present, +}; +use super::{OpenVmmBackend, RunState}; +use crate::backend::{Backend, ExecControl, ExecIo}; +use crate::capabilities::Capabilities; +use crate::error::{Error, Result}; +use crate::exec::{Completion, ExecOutcome}; +use crate::id::SandboxId; +use crate::model::{ + Access, Command, DeprovisionResult, ExecRequest, Metadata, ProvisionRequest, ProvisionResult, + StartResult, StdinMode, StopResult, duration_millis, +}; +use crate::nvxhost::{HostLibrary, LaunchInputs, Session}; + +const BACKEND_KEY: &str = "nvxhost"; +const RUNTIME_ABI: &str = "microvm-abi-v2-edge-ramfs-v1"; +const MAX_EXEC_SECONDS: u64 = 3600; +const MAX_OUTPUT_BYTES: usize = 1 << 20; +const SHELL: &str = "/bin/sh"; + +/// Artifact paths and approved native-library digest for an image-backed guest. +/// +/// The image is a caller-prepared GPT disk, not a container image reference. This backend +/// creates no disks, filesystem mappings, or network devices. +#[derive(Debug, Clone)] +pub struct NvxHostConfig { + /// OpenVMM and guest boot artifacts, hypervisor, state root, and deadlines. + pub openvmm: OpenVmmConfig, + /// GPT image attached read-only as the distro block device. + pub image: PathBuf, + /// Absolute path to a separately installed nvxhost DLL or shared library. + pub library: PathBuf, + /// SHA-256 approved by the caller's independent artifact policy. + pub library_sha256: [u8; 32], + /// Requests verbose guest kernel diagnostics on the boot console. + pub guest_debug: bool, +} + +impl NvxHostConfig { + /// Uses caller-provided artifacts and an independently supplied native-library digest. + pub fn new( + openvmm: OpenVmmConfig, + image: impl Into, + library: impl Into, + library_sha256: [u8; 32], + ) -> Self { + Self { + openvmm, + image: image.into(), + library: library.into(), + library_sha256, + guest_debug: false, + } + } + + /// Enables verbose guest boot diagnostics without changing the guest lifecycle. + #[must_use] + pub fn with_guest_debug(mut self, enabled: bool) -> Self { + self.guest_debug = enabled; + self + } +} + +/// Owns sandbox state and OpenVMM processes; the private native library owns only +/// the launch arguments and authenticated guest channel. +#[derive(Debug)] +pub struct NvxHostBackend { + config: NvxHostConfig, + base: OpenVmmBackend, + host: Arc, + console_pumps: Mutex>>>, +} + +#[derive(Clone, PartialEq, Message)] +struct GuestInfo { + #[prost(int32, tag = "10")] + readiness: i32, + #[prost(uint64, tag = "11")] + readiness_generation: u64, + #[prost(string, tag = "15")] + agent_build_id: String, + #[prost(string, tag = "16")] + runtime_abi: String, + #[prost(string, tag = "17")] + startup_error: String, +} + +#[derive(Clone, PartialEq, Message)] +struct WaitReadyRequest { + #[prost(uint64, tag = "1")] + minimum_generation: u64, +} + +#[derive(Clone, PartialEq, Message)] +struct WaitReadyResponse { + #[prost(int32, tag = "1")] + readiness: i32, + #[prost(uint64, tag = "2")] + generation: u64, + #[prost(string, tag = "3")] + startup_error: String, +} + +#[derive(Clone, PartialEq, Message)] +struct ExecuteCommandRequest { + #[prost(string, tag = "1")] + command: String, + #[prost(string, repeated, tag = "2")] + args: Vec, + #[prost(int32, tag = "5")] + timeout_seconds: i32, +} + +#[derive(Clone, PartialEq, Message)] +struct ExecuteCommandResponse { + #[prost(int32, tag = "1")] + exit_code: i32, + #[prost(string, tag = "2")] + stdout: String, + #[prost(string, tag = "3")] + stderr: String, + #[prost(bool, tag = "4")] + timed_out: bool, +} + +#[derive(Clone, PartialEq, Message)] +struct ShutdownRequest { + #[prost(int64, tag = "1")] + grace_period_milliseconds: i64, +} + +#[derive(Clone, PartialEq, Message)] +struct ShutdownResponse {} + +#[derive(Clone, PartialEq, Message)] +struct StreamLogsRequest { + #[prost(string, tag = "1")] + container_id: String, + #[prost(uint64, tag = "2")] + offset: u64, + #[prost(bool, tag = "3")] + follow: bool, + #[prost(uint32, tag = "4")] + max_chunk_bytes: u32, +} + +#[derive(Clone, PartialEq, Message)] +struct LogChunk { + #[prost(string, tag = "1")] + container_id: String, + #[prost(uint64, tag = "2")] + offset: u64, + #[prost(bytes, tag = "3")] + data: Vec, + #[prost(uint64, tag = "4")] + next_offset: u64, + #[prost(uint64, tag = "5")] + earliest_retained_offset: u64, + #[prost(bool, tag = "6")] + loss: bool, +} + +struct ExecJob { + host: Arc, + runtime: RuntimeRecord, + capability: [u8; 32], + config: OpenVmmConfig, + command: ExecuteCommandRequest, + timeout: Option, + io: ExecIo, +} + +impl NvxHostBackend { + /// Opens the state root and verifies the explicitly supplied native asset. + pub fn new(mut config: NvxHostConfig) -> Result { + config.openvmm = config.openvmm.normalized()?; + config.image = absolute(&config.image)?; + config.library = absolute(&config.library)?; + let host = HostLibrary::load(&config.library, &config.library_sha256)?; + let store_root = config.openvmm.state_root.join(BACKEND_KEY); + let store = StateStore::open_for(&store_root, BACKEND_KEY).map_err(|error| { + Error::backend_unavailable(format!( + "cannot open native sandbox state root {}", + store_root.display() + )) + .with_source(error) + })?; + let base = OpenVmmBackend { + config: config.openvmm.clone(), + store, + }; + Ok(Self { + config, + base, + host, + console_pumps: Mutex::new(HashMap::new()), + }) + } + + /// Returns the normalized configuration. + pub fn config(&self) -> &NvxHostConfig { + &self.config + } + + /// Returns the OpenVMM diagnostic log for this sandbox. + pub fn log_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.base.log_path(sandbox_id) + } + + /// Returns the guest boot-console log, including startup errors before ttrpc is ready. + pub fn console_log_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.base.store.console_path(sandbox_id) + } + + /// Returns OpenVMM's bounded outcome report path for the latest launch. + pub fn outcome_report_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.base.store.outcome_path(sandbox_id) + } + + /// Collects the guest's bounded, non-follow log snapshot through ttrpc. + pub fn guest_logs(&self, sandbox_id: &SandboxId) -> Result> { + let (runtime, capability) = self.running(sandbox_id)?; + let mut session = + self.connect(&runtime, &capability, self.config.openvmm.control_timeout)?; + let mut output = Vec::new(); + let mut offset = 0u64; + let request = StreamLogsRequest { + container_id: "guest".to_owned(), + offset, + follow: false, + max_chunk_bytes: 64 * 1024, + }; + let streamed = session.server_stream( + "StreamLogs", + &request.encode_to_vec(), + self.config.openvmm.control_timeout, + |bytes| { + let chunk = LogChunk::decode(bytes).map_err(|error| { + Error::backend_error("the guest returned an invalid log chunk") + .with_source(error) + })?; + let expected_next = offset + .checked_add(chunk.data.len() as u64) + .ok_or_else(|| Error::backend_error("guest log cursor overflowed"))?; + if chunk.container_id != "guest" + || chunk.loss + || chunk.offset != offset + || chunk.next_offset != expected_next + || chunk.earliest_retained_offset > offset + { + return Err(Error::backend_error( + "the guest returned a missing or out-of-order log chunk", + )); + } + if output + .len() + .checked_add(chunk.data.len()) + .is_none_or(|size| size > MAX_OUTPUT_BYTES) + { + return Err(Error::backend_error( + "guest logs exceed the one-megabyte collection limit", + )); + } + offset = chunk.next_offset; + output.extend_from_slice(&chunk.data); + Ok(()) + }, + ); + finish_session(session, streamed)?; + Ok(output) + } + + fn boot_endpoint(&self, sandbox_id: &SandboxId, control: &str) -> Result { + if cfg!(windows) { + let suffix = control + .strip_prefix("//./pipe/openvmm-microvm-") + .ok_or_else(|| { + Error::backend_error("the recorded control pipe is not canonical") + })?; + if suffix.is_empty() + || !suffix + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'.')) + { + return Err(Error::backend_error( + "the recorded control pipe has an invalid name", + )); + } + return Ok(format!("{control}-boot")); + } + self.base + .store + .boot_socket_path(sandbox_id) + .to_str() + .map(str::to_owned) + .ok_or_else(|| Error::backend_unavailable("the boot-console socket path is not UTF-8")) + } + + fn start_console_pump(&self, sandbox_id: &SandboxId, runtime: &RuntimeRecord) -> Result<()> { + let mut pumps = self + .console_pumps + .lock() + .map_err(|_| Error::backend_error("the boot-console pump table is unavailable"))?; + if pumps.contains_key(sandbox_id) { + return Ok(()); + } + let endpoint = self.boot_endpoint(sandbox_id, &runtime.endpoint)?; + let file = self.base.store.open_console_log(sandbox_id)?; + let runtime = runtime.clone(); + let timeout = self.config.openvmm.start_timeout; + let pump = thread::Builder::new() + .name("nvxhost-boot-console".to_owned()) + .spawn(move || pump_console(runtime, endpoint, file, timeout)) + .map_err(|error| { + Error::backend_error("cannot start the guest boot-console listener") + .with_source(error) + })?; + pumps.insert(sandbox_id.clone(), pump); + Ok(()) + } + + fn finish_console_pump(&self, sandbox_id: &SandboxId) -> Result<()> { + let pump = self + .console_pumps + .lock() + .map_err(|_| Error::backend_error("the boot-console pump table is unavailable"))? + .remove(sandbox_id); + if let Some(pump) = pump { + pump.join() + .map_err(|_| Error::backend_error("the boot-console listener panicked"))??; + } + Ok(()) + } + + fn artifact_record(&self) -> Result { + Ok(NativeArtifactRecord { + image: self.config.image.clone(), + image_sha256: file_digest(&self.config.image)?, + kernel_sha256: file_digest(&self.config.openvmm.kernel)?, + initrd_sha256: file_digest(&self.config.openvmm.initrd)?, + library_sha256: encode_digest(&self.config.library_sha256), + }) + } + + fn verify_record(&self, record: &SandboxRecord, for_launch: bool) -> Result<()> { + let recorded = record.native.as_ref().ok_or_else(|| { + Error::backend_error("the native sandbox has no pinned artifact identity") + })?; + if recorded.image != self.config.image + || recorded.library_sha256 != encode_digest(&self.config.library_sha256) + { + return Err(Error::backend_unavailable( + "the native sandbox's image or host library differs from its provisioned identity", + )); + } + if for_launch && *recorded != self.artifact_record()? { + return Err(Error::backend_unavailable( + "the native sandbox's boot artifacts changed since provision", + )); + } + Ok(()) + } + + fn running(&self, sandbox_id: &SandboxId) -> Result<(RuntimeRecord, [u8; 32])> { + let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; + self.verify_record(&record, false)?; + match self.base.reconcile(sandbox_id)? { + RunState::Provisioned => Err(Error::not_started(format!( + "sandbox {sandbox_id} is not running" + ))), + RunState::Running(runtime) => { + let capability = self.base.store.read_capability(sandbox_id)?; + self.start_console_pump(sandbox_id, &runtime)?; + Ok((runtime, capability)) + } + } + } + + fn connect( + &self, + runtime: &RuntimeRecord, + capability: &[u8; 32], + timeout: Duration, + ) -> Result { + let deadline = Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("control timeout is too long"))?; + Session::connect_verified( + Arc::clone(&self.host), + &runtime.endpoint, + capability, + runtime.pid, + deadline, + || { + platform::process_start_time(runtime.pid) + .map(|current| current == Some(runtime.start_time)) + .map_err(|error| { + Error::backend_error("cannot verify the running OpenVMM process") + .with_source(error) + }) + }, + ) + } + + fn wait_ready(&self, session: &mut Session, deadline: Instant) -> Result { + let remaining = || { + let timeout = deadline.saturating_duration_since(Instant::now()); + if timeout.is_zero() { + Err(Error::backend_error( + "the guest did not become ready before the start deadline", + )) + } else { + Ok(timeout) + } + }; + let info = GuestInfo::decode( + session + .unary("GetGuestInfo", &[], Some(remaining()?))? + .as_slice(), + ) + .map_err(|error| { + Error::backend_error("the guest returned malformed identity data").with_source(error) + })?; + if info.runtime_abi != RUNTIME_ABI || info.agent_build_id.is_empty() { + return Err(Error::backend_unavailable(format!( + "the guest reported runtime ABI {:?}, expected {RUNTIME_ABI}", + info.runtime_abi + ))); + } + if info.readiness == 3 || info.readiness == 2 { + return Err(Error::backend_error(format!( + "the guest cannot become ready: {}", + info.startup_error + ))); + } + let ready = WaitReadyResponse::decode( + session + .unary( + "WaitReady", + &WaitReadyRequest { + minimum_generation: info.readiness_generation, + } + .encode_to_vec(), + Some(remaining()?), + )? + .as_slice(), + ) + .map_err(|error| { + Error::backend_error("the guest returned malformed readiness data").with_source(error) + })?; + if ready.readiness != 1 || ready.generation < info.readiness_generation { + return Err(Error::backend_error(format!( + "the guest did not reach the required readiness generation: {}", + ready.startup_error + ))); + } + Ok(info.agent_build_id) + } +} + +impl Backend for NvxHostBackend { + fn name(&self) -> &str { + BACKEND_KEY + } + + fn capabilities(&self) -> Capabilities { + capabilities() + } + + fn probe(&self) -> Result<()> { + self.base.probe()?; + if !self.config.image.is_file() { + return Err(Error::backend_unavailable(format!( + "NVX image not found: {}", + self.config.image.display() + ))); + } + Ok(()) + } + + fn validate_provision(&self, request: &ProvisionRequest) -> Result<()> { + let memory = request + .microvm + .provision + .memory_mib + .unwrap_or(self.config.openvmm.memory_mib); + if memory == 0 || memory > i32::MAX as u32 { + return Err(Error::policy_validation( + "microvm.provision.memoryMib must fit a positive signed 32-bit integer", + )); + } + if request + .filesystem + .as_ref() + .is_some_and(|policy| !policy.is_empty()) + || request.network.as_ref().is_some_and(|policy| { + policy.egress.default != Access::Deny + || policy.ingress.default != Access::Deny + || policy.ingress.host_loopback == Some(Access::Allow) + || !policy.egress.allow.is_empty() + || !policy.egress.deny.is_empty() + }) + { + return Err(Error::policy_validation( + "the nvxhost backend currently supports no host filesystem or guest network", + )); + } + Ok(()) + } + + fn validate_exec(&self, request: &ExecRequest) -> Result<()> { + prepare_exec(request).map(|_| ()) + } + + fn provision(&self, request: &ProvisionRequest) -> Result { + self.validate_provision(request)?; + self.probe()?; + let record = SandboxRecord { + format: STATE_FORMAT, + backend: BACKEND_KEY.to_owned(), + network: None, + filesystem: None, + native: Some(self.artifact_record()?), + memory_mib: request + .microvm + .provision + .memory_mib + .unwrap_or(self.config.openvmm.memory_mib), + workload_uid: 65534, + workload_gid: 65534, + create_workload_account: false, + hostname: self.config.openvmm.hostname.clone(), + }; + let sandbox_id = SandboxId::generate()?; + self.base.store.create(&sandbox_id, &record)?; + Ok(ProvisionResult { + sandbox_id, + metadata: None, + }) + } + + fn start(&self, sandbox_id: &SandboxId) -> Result { + let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; + if let RunState::Running(_) = self.base.reconcile(sandbox_id)? { + return Err(Error::already_started(format!( + "sandbox {sandbox_id} is already running" + ))); + } + self.verify_record(&record, true)?; + self.probe()?; + + let mut capability = [0u8; 32]; + while capability == [0; 32] { + getrandom::fill(&mut capability).map_err(|error| { + Error::backend_error("cannot generate a broker capability").with_source(error) + })?; + } + let socket = self.base.store.socket_path(sandbox_id); + let boot_socket = self.base.store.boot_socket_path(sandbox_id); + remove_if_present(&socket)?; + remove_if_present(&boot_socket)?; + let endpoint = platform::control_endpoint(&socket).map_err(|error| { + Error::backend_error("cannot choose a control endpoint").with_source(error) + })?; + let boot = self.boot_endpoint(sandbox_id, &endpoint)?; + let mut arguments = self.host.launch_arguments(&LaunchInputs { + kernel: &self.config.openvmm.kernel, + initrd: &self.config.openvmm.initrd, + image: &self.config.image, + control: &endpoint, + boot: &boot, + hypervisor: self.config.openvmm.hypervisor.as_str(), + memory_mb: record.memory_mib, + guest_debug: self.config.guest_debug, + })?; + let report = self.base.store.outcome_path(sandbox_id); + remove_if_present(&report)?; + arguments.push("--microvm-report".into()); + arguments.push(report.into_os_string()); + + self.base.store.write_capability(sandbox_id, &capability)?; + let log = self.base.store.create_log(sandbox_id)?; + self.base.store.write_launch( + sandbox_id, + &LaunchRecord { + format: STATE_FORMAT, + endpoint: endpoint.clone(), + process: None, + }, + )?; + let started = Instant::now(); + let child = match process::spawn( + &self.config.openvmm, + &arguments, + &capability, + log, + &self.base.store.dir(sandbox_id), + ) { + Ok(child) => child, + Err(error) => { + self.base.store.clear_runtime(sandbox_id)?; + return Err(Error::backend_error("cannot launch OpenVMM").with_source(error)); + } + }; + let pid = child.id(); + let start_time = match platform::process_start_time(pid) { + Ok(Some(value)) => value, + other => { + if process::kill_child(child) { + self.base.store.clear_runtime(sandbox_id)?; + } + return Err(self.base.start_failure( + sandbox_id, + &format!("cannot identify OpenVMM process {pid}: {other:?}"), + )); + } + }; + let runtime = RuntimeRecord { + format: STATE_FORMAT, + pid, + start_time, + endpoint: endpoint.clone(), + }; + if let Err(error) = self + .base + .store + .write_launch( + sandbox_id, + &LaunchRecord { + format: STATE_FORMAT, + endpoint, + process: Some(ProcessIdentity { pid, start_time }), + }, + ) + .and_then(|()| self.base.store.write_runtime(sandbox_id, &runtime)) + { + if process::kill_child(child) { + self.base.store.clear_runtime(sandbox_id)?; + } + return Err(error); + } + process::detach_child(child); + self.base.store.remove_launch(sandbox_id); + if let Err(error) = self.start_console_pump(sandbox_id, &runtime) { + return Err(self + .base + .abort_start(sandbox_id, &runtime, &error.to_string())); + } + let ready = (|| { + let deadline = started + .checked_add(self.config.openvmm.start_timeout) + .ok_or_else(|| Error::backend_error("start timeout is too long"))?; + let mut session = self.connect( + &runtime, + &capability, + deadline.saturating_duration_since(Instant::now()), + )?; + let result = self.wait_ready(&mut session, deadline); + finish_session(session, result) + })(); + let agent = match ready { + Ok(agent) => agent, + Err(error) => { + let failure = self + .base + .abort_start(sandbox_id, &runtime, &error.to_string()); + match platform::process_start_time(runtime.pid) { + Ok(Some(current)) if current == runtime.start_time => return Err(failure), + Ok(_) => {} + Err(check) => { + return Err(Error::backend_error(format!( + "{failure}; cannot verify boot-console cleanup: {check}" + ))); + } + } + return match self.finish_console_pump(sandbox_id) { + Ok(()) => Err(failure), + Err(console) => Err(Error::backend_error(format!( + "{failure}; boot-console listener also failed: {console}" + ))), + }; + } + }; + let mut metadata = Metadata::new(); + let boot_ms = u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX); + metadata.insert("bootMilliseconds".to_owned(), boot_ms.into()); + metadata.insert("guestBuildId".to_owned(), agent.into()); + Ok(StartResult { + metadata: Some(metadata), + }) + } + + fn exec( + &self, + sandbox_id: &SandboxId, + request: &ExecRequest, + io: ExecIo, + ) -> Result> { + if request.stdin == StdinMode::Piped || io.stdin.is_some() { + return Err(Error::policy_validation( + "the nvxhost backend does not support piped stdin", + )); + } + let (message, timeout) = prepare_exec(request)?; + let (runtime, capability) = self.running(sandbox_id)?; + let shared = Arc::new(Completion::default()); + let result = Arc::clone(&shared); + let job = ExecJob { + host: Arc::clone(&self.host), + runtime, + capability, + config: self.config.openvmm.clone(), + command: message, + timeout, + io, + }; + thread::Builder::new() + .name("nvxhost-exec".to_owned()) + .spawn(move || { + let outcome = execute(job); + result.finish(outcome); + }) + .map_err(|error| { + Error::backend_error("cannot start an execution thread").with_source(error) + })?; + Ok(Box::new(NvxHostExecution { shared })) + } + + fn stop(&self, sandbox_id: &SandboxId) -> Result { + let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; + self.verify_record(&record, false)?; + let RunState::Running(runtime) = self.base.reconcile(sandbox_id)? else { + return Err(Error::already_stopped(format!( + "sandbox {sandbox_id} is not running" + ))); + }; + let deadline = Instant::now() + self.config.openvmm.stop_timeout; + let graceful = (|| { + let capability = self.base.store.read_capability(sandbox_id)?; + let remaining = deadline.saturating_duration_since(Instant::now()); + let mut session = self.connect(&runtime, &capability, remaining)?; + let grace_period_milliseconds = i64::try_from(remaining.as_millis()) + .map_err(|_| Error::backend_error("guest shutdown grace period is too long"))?; + let response = session.unary( + "Shutdown", + &ShutdownRequest { + grace_period_milliseconds, + } + .encode_to_vec(), + Some(remaining), + ); + let response = finish_session(session, response)?; + ShutdownResponse::decode(response.as_slice()).map_err(|error| { + Error::backend_error("the guest returned an invalid shutdown response") + .with_source(error) + })?; + if process::wait_for_exit(runtime.pid, runtime.start_time, deadline) { + Ok(()) + } else { + Err(Error::backend_error("guest shutdown timed out")) + } + })(); + let needs_force = graceful.is_err(); + let forced = needs_force + && platform::process_start_time(runtime.pid).map_err(|error| { + Error::backend_error("cannot check OpenVMM before forced stop").with_source(error) + })? == Some(runtime.start_time); + if forced + && !process::kill(runtime.pid, runtime.start_time).map_err(|error| { + Error::backend_error(format!("cannot terminate OpenVMM process {}", runtime.pid)) + .with_source(error) + })? + { + return Err(Error::backend_error(format!( + "OpenVMM process {} did not terminate", + runtime.pid + ))); + } + let mut metadata = Metadata::new(); + metadata.insert("forced".to_owned(), forced.into()); + if let Err(error) = graceful { + metadata.insert("gracefulError".to_owned(), error.to_string().into()); + } + let console = self.finish_console_pump(sandbox_id); + self.base.store.clear_runtime(sandbox_id)?; + console?; + Ok(StopResult { + metadata: Some(metadata), + }) + } + + fn deprovision(&self, sandbox_id: &SandboxId) -> Result { + let (guard, record) = self.base.store.lock_and_load(sandbox_id)?; + self.verify_record(&record, false)?; + if let RunState::Running(_) = self.base.reconcile(sandbox_id)? { + return Err(Error::already_started(format!( + "sandbox {sandbox_id} is running; stop it before deprovisioning" + ))); + } + self.finish_console_pump(sandbox_id)?; + self.base.store.remove(sandbox_id)?; + drop(guard); + self.base.store.remove_lock(sandbox_id); + Ok(DeprovisionResult::default()) + } +} + +fn capabilities() -> Capabilities { + let mut capabilities = Capabilities::new(BACKEND_KEY); + capabilities.exec.command_line = true; + capabilities.exec.argv = true; + capabilities.exec.max_timeout_ms = Some(MAX_EXEC_SECONDS * 1_000); + capabilities.exec.max_output_bytes = Some(MAX_OUTPUT_BYTES as u64); + capabilities.network.egress_deny = true; + capabilities.network.ingress_deny = true; + capabilities.network.host_loopback_deny = true; + capabilities +} + +fn encode_digest(bytes: &[u8; 32]) -> String { + bytes.iter().map(|byte| format!("{byte:02x}")).collect() +} + +fn file_digest(path: &Path) -> Result { + let mut file = File::open(path).map_err(|error| { + Error::backend_unavailable(format!("cannot open {}", path.display())).with_source(error) + })?; + let mut hasher = Sha256::new(); + io::copy(&mut file, &mut hasher).map_err(|error| { + Error::backend_unavailable(format!("cannot hash {}", path.display())).with_source(error) + })?; + Ok(format!("{:x}", hasher.finalize())) +} + +fn finish_session(session: Session, result: Result) -> Result { + match (result, session.close()) { + (Ok(value), Ok(())) => Ok(value), + (Err(error), Ok(())) | (Ok(_), Err(error)) => Err(error), + (Err(error), Err(close)) => Err(Error::backend_error(format!( + "{error}; closing the guest session also failed: {close}" + ))), + } +} + +fn pump_console( + runtime: RuntimeRecord, + endpoint: String, + mut output: File, + timeout: Duration, +) -> Result<()> { + let deadline = Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("boot-console timeout is too long"))?; + let mut console = loop { + if Instant::now() >= deadline { + return Err(Error::backend_error( + "OpenVMM did not open its guest boot-console endpoint", + )); + } + match platform::connect_endpoint(&endpoint, runtime.pid, Duration::from_millis(250)) { + Ok(console) => break console, + Err(error) + if matches!( + error.kind(), + io::ErrorKind::NotFound + | io::ErrorKind::ConnectionRefused + | io::ErrorKind::WouldBlock + | io::ErrorKind::TimedOut + ) => + { + if platform::process_start_time(runtime.pid).map_err(|error| { + Error::backend_error("cannot verify OpenVMM while connecting the console") + .with_source(error) + })? != Some(runtime.start_time) + { + return Err(Error::backend_error( + "OpenVMM exited before the guest boot console could connect", + )); + } + thread::sleep(Duration::from_millis(25)); + } + Err(error) => { + return Err( + Error::backend_error("cannot connect the guest boot console") + .with_source(error), + ); + } + } + }; + let mut buffer = [0u8; 4096]; + loop { + match console.read(&mut buffer, Some(Duration::from_millis(250))) { + Ok(0) => break, + Ok(size) => output.write_all(&buffer[..size]).map_err(|error| { + Error::backend_error("cannot write the guest boot-console log").with_source(error) + })?, + Err(error) if error.kind() == io::ErrorKind::TimedOut => { + if platform::process_start_time(runtime.pid).map_err(|error| { + Error::backend_error("cannot verify OpenVMM while reading the console") + .with_source(error) + })? != Some(runtime.start_time) + { + break; + } + } + Err(error) => { + return Err( + Error::backend_error("cannot read the guest boot console").with_source(error) + ); + } + } + } + output.sync_all().map_err(|error| { + Error::backend_error("cannot flush the guest boot-console log").with_source(error) + }) +} + +fn prepare_exec(request: &ExecRequest) -> Result<(ExecuteCommandRequest, Option)> { + if request.stdin == StdinMode::Piped + || request.process.cwd.is_some() + || request.process.env.is_some() + { + return Err(Error::policy_validation( + "the nvxhost backend supports no piped stdin, custom working directory, or environment", + )); + } + let argv = match &request.process.command { + Command::CommandLine(line) => vec![SHELL.to_owned(), "-c".to_owned(), line.clone()], + Command::Argv(argv) => argv.clone(), + }; + if argv.is_empty() + || argv.iter().any(|argument| argument.contains('\0')) + || argv.iter().map(String::len).sum::() > 4096 + { + return Err(Error::policy_validation( + "the guest command must contain non-NUL arguments totaling at most 4096 bytes", + )); + } + let requested_ms = request.process.timeout.map(duration_millis).unwrap_or(0); + if requested_ms > MAX_EXEC_SECONDS * 1_000 { + return Err(Error::policy_validation( + "the guest command timeout exceeds one hour", + )); + } + let seconds = if requested_ms == 0 { + 0 + } else { + requested_ms.div_ceil(1_000) + }; + let timeout = (seconds != 0).then(|| Duration::from_secs(seconds)); + Ok(( + ExecuteCommandRequest { + command: argv[0].clone(), + args: argv[1..].to_vec(), + timeout_seconds: seconds as i32, + }, + timeout, + )) +} + +fn execute(job: ExecJob) -> Result { + let ExecJob { + host, + runtime, + capability, + config, + command, + timeout, + mut io, + } = job; + let mut session = Session::connect_verified( + host, + &runtime.endpoint, + &capability, + runtime.pid, + Instant::now() + config.control_timeout, + || { + platform::process_start_time(runtime.pid) + .map(|current| current == Some(runtime.start_time)) + .map_err(|error| { + Error::backend_error("cannot verify the running OpenVMM process") + .with_source(error) + }) + }, + )?; + let response = session.unary( + "ExecuteCommand", + &command.encode_to_vec(), + timeout.map(|timeout| timeout + config.exec_response_grace), + ); + let response = finish_session(session, response)?; + let reply = ExecuteCommandResponse::decode(response.as_slice()).map_err(|error| { + Error::backend_error("the guest returned an invalid execution result").with_source(error) + })?; + let _ = io.stdout.write(reply.stdout.as_bytes()); + let _ = io.stderr.write(reply.stderr.as_bytes()); + drop(io); + if reply.timed_out { + Ok(ExecOutcome::TimedOut) + } else { + Ok(ExecOutcome::Exited(reply.exit_code)) + } +} + +struct NvxHostExecution { + shared: Arc, +} + +impl ExecControl for NvxHostExecution { + fn wait(&self) -> Result { + self.shared.wait() + } + + fn cancel(&self) -> Result<()> { + Err(Error::unsupported( + "the nvxhost backend does not support canceling an active guest command", + )) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ErrorCode; + + #[test] + fn simple_exec_has_no_container_metadata_and_respects_timeout() { + let (request, timeout) = prepare_exec( + &ExecRequest::command_line("printf READY").with_timeout(Duration::from_millis(1500)), + ) + .unwrap(); + assert_eq!(request.command, SHELL); + assert_eq!(request.args, ["-c", "printf READY"]); + assert_eq!(request.timeout_seconds, 2); + assert_eq!(timeout, Some(Duration::from_secs(2))); + assert!(request.encode_to_vec().len() < 100); + assert!(capabilities().exec.command_line); + assert!(!capabilities().exec.cancel); + assert!(!capabilities().filesystem.readonly_paths); + } + + #[test] + fn unsupported_exec_options_fail_before_guest_work() { + let request = ExecRequest::command_line("echo READY").with_cwd("/tmp"); + assert_eq!( + prepare_exec(&request).unwrap_err().code(), + ErrorCode::PolicyValidation + ); + let request = ExecRequest::command_line("echo READY").with_env("K=V"); + assert_eq!( + prepare_exec(&request).unwrap_err().code(), + ErrorCode::PolicyValidation + ); + } +} diff --git a/aci_edge_sandboxes/src/openvmm/state.rs b/aci_edge_sandboxes/src/openvmm/state.rs index 064ef51bb..9a4ed434f 100644 --- a/aci_edge_sandboxes/src/openvmm/state.rs +++ b/aci_edge_sandboxes/src/openvmm/state.rs @@ -23,6 +23,8 @@ use serde::{Deserialize, Serialize}; use super::filesystem::HostMapping; use super::platform; use super::protocol::CAPABILITY_LEN; +#[cfg(any(test, feature = "nvxhost"))] +use crate::error::ErrorCode; use crate::error::{Error, Result}; use crate::id::SandboxId; use crate::model::NetworkPolicy; @@ -35,12 +37,15 @@ pub(crate) const STATE_FORMAT: u32 = 2; pub(crate) const BACKEND_KEY: &str = "openvmm"; /// Linux control socket name. pub(crate) const SOCKET_NAME: &str = "control.sock"; +/// Linux boot-console socket name. +pub(crate) const BOOT_SOCKET_NAME: &str = "boot.sock"; const RECORD_NAME: &str = "sandbox.json"; const LAUNCH_NAME: &str = "launch.json"; const RUNTIME_NAME: &str = "runtime.json"; const CAPABILITY_NAME: &str = "control.capability"; const LOG_NAME: &str = "openvmm.log"; +const CONSOLE_LOG_NAME: &str = "console.log"; const OUTCOME_NAME: &str = "outcome.json"; const LOCKS_NAME: &str = ".locks"; @@ -54,6 +59,8 @@ pub(crate) struct SandboxRecord { pub(crate) network: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub(crate) filesystem: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(crate) native: Option, pub(crate) memory_mib: u32, pub(crate) workload_uid: u32, pub(crate) workload_gid: u32, @@ -64,6 +71,16 @@ pub(crate) struct SandboxRecord { pub(crate) hostname: String, } +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub(crate) struct NativeArtifactRecord { + pub(crate) image: PathBuf, + pub(crate) image_sha256: String, + pub(crate) kernel_sha256: String, + pub(crate) initrd_sha256: String, + pub(crate) library_sha256: String, +} + /// Identity of the OpenVMM process of a running sandbox. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "camelCase", deny_unknown_fields)] @@ -104,6 +121,7 @@ pub(crate) struct LockGuard { #[derive(Debug)] pub(crate) struct StateStore { root: PathBuf, + backend: &'static str, } fn io_error(context: String) -> impl FnOnce(io::Error) -> Error { @@ -113,6 +131,10 @@ fn io_error(context: String) -> impl FnOnce(io::Error) -> Error { impl StateStore { /// Opens `root`, creating it and restricting it to the current user if needed. pub(crate) fn open(root: &Path) -> io::Result { + Self::open_for(root, BACKEND_KEY) + } + + pub(crate) fn open_for(root: &Path, backend: &'static str) -> io::Result { if let Some(parent) = root.parent() { fs::create_dir_all(parent)?; } @@ -120,6 +142,7 @@ impl StateStore { platform::create_private_dir(&root.join(LOCKS_NAME))?; Ok(Self { root: root.to_path_buf(), + backend, }) } @@ -131,6 +154,11 @@ impl StateStore { self.dir(sandbox_id).join(LOG_NAME) } + #[cfg(feature = "nvxhost")] + pub(crate) fn console_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.dir(sandbox_id).join(CONSOLE_LOG_NAME) + } + pub(crate) fn outcome_path(&self, sandbox_id: &SandboxId) -> PathBuf { self.dir(sandbox_id).join(OUTCOME_NAME) } @@ -139,6 +167,11 @@ impl StateStore { self.dir(sandbox_id).join(SOCKET_NAME) } + #[cfg(feature = "nvxhost")] + pub(crate) fn boot_socket_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.dir(sandbox_id).join(BOOT_SOCKET_NAME) + } + fn lock_path(&self, sandbox_id: &SandboxId) -> PathBuf { self.root .join(LOCKS_NAME) @@ -150,6 +183,34 @@ impl StateStore { lock(&self.lock_path(sandbox_id)) } + #[cfg(any(test, feature = "nvxhost"))] + pub(crate) fn lock_and_load( + &self, + sandbox_id: &SandboxId, + ) -> Result<(LockGuard, SandboxRecord)> { + match self.dir(sandbox_id).try_exists() { + Ok(false) => { + return Err(Error::stale_id(format!( + "sandbox {sandbox_id} is not provisioned" + ))); + } + Err(error) => { + return Err(Error::backend_error("cannot inspect sandbox state").with_source(error)); + } + Ok(true) => {} + } + let guard = self.lock(sandbox_id)?; + match self.load(sandbox_id) { + Ok(record) => Ok((guard, record)), + Err(error) if error.code() == ErrorCode::StaleId => { + drop(guard); + self.remove_lock(sandbox_id); + Err(error) + } + Err(error) => Err(error), + } + } + /// Removes the lock file of a deprovisioned sandbox. pub(crate) fn remove_lock(&self, sandbox_id: &SandboxId) { let _ = fs::remove_file(self.lock_path(sandbox_id)); @@ -157,6 +218,11 @@ impl StateStore { /// Creates the state directory of a new sandbox. pub(crate) fn create(&self, sandbox_id: &SandboxId, record: &SandboxRecord) -> Result<()> { + if record.backend != self.backend { + return Err(Error::backend_error( + "the sandbox record does not belong to this backend", + )); + } let dir = self.dir(sandbox_id); fs::create_dir(&dir).map_err(io_error(format!( "cannot create sandbox state {}", @@ -185,7 +251,7 @@ impl StateStore { ))); } }; - if record.format != STATE_FORMAT || record.backend != BACKEND_KEY { + if record.format != STATE_FORMAT || record.backend != self.backend { return Err(Error::backend_error(format!( "sandbox state {} has an unsupported format", path.display() @@ -263,7 +329,13 @@ impl StateStore { /// Removes the files that exist only while the sandbox starts or runs. pub(crate) fn clear_runtime(&self, sandbox_id: &SandboxId) -> Result<()> { let dir = self.dir(sandbox_id); - for name in [RUNTIME_NAME, LAUNCH_NAME, CAPABILITY_NAME, SOCKET_NAME] { + for name in [ + RUNTIME_NAME, + LAUNCH_NAME, + CAPABILITY_NAME, + SOCKET_NAME, + BOOT_SOCKET_NAME, + ] { remove_if_present(&dir.join(name))?; } Ok(()) @@ -278,6 +350,15 @@ impl StateStore { .map_err(io_error(format!("cannot create {}", path.display()))) } + #[cfg(feature = "nvxhost")] + pub(crate) fn open_console_log(&self, sandbox_id: &SandboxId) -> Result { + let path = self.console_path(sandbox_id); + private_options() + .append(true) + .open(&path) + .map_err(io_error(format!("cannot open {}", path.display()))) + } + /// Deletes the state of a stopped sandbox. /// /// Unknown files abort the removal before anything is deleted. The configuration is removed @@ -293,8 +374,10 @@ impl StateStore { LAUNCH_NAME, CAPABILITY_NAME, SOCKET_NAME, + BOOT_SOCKET_NAME, OUTCOME_NAME, LOG_NAME, + CONSOLE_LOG_NAME, ]; let mut owned = Vec::new(); for entry in entries { @@ -408,6 +491,7 @@ mod tests { backend: BACKEND_KEY.to_owned(), network: None, filesystem: None, + native: None, memory_mib: 256, workload_uid: 65534, workload_gid: 65534, @@ -422,11 +506,50 @@ mod tests { let store = StateStore::open(&root.path().join("sandboxes")).unwrap(); let id = SandboxId::generate().unwrap(); assert_eq!(store.load(&id).unwrap_err().code(), ErrorCode::StaleId); + assert_eq!( + store.lock_and_load(&id).unwrap_err().code(), + ErrorCode::StaleId + ); + assert!(!store.dir(&id).exists()); + assert!(!store.lock_path(&id).exists()); store.create(&id, &record()).unwrap(); assert_eq!(store.load(&id).unwrap(), record()); assert!(store.runtime(&id).unwrap().is_none()); } + #[test] + fn native_records_round_trip_only_through_their_backend() { + let root = tempfile::tempdir().unwrap(); + let store = StateStore::open_for(root.path(), "nvxhost").unwrap(); + let id = SandboxId::generate().unwrap(); + assert_eq!( + store.create(&id, &record()).unwrap_err().code(), + ErrorCode::BackendError + ); + assert!(!store.dir(&id).exists()); + + let mut native = record(); + native.backend = "nvxhost".to_owned(); + native.native = Some(NativeArtifactRecord { + image: PathBuf::from("image.vhd"), + image_sha256: "1".repeat(64), + kernel_sha256: "2".repeat(64), + initrd_sha256: "3".repeat(64), + library_sha256: "4".repeat(64), + }); + store.create(&id, &native).unwrap(); + let (_guard, loaded) = store.lock_and_load(&id).unwrap(); + assert_eq!(loaded, native); + assert_eq!( + StateStore::open(root.path()) + .unwrap() + .load(&id) + .unwrap_err() + .code(), + ErrorCode::BackendError + ); + } + #[test] fn runtime_files_are_cleared_and_state_removed() { let root = tempfile::tempdir().unwrap(); diff --git a/aci_edge_sandboxes/tests/nvxhost_guest.rs b/aci_edge_sandboxes/tests/nvxhost_guest.rs new file mode 100644 index 000000000..b6601ac35 --- /dev/null +++ b/aci_edge_sandboxes/tests/nvxhost_guest.rs @@ -0,0 +1,150 @@ +//! Opt-in WHP lifecycle proof with a privately supplied native library and guest. +#![cfg(feature = "nvxhost")] + +use std::path::PathBuf; +use std::sync::Arc; + +use aci_edge_sandboxes::openvmm::{Hypervisor, NvxHostBackend, NvxHostConfig, OpenVmmConfig}; +use aci_edge_sandboxes::{ + AciEdgeSandbox, Error, ErrorCode, ExecOutcome, ExecRequest, ProvisionRequest, +}; + +fn required(name: &str) -> PathBuf { + PathBuf::from(std::env::var_os(name).unwrap_or_else(|| panic!("{name} must be set"))) +} + +fn approved_digest() -> [u8; 32] { + let hex = std::env::var("NVXHOST_TEST_SHA256").expect("NVXHOST_TEST_SHA256 must be set"); + assert_eq!(hex.len(), 64, "the approved digest must contain 64 digits"); + let mut digest = [0u8; 32]; + for (index, value) in digest.iter_mut().enumerate() { + *value = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16) + .expect("the approved digest must be hexadecimal"); + } + digest +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_guest_lifecycle_stops_gracefully_and_cleans_up() { + let state = tempfile::tempdir().unwrap(); + let config = OpenVmmConfig::new( + required("NVXHOST_TEST_OPENVMM"), + required("NVXHOST_TEST_KERNEL"), + required("NVXHOST_TEST_INITRD"), + Hypervisor::Whp, + state.path(), + ); + let backend = Arc::new( + NvxHostBackend::new( + NvxHostConfig::new( + config, + required("NVXHOST_TEST_IMAGE"), + required("NVXHOST_TEST_LIBRARY"), + approved_digest(), + ) + .with_guest_debug(true), + ) + .unwrap(), + ); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox_id = client + .provision(&ProvisionRequest::new()) + .unwrap() + .sandbox_id; + let result = (|| { + let started = client.start(&sandbox_id)?; + if !started + .metadata + .as_ref() + .is_some_and(|metadata| metadata.contains_key("guestBuildId")) + { + return Err(Error::new( + ErrorCode::BackendError, + "the guest did not report a build ID", + )); + } + let executed = client + .exec(&sandbox_id, &ExecRequest::command_line("printf READY"))? + .wait_with_output()?; + if executed.outcome != ExecOutcome::Exited(0) || executed.stdout != b"READY" { + return Err(Error::new( + ErrorCode::BackendError, + format!( + "guest exec outcome {:?}, stdout {:?}, stderr {:?}", + executed.outcome, + String::from_utf8_lossy(&executed.stdout), + String::from_utf8_lossy(&executed.stderr), + ), + )); + } + let logs = backend.guest_logs(&sandbox_id)?; + if !logs.windows(8).any(|window| window == b"execute:") { + return Err(Error::new( + ErrorCode::BackendError, + "guest log stream omitted the executed command", + )); + } + let stopped = client.stop(&sandbox_id)?; + if stopped + .metadata + .as_ref() + .and_then(|metadata| metadata.get("forced")) + .and_then(serde_json::Value::as_bool) + != Some(false) + { + return Err(Error::new( + ErrorCode::BackendError, + format!("guest shutdown was not graceful: {:?}", stopped.metadata), + )); + } + let console = std::fs::read(backend.console_log_path(&sandbox_id)).map_err(|error| { + Error::new( + ErrorCode::BackendError, + "cannot read the guest boot console", + ) + .with_source(error) + })?; + if [b"NVX-EDGE-FATAL".as_slice(), b"NVX-EDGE-SHUTDOWN-ERROR"] + .iter() + .any(|marker| { + console + .windows(marker.len()) + .any(|window| window == *marker) + }) + { + return Err(Error::new( + ErrorCode::BackendError, + "the guest reported a fatal or shutdown error", + )); + } + client.deprovision(&sandbox_id)?; + if state + .path() + .join("nvxhost") + .join(sandbox_id.token()) + .exists() + { + return Err(Error::new( + ErrorCode::BackendError, + "deprovision left sandbox state behind", + )); + } + Ok::<_, aci_edge_sandboxes::Error>(()) + })(); + if let Err(error) = result { + let log = std::fs::read_to_string(backend.log_path(&sandbox_id)) + .unwrap_or_else(|read| format!("cannot read OpenVMM diagnostic log: {read}")); + let console = std::fs::read_to_string(backend.console_log_path(&sandbox_id)) + .unwrap_or_else(|read| format!("cannot read guest boot console: {read}")); + let outcome = std::fs::read_to_string(backend.outcome_report_path(&sandbox_id)) + .unwrap_or_else(|read| format!("cannot read OpenVMM outcome report: {read}")); + let stop = client.stop(&sandbox_id); + let deprovision = client.deprovision(&sandbox_id); + panic!( + "WHP lifecycle failed: {error}; recovery stop: {stop:?}; \ + recovery deprovision: {deprovision:?}; OpenVMM log: {log}; \ + guest console: {console}; OpenVMM outcome: {outcome}" + ); + } +} diff --git a/openvmm b/openvmm index 4355c010e..952406552 160000 --- a/openvmm +++ b/openvmm @@ -1 +1 @@ -Subproject commit 4355c010e726128c751138e7f74ad969e76a4c19 +Subproject commit 9524065522764c8b80b3a4d48ec7877ddc16090b From e17cfaeeb0b8d9b3a7a90e0b6d3825dd3ceca87d Mon Sep 17 00:00:00 2001 From: Enrique Saurez Date: Mon, 5 Oct 2026 13:07:22 -0700 Subject: [PATCH 2/5] Preserve native deadlines and backend state isolation Bound guest-session disposal by each operation deadline, preserve zero as an unbounded command timeout while rejecting sub-second precision, keep native-only files outside direct-backend cleanup, and validate both derived Unix socket paths before loading the library. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- aci_edge_sandboxes/README.md | 4 +- aci_edge_sandboxes/src/nvxhost.rs | 35 ++++-- aci_edge_sandboxes/src/openvmm/config.rs | 32 +++--- aci_edge_sandboxes/src/openvmm/native.rs | 130 +++++++++++++++++------ aci_edge_sandboxes/src/openvmm/state.rs | 48 ++++++--- 5 files changed, 185 insertions(+), 64 deletions(-) diff --git a/aci_edge_sandboxes/README.md b/aci_edge_sandboxes/README.md index 88a80fbac..d3ef0f54f 100644 --- a/aci_edge_sandboxes/README.md +++ b/aci_edge_sandboxes/README.md @@ -132,7 +132,9 @@ converted to the direct backend. `NvxHostBackend::guest_logs` reads a bounded no guest-log snapshot before stop. This first backend supports provision/start/exec/stop/deprovision with shell commands or -argv and a fixed caller-provided image. It explicitly rejects host file mappings, network +argv and a fixed caller-provided image. Positive execution timeouts must be whole seconds, +matching the guest RPC's precision; finer-grained timeouts fail validation rather than +silently extending execution. It explicitly rejects host file mappings, network configuration beyond deny-all, piped stdin, execution cancellation, custom working directories and environments. Snapshot/restore, image selection, and richer guest operations are not part of this profile. See diff --git a/aci_edge_sandboxes/src/nvxhost.rs b/aci_edge_sandboxes/src/nvxhost.rs index c052f10fa..e3c0593de 100644 --- a/aci_edge_sandboxes/src/nvxhost.rs +++ b/aci_edge_sandboxes/src/nvxhost.rs @@ -456,6 +456,11 @@ pub(crate) struct Session { next_token: u64, } +fn disposal_deadline(operation: Option, now: Instant) -> Instant { + let maximum = now + Duration::from_secs(5); + operation.map_or(maximum, |deadline| deadline.min(maximum)) +} + impl Session { pub(crate) fn connect_verified( host: Arc, @@ -773,16 +778,19 @@ impl Session { Ok(()) } - pub(crate) fn close(mut self) -> Result<()> { + pub(crate) fn close(mut self, deadline: Option) -> Result<()> { + let now = Instant::now(); + let deadline = disposal_deadline(deadline, now); + if now >= deadline { + return Err(Error::backend_error( + "guest session disposal exceeded the operation deadline", + )); + } let token = self.token()?; let status = unsafe { (self.host.exports.session_dispose)(self.session, token) }; self.host .check_status("closing a guest session", status, ptr::null_mut())?; - let disposed = self.next( - COMPLETION_DISPOSE, - token, - Some(Instant::now() + Duration::from_secs(5)), - )?; + let disposed = self.next(COMPLETION_DISPOSE, token, Some(deadline))?; self.completed_ok("closing a guest session", &disposed)?; let status = unsafe { (self.host.exports.session_release)(self.session) }; self.host @@ -827,6 +835,21 @@ impl Drop for Call { mod tests { use super::*; + #[test] + fn disposal_never_extends_a_short_operation_deadline() { + let now = Instant::now(); + let short = now + Duration::from_millis(100); + assert_eq!(disposal_deadline(Some(short), now), short); + let maximum = now + Duration::from_secs(5); + assert_eq!( + disposal_deadline(Some(now + Duration::from_secs(60)), now), + maximum + ); + assert_eq!(disposal_deadline(None, now), maximum); + let expired = now - Duration::from_millis(1); + assert_eq!(disposal_deadline(Some(expired), now), expired); + } + #[test] fn ttrpc_response_decodes_the_existing_wire_vector() { let bytes = [0x0a, 0x00, 0x12, 0x06, 0x7a, 0x04, b'r', b'u', b's', b't']; diff --git a/aci_edge_sandboxes/src/openvmm/config.rs b/aci_edge_sandboxes/src/openvmm/config.rs index 5cbd5253b..61536dc87 100644 --- a/aci_edge_sandboxes/src/openvmm/config.rs +++ b/aci_edge_sandboxes/src/openvmm/config.rs @@ -349,23 +349,29 @@ impl OpenVmmConfig { )); } } - if cfg!(unix) { - // sockaddr_un holds 108 bytes including the terminating NUL. - let socket = self - .state_root - .join("0".repeat(32)) - .join(super::state::SOCKET_NAME); - if socket.as_os_str().len() >= 108 { - return invalid(format!( - "state_root {} is too long for a Unix control socket path", - self.state_root.display() - )); - } - } + validate_unix_socket_path(&self.state_root, super::state::SOCKET_NAME, "control")?; Ok(()) } } +pub(crate) fn validate_unix_socket_path( + state_root: &Path, + socket_name: &str, + purpose: &str, +) -> Result<()> { + if cfg!(unix) { + // sockaddr_un holds 108 bytes including the terminating NUL. + let socket = state_root.join("0".repeat(32)).join(socket_name); + if socket.as_os_str().len() >= 108 { + return Err(Error::backend_unavailable(format!( + "state_root {} is too long for a Unix {purpose} socket path", + state_root.display() + ))); + } + } + Ok(()) +} + fn valid_hostname(hostname: &str) -> bool { let bytes = hostname.as_bytes(); !bytes.is_empty() diff --git a/aci_edge_sandboxes/src/openvmm/native.rs b/aci_edge_sandboxes/src/openvmm/native.rs index 4dcd37729..29be94220 100644 --- a/aci_edge_sandboxes/src/openvmm/native.rs +++ b/aci_edge_sandboxes/src/openvmm/native.rs @@ -12,12 +12,12 @@ use prost::Message; use sha2_runtime::{Digest, Sha256}; use super::artifacts::absolute; -use super::config::OpenVmmConfig; +use super::config::{OpenVmmConfig, validate_unix_socket_path}; use super::platform; use super::process; use super::state::{ - LaunchRecord, NativeArtifactRecord, ProcessIdentity, RuntimeRecord, STATE_FORMAT, - SandboxRecord, StateStore, remove_if_present, + BOOT_SOCKET_NAME, LaunchRecord, NATIVE_BACKEND_KEY, NativeArtifactRecord, ProcessIdentity, + RuntimeRecord, SOCKET_NAME, STATE_FORMAT, SandboxRecord, StateStore, remove_if_present, }; use super::{OpenVmmBackend, RunState}; use crate::backend::{Backend, ExecControl, ExecIo}; @@ -27,11 +27,11 @@ use crate::exec::{Completion, ExecOutcome}; use crate::id::SandboxId; use crate::model::{ Access, Command, DeprovisionResult, ExecRequest, Metadata, ProvisionRequest, ProvisionResult, - StartResult, StdinMode, StopResult, duration_millis, + StartResult, StdinMode, StopResult, }; use crate::nvxhost::{HostLibrary, LaunchInputs, Session}; -const BACKEND_KEY: &str = "nvxhost"; +const BACKEND_KEY: &str = NATIVE_BACKEND_KEY; const RUNTIME_ABI: &str = "microvm-abi-v2-edge-ramfs-v1"; const MAX_EXEC_SECONDS: u64 = 3600; const MAX_OUTPUT_BYTES: usize = 1 << 20; @@ -195,8 +195,10 @@ impl NvxHostBackend { config.openvmm = config.openvmm.normalized()?; config.image = absolute(&config.image)?; config.library = absolute(&config.library)?; - let host = HostLibrary::load(&config.library, &config.library_sha256)?; let store_root = config.openvmm.state_root.join(BACKEND_KEY); + validate_unix_socket_path(&store_root, SOCKET_NAME, "control")?; + validate_unix_socket_path(&store_root, BOOT_SOCKET_NAME, "boot-console")?; + let host = HostLibrary::load(&config.library, &config.library_sha256)?; let store = StateStore::open_for(&store_root, BACKEND_KEY).map_err(|error| { Error::backend_unavailable(format!( "cannot open native sandbox state root {}", @@ -239,8 +241,12 @@ impl NvxHostBackend { /// Collects the guest's bounded, non-follow log snapshot through ttrpc. pub fn guest_logs(&self, sandbox_id: &SandboxId) -> Result> { let (runtime, capability) = self.running(sandbox_id)?; - let mut session = - self.connect(&runtime, &capability, self.config.openvmm.control_timeout)?; + let deadline = Instant::now() + self.config.openvmm.control_timeout; + let mut session = self.connect( + &runtime, + &capability, + deadline.saturating_duration_since(Instant::now()), + )?; let mut output = Vec::new(); let mut offset = 0u64; let request = StreamLogsRequest { @@ -252,7 +258,7 @@ impl NvxHostBackend { let streamed = session.server_stream( "StreamLogs", &request.encode_to_vec(), - self.config.openvmm.control_timeout, + deadline.saturating_duration_since(Instant::now()), |bytes| { let chunk = LogChunk::decode(bytes).map_err(|error| { Error::backend_error("the guest returned an invalid log chunk") @@ -285,7 +291,7 @@ impl NvxHostBackend { Ok(()) }, ); - finish_session(session, streamed)?; + finish_session(session, streamed, Some(deadline))?; Ok(output) } @@ -675,7 +681,7 @@ impl Backend for NvxHostBackend { deadline.saturating_duration_since(Instant::now()), )?; let result = self.wait_ready(&mut session, deadline); - finish_session(session, result) + finish_session(session, result, Some(deadline)) })(); let agent = match ready { Ok(agent) => agent, @@ -768,7 +774,7 @@ impl Backend for NvxHostBackend { .encode_to_vec(), Some(remaining), ); - let response = finish_session(session, response)?; + let response = finish_session(session, response, Some(deadline))?; ShutdownResponse::decode(response.as_slice()).map_err(|error| { Error::backend_error("the guest returned an invalid shutdown response") .with_source(error) @@ -851,13 +857,14 @@ fn file_digest(path: &Path) -> Result { Ok(format!("{:x}", hasher.finalize())) } -fn finish_session(session: Session, result: Result) -> Result { - match (result, session.close()) { +fn finish_session(session: Session, result: Result, deadline: Option) -> Result { + match (result, session.close(deadline)) { (Ok(value), Ok(())) => Ok(value), (Err(error), Ok(())) | (Ok(_), Err(error)) => Err(error), - (Err(error), Err(close)) => Err(Error::backend_error(format!( - "{error}; closing the guest session also failed: {close}" - ))), + (Err(error), Err(close)) => { + eprintln!("nvxhost could not close a failed guest session: {close}"); + Err(error) + } } } @@ -955,23 +962,25 @@ fn prepare_exec(request: &ExecRequest) -> Result<(ExecuteCommandRequest, Option< "the guest command must contain non-NUL arguments totaling at most 4096 bytes", )); } - let requested_ms = request.process.timeout.map(duration_millis).unwrap_or(0); - if requested_ms > MAX_EXEC_SECONDS * 1_000 { - return Err(Error::policy_validation( - "the guest command timeout exceeds one hour", - )); + let timeout = request.process.timeout.filter(|timeout| !timeout.is_zero()); + if let Some(timeout) = timeout { + if timeout > Duration::from_secs(MAX_EXEC_SECONDS) { + return Err(Error::policy_validation( + "the guest command timeout exceeds one hour", + )); + } + if timeout.subsec_nanos() != 0 { + return Err(Error::policy_validation( + "the nvxhost backend requires whole-second precision for positive command timeouts", + )); + } } - let seconds = if requested_ms == 0 { - 0 - } else { - requested_ms.div_ceil(1_000) - }; - let timeout = (seconds != 0).then(|| Duration::from_secs(seconds)); + let seconds = timeout.map_or(0, |duration| duration.as_secs() as i32); Ok(( ExecuteCommandRequest { command: argv[0].clone(), args: argv[1..].to_vec(), - timeout_seconds: seconds as i32, + timeout_seconds: seconds, }, timeout, )) @@ -1002,12 +1011,13 @@ fn execute(job: ExecJob) -> Result { }) }, )?; + let deadline = timeout.map(|timeout| Instant::now() + timeout + config.exec_response_grace); let response = session.unary( "ExecuteCommand", &command.encode_to_vec(), - timeout.map(|timeout| timeout + config.exec_response_grace), + deadline.map(|deadline| deadline.saturating_duration_since(Instant::now())), ); - let response = finish_session(session, response)?; + let response = finish_session(session, response, deadline)?; let reply = ExecuteCommandResponse::decode(response.as_slice()).map_err(|error| { Error::backend_error("the guest returned an invalid execution result").with_source(error) })?; @@ -1045,7 +1055,7 @@ mod tests { #[test] fn simple_exec_has_no_container_metadata_and_respects_timeout() { let (request, timeout) = prepare_exec( - &ExecRequest::command_line("printf READY").with_timeout(Duration::from_millis(1500)), + &ExecRequest::command_line("printf READY").with_timeout(Duration::from_secs(2)), ) .unwrap(); assert_eq!(request.command, SHELL); @@ -1058,6 +1068,62 @@ mod tests { assert!(!capabilities().filesystem.readonly_paths); } + #[test] + fn fractional_second_exec_timeouts_are_rejected_but_zero_is_unbounded() { + let (unbounded, timeout) = + prepare_exec(&ExecRequest::command_line("true").with_timeout(Duration::ZERO)).unwrap(); + assert_eq!(unbounded.timeout_seconds, 0); + assert_eq!(timeout, None); + for timeout in [ + Duration::from_nanos(1), + Duration::from_millis(1), + Duration::from_millis(999), + Duration::from_millis(1001), + ] { + let error = + prepare_exec(&ExecRequest::command_line("true").with_timeout(timeout)).unwrap_err(); + assert_eq!(error.code(), ErrorCode::PolicyValidation); + assert!(error.message().contains("whole-second precision")); + } + assert_eq!( + prepare_exec( + &ExecRequest::command_line("true") + .with_timeout(Duration::from_secs(MAX_EXEC_SECONDS + 1)), + ) + .unwrap_err() + .code(), + ErrorCode::PolicyValidation, + ); + } + + #[cfg(unix)] + #[test] + fn native_socket_paths_are_checked_before_loading_the_library() { + let suffix = PathBuf::from("0".repeat(32)).join(SOCKET_NAME); + let target_root_len = 107 - 1 - suffix.as_os_str().len(); + let state_root = PathBuf::from("/").join("x".repeat(target_root_len - 1)); + validate_unix_socket_path(&state_root, SOCKET_NAME, "control").unwrap(); + let native_root = state_root.join(BACKEND_KEY); + assert!(validate_unix_socket_path(&native_root, BOOT_SOCKET_NAME, "boot-console").is_err()); + let config = OpenVmmConfig::new( + "/tmp/openvmm", + "/tmp/vmlinux", + "/tmp/initramfs", + super::super::Hypervisor::Kvm, + &state_root, + ); + let error = NvxHostBackend::new(NvxHostConfig::new( + config, + "/tmp/image.vhd", + "/tmp/libnvxhost.so", + [0; 32], + )) + .unwrap_err(); + assert_eq!(error.code(), ErrorCode::BackendUnavailable); + assert!(error.message().contains("Unix control socket path")); + assert!(!native_root.exists()); + } + #[test] fn unsupported_exec_options_fail_before_guest_work() { let request = ExecRequest::command_line("echo READY").with_cwd("/tmp"); diff --git a/aci_edge_sandboxes/src/openvmm/state.rs b/aci_edge_sandboxes/src/openvmm/state.rs index 9a4ed434f..7138a4533 100644 --- a/aci_edge_sandboxes/src/openvmm/state.rs +++ b/aci_edge_sandboxes/src/openvmm/state.rs @@ -35,6 +35,8 @@ use crate::model::NetworkPolicy; pub(crate) const STATE_FORMAT: u32 = 2; /// Backend key recorded in every sandbox. pub(crate) const BACKEND_KEY: &str = "openvmm"; +/// Backend key for the separately supplied native host library. +pub(crate) const NATIVE_BACKEND_KEY: &str = "nvxhost"; /// Linux control socket name. pub(crate) const SOCKET_NAME: &str = "control.sock"; /// Linux boot-console socket name. @@ -329,15 +331,12 @@ impl StateStore { /// Removes the files that exist only while the sandbox starts or runs. pub(crate) fn clear_runtime(&self, sandbox_id: &SandboxId) -> Result<()> { let dir = self.dir(sandbox_id); - for name in [ - RUNTIME_NAME, - LAUNCH_NAME, - CAPABILITY_NAME, - SOCKET_NAME, - BOOT_SOCKET_NAME, - ] { + for name in [RUNTIME_NAME, LAUNCH_NAME, CAPABILITY_NAME, SOCKET_NAME] { remove_if_present(&dir.join(name))?; } + if self.backend == NATIVE_BACKEND_KEY { + remove_if_present(&dir.join(BOOT_SOCKET_NAME))?; + } Ok(()) } @@ -374,11 +373,10 @@ impl StateStore { LAUNCH_NAME, CAPABILITY_NAME, SOCKET_NAME, - BOOT_SOCKET_NAME, OUTCOME_NAME, LOG_NAME, - CONSOLE_LOG_NAME, ]; + let native_only = [BOOT_SOCKET_NAME, CONSOLE_LOG_NAME]; let mut owned = Vec::new(); for entry in entries { let entry = entry.map_err(io_error(format!( @@ -388,7 +386,10 @@ impl StateStore { let name = entry.file_name(); let name = name.to_string_lossy(); let temporary = name.starts_with('.') && name.ends_with(".tmp"); - if known.contains(&name.as_ref()) || temporary { + if known.contains(&name.as_ref()) + || (self.backend == NATIVE_BACKEND_KEY && native_only.contains(&name.as_ref())) + || temporary + { owned.push(entry.path()); } else if name != RECORD_NAME { return Err(Error::backend_error(format!( @@ -520,7 +521,7 @@ mod tests { #[test] fn native_records_round_trip_only_through_their_backend() { let root = tempfile::tempdir().unwrap(); - let store = StateStore::open_for(root.path(), "nvxhost").unwrap(); + let store = StateStore::open_for(root.path(), NATIVE_BACKEND_KEY).unwrap(); let id = SandboxId::generate().unwrap(); assert_eq!( store.create(&id, &record()).unwrap_err().code(), @@ -529,7 +530,7 @@ mod tests { assert!(!store.dir(&id).exists()); let mut native = record(); - native.backend = "nvxhost".to_owned(); + native.backend = NATIVE_BACKEND_KEY.to_owned(); native.native = Some(NativeArtifactRecord { image: PathBuf::from("image.vhd"), image_sha256: "1".repeat(64), @@ -548,6 +549,29 @@ mod tests { .code(), ErrorCode::BackendError ); + drop(_guard); + fs::write(store.dir(&id).join(BOOT_SOCKET_NAME), b"socket").unwrap(); + fs::write(store.dir(&id).join(CONSOLE_LOG_NAME), b"console").unwrap(); + store.clear_runtime(&id).unwrap(); + assert!(!store.dir(&id).join(BOOT_SOCKET_NAME).exists()); + store.remove(&id).unwrap(); + assert!(!store.dir(&id).exists()); + } + + #[test] + fn direct_records_preserve_native_only_filenames() { + let root = tempfile::tempdir().unwrap(); + let store = StateStore::open(root.path()).unwrap(); + let id = SandboxId::generate().unwrap(); + store.create(&id, &record()).unwrap(); + for name in [BOOT_SOCKET_NAME, CONSOLE_LOG_NAME] { + fs::write(store.dir(&id).join(name), b"not owned").unwrap(); + } + store.clear_runtime(&id).unwrap(); + assert!(store.dir(&id).join(BOOT_SOCKET_NAME).exists()); + assert!(store.remove(&id).is_err()); + assert!(store.dir(&id).join(CONSOLE_LOG_NAME).exists()); + assert_eq!(store.load(&id).unwrap(), record()); } #[test] From adfa16761b862af949577538147bd84229d95561 Mon Sep 17 00:00:00 2001 From: Enrique Saurez Date: Mon, 5 Oct 2026 16:12:02 -0700 Subject: [PATCH 3/5] Document opt-in native-host edge example inputs Show a concrete WHP invocation, the separately supplied artifacts and trust requirement, and the unverified MSHV path. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- aci_edge_sandboxes/README.md | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/aci_edge_sandboxes/README.md b/aci_edge_sandboxes/README.md index d3ef0f54f..432063f80 100644 --- a/aci_edge_sandboxes/README.md +++ b/aci_edge_sandboxes/README.md @@ -144,6 +144,33 @@ operations are not part of this profile. See `tests/nvxhost_guest.rs` exercises an actual WHP guest when the corresponding `NVXHOST_TEST_*` paths and approved DLL digest are set. +For example, from `aci_edge_sandboxes` on a Windows WHP host, set the following +paths to compatible, separately built artifacts and a caller-prepared GPT disk +with p2+ ext4 layers (not a container image reference). Use the **edge** guest +initramfs, not the default Alpine or standard container guest initramfs: + +```powershell +$openvmm = 'C:\path\to\openvmm.exe' +$kernel = 'C:\path\to\vmlinux' +$initrd = 'C:\path\to\nvx-edge-initramfs.cpio.gz' +$image = 'C:\path\to\layers.gpt' +$hostLibrary = 'C:\path\to\nvxhost.dll' +$approvedHostSha256 = '<64 hexadecimal digits from an independent trust policy>' +$stateRoot = 'C:\nvx-edge-state' + +cargo run --release --locked --features nvxhost --example nvxhost_lifecycle -- ` + --openvmm $openvmm --kernel $kernel --initrd $initrd --image $image ` + --host-library $hostLibrary --host-sha256 $approvedHostSha256 ` + --state-root $stateRoot --hypervisor whp -- 'printf READY' +``` + +Build the native library and the static edge initramfs separately from their +matching private sources; this example neither fetches nor builds them. Use the +pinned OpenVMM, which includes the scratchless RAM-overlay topology, and a kernel +compatible with that OpenVMM and guest revision. On Linux, supply a matching +`libnvxhost.so` and OpenVMM build and select `--hypervisor mshv`; the Linux edge +lifecycle has not yet been verified end to end. + The caller must independently approve and protect the native asset. Checking a caller-supplied digest does not make a writable path or a self-declared digest trustworthy; use this profile only with an immutable, externally authorized library installation. From 22cfcb118aa0ad3c044c6cc709d42e0ed653a4ea Mon Sep 17 00:00:00 2001 From: Enrique Saurez Date: Mon, 5 Oct 2026 20:07:47 -0700 Subject: [PATCH 4/5] Keep the native boot console attached across reattach and crashes OpenVMM serves one boot-console client at a time and retains guest output while none is connected. The native backend keyed its console listener by sandbox only, so a restart after an OpenVMM crash reused the finished listener of the crashed launch; with no listener, a verbose guest stalled until start timed out. A second backend instance that reattached to a running guest could not connect while the first one held the console, and stop then failed after the guest had already stopped. Key listeners by the OpenVMM process identity and retire stale ones before a launch. A listener that finds the console held elsewhere waits, without the start deadline, to take it over and ends quietly when OpenVMM exits; a reset after OpenVMM exits ends the log. Stop and deprovision report capture failures as consoleError metadata instead of failing. Extend the opt-in WHP suite with exec semantics, lifecycle errors and restarts, reattachment from a new backend, a crashed guest, a guest that outlives the process that started it, starts killed at several points, parallel sandboxes, and concurrent commands. List OpenVMM processes in a way that fails loudly, so the leak checks cannot pass vacuously. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- aci_edge_sandboxes/README.md | 13 + aci_edge_sandboxes/src/openvmm/native.rs | 259 ++++++++-- aci_edge_sandboxes/tests/nvxhost_guest.rs | 567 +++++++++++++++++++++- 3 files changed, 803 insertions(+), 36 deletions(-) diff --git a/aci_edge_sandboxes/README.md b/aci_edge_sandboxes/README.md index 432063f80..140fec1d1 100644 --- a/aci_edge_sandboxes/README.md +++ b/aci_edge_sandboxes/README.md @@ -131,6 +131,19 @@ State is kept separately under `/nvxhost/`; a running VM is never si converted to the direct backend. `NvxHostBackend::guest_logs` reads a bounded non-follow guest-log snapshot before stop. +Provision records the SHA-256 of the image, kernel, and initramfs, and every start hashes +them again and refuses changed artifacts, so start latency grows with the image size. As +with the direct backend, executions on one sandbox share its single control connection: a +concurrent exec waits up to `control_timeout` for the running one to finish. The backend +copies the guest boot console to `NvxHostBackend::console_log_path` from the process that +started or last used the sandbox. OpenVMM serves one console client at a time and holds +guest output while none is connected, so a guest that writes enough console output stalls +until a listener reconnects: keep that process running, or use the sandbox from its +successor, whose listener waits to take the capture over. Start reports `bootMilliseconds` +and `guestBuildId`; stop reports `forced` and, after a failed graceful shutdown, +`gracefulError`. A capture failure never fails stop or deprovision: both report it as +`consoleError`. + This first backend supports provision/start/exec/stop/deprovision with shell commands or argv and a fixed caller-provided image. Positive execution timeouts must be whole seconds, matching the guest RPC's precision; finer-grained timeouts fail validation rather than diff --git a/aci_edge_sandboxes/src/openvmm/native.rs b/aci_edge_sandboxes/src/openvmm/native.rs index 29be94220..674226a65 100644 --- a/aci_edge_sandboxes/src/openvmm/native.rs +++ b/aci_edge_sandboxes/src/openvmm/native.rs @@ -13,7 +13,7 @@ use sha2_runtime::{Digest, Sha256}; use super::artifacts::absolute; use super::config::{OpenVmmConfig, validate_unix_socket_path}; -use super::platform; +use super::platform::{self, Transport}; use super::process; use super::state::{ BOOT_SOCKET_NAME, LaunchRecord, NATIVE_BACKEND_KEY, NativeArtifactRecord, ProcessIdentity, @@ -36,6 +36,7 @@ const RUNTIME_ABI: &str = "microvm-abi-v2-edge-ramfs-v1"; const MAX_EXEC_SECONDS: u64 = 3600; const MAX_OUTPUT_BYTES: usize = 1 << 20; const SHELL: &str = "/bin/sh"; +const CONSOLE_POLL: Duration = Duration::from_millis(250); /// Artifact paths and approved native-library digest for an image-backed guest. /// @@ -87,7 +88,15 @@ pub struct NvxHostBackend { config: NvxHostConfig, base: OpenVmmBackend, host: Arc, - console_pumps: Mutex>>>, + console_pumps: Mutex>, +} + +/// Copies the boot console of one OpenVMM launch into the sandbox's console log. +#[derive(Debug)] +struct ConsolePump { + pid: u32, + start_time: u64, + thread: thread::JoinHandle>, } #[derive(Clone, PartialEq, Message)] @@ -326,21 +335,37 @@ impl NvxHostBackend { .console_pumps .lock() .map_err(|_| Error::backend_error("the boot-console pump table is unavailable"))?; - if pumps.contains_key(sandbox_id) { + if let Some(pump) = pumps.get(sandbox_id) + && pump.pid == runtime.pid + && pump.start_time == runtime.start_time + { return Ok(()); } + // A listener of an earlier launch ends promptly because its OpenVMM has exited; its + // result only describes that launch. + if let Some(stale) = pumps.remove(sandbox_id) { + let _ = stale.thread.join(); + } let endpoint = self.boot_endpoint(sandbox_id, &runtime.endpoint)?; let file = self.base.store.open_console_log(sandbox_id)?; let runtime = runtime.clone(); + let (pid, start_time) = (runtime.pid, runtime.start_time); let timeout = self.config.openvmm.start_timeout; - let pump = thread::Builder::new() + let thread = thread::Builder::new() .name("nvxhost-boot-console".to_owned()) .spawn(move || pump_console(runtime, endpoint, file, timeout)) .map_err(|error| { Error::backend_error("cannot start the guest boot-console listener") .with_source(error) })?; - pumps.insert(sandbox_id.clone(), pump); + pumps.insert( + sandbox_id.clone(), + ConsolePump { + pid, + start_time, + thread, + }, + ); Ok(()) } @@ -351,7 +376,8 @@ impl NvxHostBackend { .map_err(|_| Error::backend_error("the boot-console pump table is unavailable"))? .remove(sandbox_id); if let Some(pump) = pump { - pump.join() + pump.thread + .join() .map_err(|_| Error::backend_error("the boot-console listener panicked"))??; } Ok(()) @@ -571,6 +597,9 @@ impl Backend for NvxHostBackend { "sandbox {sandbox_id} is already running" ))); } + // A listener of a launch that exited on its own ends promptly. Retire it now so that the + // new launch's listener connects before the guest writes its first boot output. + let _ = self.finish_console_pump(sandbox_id); self.verify_record(&record, true)?; self.probe()?; @@ -806,9 +835,11 @@ impl Backend for NvxHostBackend { if let Err(error) = graceful { metadata.insert("gracefulError".to_owned(), error.to_string().into()); } - let console = self.finish_console_pump(sandbox_id); + // The guest has stopped; a boot-console capture failure is only a diagnostic. + if let Err(error) = self.finish_console_pump(sandbox_id) { + metadata.insert("consoleError".to_owned(), error.to_string().into()); + } self.base.store.clear_runtime(sandbox_id)?; - console?; Ok(StopResult { metadata: Some(metadata), }) @@ -822,11 +853,16 @@ impl Backend for NvxHostBackend { "sandbox {sandbox_id} is running; stop it before deprovisioning" ))); } - self.finish_console_pump(sandbox_id)?; + // A listener remains only if the guest exited without a stop through this backend. + let console = self.finish_console_pump(sandbox_id); self.base.store.remove(sandbox_id)?; drop(guard); self.base.store.remove_lock(sandbox_id); - Ok(DeprovisionResult::default()) + Ok(DeprovisionResult { + metadata: console.err().map(|error| { + Metadata::from_iter([("consoleError".to_owned(), error.to_string().into())]) + }), + }) } } @@ -871,19 +907,42 @@ fn finish_session(session: Session, result: Result, deadline: Option Result<()> { + copy_console( + || platform::connect_endpoint(&endpoint, runtime.pid, CONSOLE_POLL), + |activity| { + let current = platform::process_start_time(runtime.pid).map_err(|error| { + Error::backend_error(format!( + "cannot verify OpenVMM while {activity} the console" + )) + .with_source(error) + })?; + Ok(current != Some(runtime.start_time)) + }, + output, + timeout, + ) +} + +/// Copies the guest boot console into `output` until OpenVMM exits. +/// +/// OpenVMM serves one console client at a time, so the listener of another backend instance or +/// process may hold the console. This listener then waits, past `timeout`, to take over, and +/// ends without error if OpenVMM exits first. +fn copy_console( + mut connect: impl FnMut() -> io::Result>, + exited: impl Fn(&str) -> Result, mut output: File, timeout: Duration, ) -> Result<()> { let deadline = Instant::now() .checked_add(timeout) .ok_or_else(|| Error::backend_error("boot-console timeout is too long"))?; + let mut held_elsewhere = false; let mut console = loop { - if Instant::now() >= deadline { - return Err(Error::backend_error( - "OpenVMM did not open its guest boot-console endpoint", - )); - } - match platform::connect_endpoint(&endpoint, runtime.pid, Duration::from_millis(250)) { + match connect() { Ok(console) => break console, Err(error) if matches!( @@ -894,15 +953,20 @@ fn pump_console( | io::ErrorKind::TimedOut ) => { - if platform::process_start_time(runtime.pid).map_err(|error| { - Error::backend_error("cannot verify OpenVMM while connecting the console") - .with_source(error) - })? != Some(runtime.start_time) - { + held_elsewhere |= error.kind() == io::ErrorKind::WouldBlock; + if exited("connecting")? { + if held_elsewhere { + return Ok(()); + } return Err(Error::backend_error( "OpenVMM exited before the guest boot console could connect", )); } + if !held_elsewhere && Instant::now() >= deadline { + return Err(Error::backend_error( + "OpenVMM did not open its guest boot-console endpoint", + )); + } thread::sleep(Duration::from_millis(25)); } Err(error) => { @@ -915,24 +979,21 @@ fn pump_console( }; let mut buffer = [0u8; 4096]; loop { - match console.read(&mut buffer, Some(Duration::from_millis(250))) { + match console.read(&mut buffer, Some(CONSOLE_POLL)) { Ok(0) => break, Ok(size) => output.write_all(&buffer[..size]).map_err(|error| { Error::backend_error("cannot write the guest boot-console log").with_source(error) })?, - Err(error) if error.kind() == io::ErrorKind::TimedOut => { - if platform::process_start_time(runtime.pid).map_err(|error| { - Error::backend_error("cannot verify OpenVMM while reading the console") - .with_source(error) - })? != Some(runtime.start_time) - { + // A client still queued in a socket's listen backlog is reset, rather than reaching + // end of stream, when OpenVMM exits. + Err(error) => { + if exited("reading")? { break; } - } - Err(error) => { - return Err( - Error::backend_error("cannot read the guest boot console").with_source(error) - ); + if error.kind() != io::ErrorKind::TimedOut { + return Err(Error::backend_error("cannot read the guest boot console") + .with_source(error)); + } } } } @@ -1049,9 +1110,141 @@ impl ExecControl for NvxHostExecution { #[cfg(test)] mod tests { + use std::cell::Cell; + use std::collections::VecDeque; + use std::io::{Read, Seek, SeekFrom}; + use super::*; use crate::ErrorCode; + /// Replays scripted boot-console reads. + struct Replay(VecDeque>); + + impl Transport for Replay { + fn read(&mut self, buffer: &mut [u8], _timeout: Option) -> io::Result { + let chunk = self + .0 + .pop_front() + .expect("the console read past its script")?; + buffer[..chunk.len()].copy_from_slice(chunk); + Ok(chunk.len()) + } + + fn write_all(&mut self, _data: &[u8], _timeout: Option) -> io::Result<()> { + unreachable!("the console listener never writes") + } + } + + fn replay( + reads: [io::Result<&'static [u8]>; N], + ) -> io::Result> { + Ok(Box::new(Replay(reads.into()))) + } + + fn busy() -> io::Result> { + Err(io::ErrorKind::WouldBlock.into()) + } + + fn read_log(mut log: File) -> String { + let mut text = String::new(); + log.seek(SeekFrom::Start(0)).unwrap(); + log.read_to_string(&mut text).unwrap(); + text + } + + #[test] + fn a_console_held_elsewhere_is_awaited_past_the_deadline_until_openvmm_exits() { + let log = tempfile::tempfile().unwrap(); + let (attempts, checks) = (Cell::new(0), Cell::new(0)); + copy_console( + || { + attempts.set(attempts.get() + 1); + busy() + }, + |_| { + checks.set(checks.get() + 1); + Ok(checks.get() > 4) + }, + log.try_clone().unwrap(), + Duration::ZERO, + ) + .unwrap(); + assert_eq!(attempts.get(), 5); + assert_eq!(read_log(log), ""); + } + + #[test] + fn a_released_console_is_taken_over_and_copied_to_its_end() { + let log = tempfile::tempfile().unwrap(); + let mut attempts = VecDeque::from([ + busy(), + replay([ + Ok(b"late ".as_slice()), + Err(io::ErrorKind::TimedOut.into()), + Ok(b"output".as_slice()), + Ok(b"".as_slice()), + ]), + ]); + copy_console( + || attempts.pop_front().unwrap(), + |_| Ok(false), + log.try_clone().unwrap(), + Duration::ZERO, + ) + .unwrap(); + assert_eq!(read_log(log), "late output"); + } + + #[test] + fn a_console_that_never_opens_fails_at_its_deadline_or_when_openvmm_exits() { + for (exited, expected) in [ + (false, "did not open its guest boot-console endpoint"), + (true, "exited before the guest boot console could connect"), + ] { + let error = copy_console( + || Err(io::ErrorKind::NotFound.into()), + |_| Ok(exited), + tempfile::tempfile().unwrap(), + Duration::ZERO, + ) + .unwrap_err(); + assert!(error.message().contains(expected), "{error}"); + } + } + + #[test] + fn a_console_reset_ends_the_log_only_after_openvmm_exits() { + let log = tempfile::tempfile().unwrap(); + let reset = || { + replay([ + Ok(b"boot".as_slice()), + Err(io::ErrorKind::ConnectionReset.into()), + ]) + }; + copy_console( + reset, + |_| Ok(true), + log.try_clone().unwrap(), + Duration::ZERO, + ) + .unwrap(); + assert_eq!(read_log(log), "boot"); + + let error = copy_console( + reset, + |_| Ok(false), + tempfile::tempfile().unwrap(), + Duration::ZERO, + ) + .unwrap_err(); + assert!( + error + .message() + .contains("cannot read the guest boot console"), + "{error}" + ); + } + #[test] fn simple_exec_has_no_container_metadata_and_respects_timeout() { let (request, timeout) = prepare_exec( diff --git a/aci_edge_sandboxes/tests/nvxhost_guest.rs b/aci_edge_sandboxes/tests/nvxhost_guest.rs index b6601ac35..f759fa77b 100644 --- a/aci_edge_sandboxes/tests/nvxhost_guest.rs +++ b/aci_edge_sandboxes/tests/nvxhost_guest.rs @@ -1,14 +1,28 @@ -//! Opt-in WHP lifecycle proof with a privately supplied native library and guest. +//! Opt-in WHP lifecycle and robustness proofs with a separately supplied native library and guest. +//! +//! The tests need `NVXHOST_TEST_OPENVMM`, `NVXHOST_TEST_KERNEL`, `NVXHOST_TEST_INITRD`, +//! `NVXHOST_TEST_IMAGE`, `NVXHOST_TEST_LIBRARY`, and the approved `NVXHOST_TEST_SHA256`. Run them +//! one at a time so that the check for leftover OpenVMM processes is exact, and optimized, since +//! every provision and start hashes the image: +//! `cargo test --release --features nvxhost --test nvxhost_guest -- --ignored --test-threads=1`. #![cfg(feature = "nvxhost")] -use std::path::PathBuf; +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::{Child, Command, Stdio}; use std::sync::Arc; +use std::thread; +use std::time::{Duration, Instant}; use aci_edge_sandboxes::openvmm::{Hypervisor, NvxHostBackend, NvxHostConfig, OpenVmmConfig}; use aci_edge_sandboxes::{ - AciEdgeSandbox, Error, ErrorCode, ExecOutcome, ExecRequest, ProvisionRequest, + AciEdgeSandbox, Error, ErrorCode, ExecOutcome, ExecOutput, ExecRequest, ProvisionRequest, + Result, SandboxId, StdinMode, StopResult, }; +const HELPER_STATE: &str = "NVXHOST_TEST_HELPER_STATE"; +const HELPER_SANDBOX: &str = "NVXHOST_TEST_HELPER_SANDBOX"; + fn required(name: &str) -> PathBuf { PathBuf::from(std::env::var_os(name).unwrap_or_else(|| panic!("{name} must be set"))) } @@ -24,6 +38,226 @@ fn approved_digest() -> [u8; 32] { digest } +fn backend_with( + state: &Path, + guest_debug: bool, + adjust: impl FnOnce(&mut OpenVmmConfig), +) -> Arc { + let mut config = OpenVmmConfig::new( + required("NVXHOST_TEST_OPENVMM"), + required("NVXHOST_TEST_KERNEL"), + required("NVXHOST_TEST_INITRD"), + Hypervisor::Whp, + state, + ); + adjust(&mut config); + Arc::new( + NvxHostBackend::new( + NvxHostConfig::new( + config, + required("NVXHOST_TEST_IMAGE"), + required("NVXHOST_TEST_LIBRARY"), + approved_digest(), + ) + .with_guest_debug(guest_debug), + ) + .unwrap(), + ) +} + +fn backend(state: &Path) -> Arc { + backend_with(state, false, |_| {}) +} + +/// Lists the OpenVMM processes running on this host, failing if they cannot be listed. +fn openvmm_processes() -> BTreeSet { + let executable = required("NVXHOST_TEST_OPENVMM"); + let name = executable + .file_stem() + .unwrap() + .to_string_lossy() + .into_owned(); + #[cfg(windows)] + let output = Command::new("powershell.exe") + .args([ + "-NoProfile", + "-NonInteractive", + "-Command", + &format!("[Diagnostics.Process]::GetProcessesByName('{name}') | ForEach-Object Id"), + ]) + .output() + .unwrap(); + #[cfg(not(windows))] + let output = Command::new("pgrep").args(["-x", &name]).output().unwrap(); + // pgrep exits with 1 when no process matches. + assert!( + output.status.success() || (cfg!(not(windows)) && output.status.code() == Some(1)), + "cannot list OpenVMM processes: {output:?}" + ); + String::from_utf8_lossy(&output.stdout) + .lines() + .map(str::trim) + .filter(|line| !line.is_empty()) + .map(|line| { + line.parse() + .unwrap_or_else(|_| panic!("unexpected process ID {line:?}")) + }) + .collect() +} + +/// Fails if an OpenVMM process that was not running before the test is still running. +fn assert_no_new_openvmm(before: &BTreeSet) { + let deadline = Instant::now() + Duration::from_secs(10); + loop { + let leaked: Vec = openvmm_processes().difference(before).copied().collect(); + if leaked.is_empty() { + return; + } + assert!( + Instant::now() < deadline, + "OpenVMM processes {leaked:?} outlived their sandboxes" + ); + thread::sleep(Duration::from_millis(100)); + } +} + +/// Terminates a process abruptly, as a crash would. +fn kill_process(pid: u32) { + #[cfg(windows)] + let status = Command::new("powershell.exe") + .args([ + "-NoProfile", + "-NonInteractive", + "-Command", + &format!("Stop-Process -Id {pid} -Force -ErrorAction Stop"), + ]) + .status() + .unwrap(); + #[cfg(not(windows))] + let status = Command::new("kill") + .args(["-KILL", &pid.to_string()]) + .status() + .unwrap(); + assert!(status.success(), "cannot kill process {pid}: {status}"); +} + +/// Stops and removes a sandbox when a test ends, printing its diagnostics if the test failed. +struct Cleanup<'a> { + client: &'a AciEdgeSandbox, + backend: &'a NvxHostBackend, + id: SandboxId, + armed: bool, +} + +impl Cleanup<'_> { + fn release(mut self) -> SandboxId { + self.armed = false; + self.id.clone() + } +} + +impl Drop for Cleanup<'_> { + fn drop(&mut self) { + if !self.armed { + return; + } + if thread::panicking() { + let read = |path: PathBuf| { + std::fs::read_to_string(&path) + .unwrap_or_else(|error| format!("cannot read {}: {error}", path.display())) + }; + eprintln!( + "sandbox {}\nOpenVMM log: {}\nguest console: {}\nOpenVMM outcome: {}", + self.id, + read(self.backend.log_path(&self.id)), + read(self.backend.console_log_path(&self.id)), + read(self.backend.outcome_report_path(&self.id)), + ); + } + let _ = self.client.stop(&self.id); + let _ = self.client.deprovision(&self.id); + } +} + +fn provision<'a>(client: &'a AciEdgeSandbox, backend: &'a NvxHostBackend) -> Cleanup<'a> { + let id = client + .provision(&ProvisionRequest::new()) + .unwrap() + .sandbox_id; + Cleanup { + client, + backend, + id, + armed: true, + } +} + +fn exec(client: &AciEdgeSandbox, id: &SandboxId, request: &ExecRequest) -> ExecOutput { + client + .exec(id, request) + .and_then(|execution| execution.wait_with_output()) + .unwrap_or_else(|error| panic!("{request:?} failed: {error}")) +} + +fn assert_prints(client: &AciEdgeSandbox, id: &SandboxId, text: &str) { + let output = exec( + client, + id, + &ExecRequest::command_line(format!("printf %s {text}")), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(0), "{output:?}"); + assert_eq!(String::from_utf8_lossy(&output.stdout), text); +} + +fn forced(stopped: &StopResult) -> Option { + stopped + .metadata + .as_ref() + .and_then(|metadata| metadata.get("forced")) + .and_then(serde_json::Value::as_bool) +} + +/// Fails if the boot-console capture reported an error. +fn assert_console_captured(stopped: &StopResult) { + assert!( + stopped + .metadata + .as_ref() + .is_none_or(|metadata| !metadata.contains_key("consoleError")), + "{:?}", + stopped.metadata + ); +} + +fn assert_graceful(stopped: Result) { + let stopped = stopped.unwrap(); + assert_eq!(forced(&stopped), Some(false), "{:?}", stopped.metadata); + assert_console_captured(&stopped); +} + +fn failure(result: Result) -> ErrorCode { + match result { + Ok(_) => panic!("the operation unexpectedly succeeded"), + Err(error) => error.code(), + } +} + +fn start_in_helper(state: &Path, id: &SandboxId) -> Child { + Command::new(std::env::current_exe().unwrap()) + .args([ + "helper_starts_a_sandbox_in_another_process", + "--exact", + "--ignored", + "--nocapture", + "--test-threads=1", + ]) + .env(HELPER_STATE, state) + .env(HELPER_SANDBOX, id.as_str()) + .stdin(Stdio::null()) + .spawn() + .unwrap() +} + #[test] #[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] fn whp_guest_lifecycle_stops_gracefully_and_cleans_up() { @@ -148,3 +382,330 @@ fn whp_guest_lifecycle_stops_gracefully_and_cleans_up() { ); } } + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_exec_reports_exit_codes_streams_timeouts_and_output_limits() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let id = &sandbox.id; + client.start(id).unwrap(); + + let output = exec( + &client, + id, + &ExecRequest::command_line("printf out; printf err >&2; exit 7"), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(7)); + assert_eq!(output.stdout, b"out"); + assert_eq!(output.stderr, b"err"); + + let output = exec( + &client, + id, + &ExecRequest::argv([ + "/bin/sh", + "-c", + "printf '%s|' \"$@\"", + "sh", + "a b", + "$HOME", + ]), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(0)); + assert_eq!(output.stdout, b"a b|$HOME|"); + + let started = Instant::now(); + let output = exec( + &client, + id, + &ExecRequest::command_line("sleep 30").with_timeout(Duration::from_secs(1)), + ); + assert_eq!(output.outcome, ExecOutcome::TimedOut); + assert!( + started.elapsed() < Duration::from_secs(20), + "a one-second timeout took {:?}", + started.elapsed() + ); + + let error = client + .exec(id, &ExecRequest::command_line("yes | head -c 1100000")) + .and_then(|execution| execution.wait_with_output()) + .expect_err("output above the guest limit must fail rather than truncate"); + assert_eq!(error.code(), ErrorCode::BackendError, "{error}"); + assert!(error.message().contains("status 8"), "{error}"); + assert_prints(&client, id, "after-overflow"); + + for index in 0..20 { + assert_prints(&client, id, &format!("command{index}")); + } + for request in [ + ExecRequest::command_line("pwd").with_cwd("/tmp"), + ExecRequest::command_line("env").with_env("NAME=value"), + ExecRequest::command_line("cat").with_stdin(StdinMode::Piped), + ] { + assert_eq!( + failure(client.exec(id, &request)), + ErrorCode::PolicyValidation + ); + } + let logs = backend.guest_logs(id).unwrap(); + assert!(logs.windows(8).any(|window| window == b"execute:")); + assert!(logs.windows(13).any(|window| window == b"output-limit:")); + + assert_graceful(client.stop(id)); + client.deprovision(id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_lifecycle_errors_follow_state_and_a_stopped_sandbox_restarts() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let id = &sandbox.id; + + assert_eq!( + failure(client.exec(id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + client.start(id).unwrap(); + assert_eq!(failure(client.start(id)), ErrorCode::AlreadyStarted); + assert_eq!(failure(client.deprovision(id)), ErrorCode::AlreadyStarted); + assert_prints(&client, id, "first"); + assert_graceful(client.stop(id)); + assert_eq!(failure(client.stop(id)), ErrorCode::AlreadyStopped); + assert_eq!( + failure(client.exec(id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + + client.start(id).unwrap(); + assert_prints(&client, id, "second"); + assert_graceful(client.stop(id)); + client.deprovision(id).unwrap(); + assert_eq!(failure(client.start(id)), ErrorCode::StaleId); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_new_backend_reattaches_to_a_running_guest_and_forces_a_stop() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let id = { + let first = backend(state.path()); + let client = AciEdgeSandbox::from_shared(first.clone()); + let sandbox = provision(&client, &first); + client.start(&sandbox.id).unwrap(); + sandbox.release() + }; + + let second = backend_with(state.path(), false, |config| { + config.stop_timeout = Duration::from_millis(1); + }); + let client = AciEdgeSandbox::from_shared(second.clone()); + let sandbox = Cleanup { + client: &client, + backend: &second, + id, + armed: true, + }; + assert_prints(&client, &sandbox.id, "reattached"); + let logs = second.guest_logs(&sandbox.id).unwrap(); + assert!(logs.windows(8).any(|window| window == b"execute:")); + let stopped = client.stop(&sandbox.id).unwrap(); + assert_eq!(forced(&stopped), Some(true), "{:?}", stopped.metadata); + // The first backend's listener still held the boot console, so the second one waited. + assert_console_captured(&stopped); + assert_eq!( + failure(client.exec(&sandbox.id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + client.deprovision(&sandbox.id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_a_crashed_guest_is_not_running_and_restarts_with_its_console_captured() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend_with(state.path(), true, |_| {}); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let id = &sandbox.id; + client.start(id).unwrap(); + assert_prints(&client, id, "before-crash"); + + let launched: Vec = openvmm_processes().difference(&before).copied().collect(); + assert_eq!(launched.len(), 1, "{launched:?}"); + kill_process(launched[0]); + assert_no_new_openvmm(&before); + assert_eq!( + failure(client.exec(id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + assert_eq!(failure(client.stop(id)), ErrorCode::AlreadyStopped); + + let console = backend.console_log_path(id); + let crashed = std::fs::metadata(&console).unwrap().len(); + client.start(id).unwrap(); + assert_prints(&client, id, "after-crash"); + assert_graceful(client.stop(id)); + let restarted = std::fs::metadata(&console).unwrap().len(); + assert!( + restarted > crashed, + "the restarted guest's boot console was not captured ({crashed} -> {restarted} bytes)" + ); + client.deprovision(id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "helper process for whp_guest_survives_its_caller_and_a_killed_start_leaves_no_vm"] +fn helper_starts_a_sandbox_in_another_process() { + let Some(sandbox) = std::env::var_os(HELPER_SANDBOX) else { + return; + }; + let client = AciEdgeSandbox::from_shared(backend(&required(HELPER_STATE))); + let id = SandboxId::parse(&sandbox.to_string_lossy()).unwrap(); + client.start(&id).unwrap(); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_guest_survives_its_caller_and_a_killed_start_leaves_no_vm() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend_with(state.path(), false, |config| { + config.stop_timeout = Duration::from_secs(5); + }); + let client = AciEdgeSandbox::from_shared(backend.clone()); + + let sandbox = provision(&client, &backend); + let status = start_in_helper(state.path(), &sandbox.id).wait().unwrap(); + assert!( + status.success(), + "the helper process could not start the guest: {status}" + ); + assert_prints(&client, &sandbox.id, "survived"); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.id).unwrap(); + drop(sandbox); + assert_no_new_openvmm(&before); + + for delay in [0, 100, 300, 700, 1500] { + let sandbox = provision(&client, &backend); + let mut helper = start_in_helper(state.path(), &sandbox.id); + thread::sleep(Duration::from_millis(delay)); + let _ = helper.kill(); + let _ = helper.wait(); + match client.stop(&sandbox.id) { + Ok(_) => {} + Err(error) if error.code() == ErrorCode::AlreadyStopped => {} + Err(error) => panic!("stop after a start killed at {delay} ms failed: {error}"), + } + client.deprovision(&sandbox.id).unwrap_or_else(|error| { + panic!("deprovision after a start killed at {delay} ms failed: {error}") + }); + drop(sandbox); + assert_no_new_openvmm(&before); + } +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_parallel_sandboxes_and_repeated_restarts_stay_isolated() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + + thread::scope(|scope| { + let workers: Vec<_> = (0..4) + .map(|index| { + let (client, backend) = (&client, &backend); + scope.spawn(move || { + let sandbox = provision(client, backend); + client.start(&sandbox.id).unwrap(); + assert_prints(client, &sandbox.id, &format!("sandbox{index}")); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.id).unwrap(); + }) + }) + .collect(); + for worker in workers { + worker.join().unwrap(); + } + }); + assert_no_new_openvmm(&before); + + let sandbox = provision(&client, &backend); + let mut starts = Vec::new(); + for cycle in 0..8 { + let called = Instant::now(); + let started = client.start(&sandbox.id).unwrap(); + let total = called.elapsed().as_millis(); + let boot = started + .metadata + .as_ref() + .and_then(|metadata| metadata.get("bootMilliseconds")) + .and_then(serde_json::Value::as_u64) + .unwrap(); + starts.push((total, boot)); + assert_prints(&client, &sandbox.id, &format!("cycle{cycle}")); + assert_graceful(client.stop(&sandbox.id)); + } + client.deprovision(&sandbox.id).unwrap(); + eprintln!("(start call, boot) milliseconds across restarts: {starts:?}"); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_concurrent_commands_share_one_guest() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + client.start(&sandbox.id).unwrap(); + + let started = Instant::now(); + thread::scope(|scope| { + let workers: Vec<_> = (0..4) + .map(|index| { + let (client, id) = (&client, &sandbox.id); + scope.spawn(move || { + let output = exec( + client, + id, + &ExecRequest::command_line(format!("sleep 1; printf parallel{index}")), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(0), "{output:?}"); + assert_eq!( + String::from_utf8_lossy(&output.stdout), + format!("parallel{index}") + ); + }) + }) + .collect(); + for worker in workers { + worker.join().unwrap(); + } + }); + eprintln!( + "four concurrent one-second commands took {:?}", + started.elapsed() + ); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.id).unwrap(); + assert_no_new_openvmm(&before); +} From ac47d53d2a6f1d53ca2849ad0bedeb97fed8673d Mon Sep 17 00:00:00 2001 From: Enrique Saurez Date: Mon, 5 Oct 2026 23:05:03 -0700 Subject: [PATCH 5/5] Register native guest images to keep hashing off the start path Hash the image once when it is registered, or record a digest that the caller already verified, and refer to it by content ID. Provision and start compare cheap file seals instead of hashing again, and fail closed when a file changed. Hash OpenVMM, the kernel, and the initramfs once per backend, optionally against approved digests, and keep full re-hashing as an opt-in diagnostic. Starts now cost about the guest boot time. Windows seals omit the change time: writers can set it, and the system updates it when it caches a file hash in an extended attribute, which failed an unchanged OpenVMM executable closed during testing. Claim the OpenVMM log before a native launch. OpenVMM inherits the claim, so recovery can clear an interrupted start that recorded no process identity once nothing holds the log, instead of leaving the sandbox unusable. Markers of the default backend are unchanged. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- aci_edge_sandboxes/README.md | 30 +- .../examples/nvxhost_lifecycle.rs | 33 +- aci_edge_sandboxes/src/openvmm/images.rs | 891 ++++++++++++++++++ aci_edge_sandboxes/src/openvmm/mod.rs | 66 +- aci_edge_sandboxes/src/openvmm/native.rs | 466 ++++++++- .../src/openvmm/platform/linux.rs | 84 ++ .../src/openvmm/platform/mod.rs | 31 + .../src/openvmm/platform/other.rs | 19 + .../src/openvmm/platform/windows.rs | 124 ++- aci_edge_sandboxes/src/openvmm/state.rs | 66 +- aci_edge_sandboxes/tests/nvxhost_guest.rs | 240 ++++- 11 files changed, 1937 insertions(+), 113 deletions(-) create mode 100644 aci_edge_sandboxes/src/openvmm/images.rs diff --git a/aci_edge_sandboxes/README.md b/aci_edge_sandboxes/README.md index 140fec1d1..846e0e430 100644 --- a/aci_edge_sandboxes/README.md +++ b/aci_edge_sandboxes/README.md @@ -131,9 +131,30 @@ State is kept separately under `/nvxhost/`; a running VM is never si converted to the direct backend. `NvxHostBackend::guest_logs` reads a bounded non-follow guest-log snapshot before stop. -Provision records the SHA-256 of the image, kernel, and initramfs, and every start hashes -them again and refuses changed artifacts, so start latency grows with the image size. As -with the direct backend, executions on one sandbox share its single control connection: a +The backend hashes OpenVMM, the kernel, and the initramfs once, when it is created +(`NvxHostBackend::runtime_digests`); `NvxHostConfig::with_runtime_digests` requires approved +digests. It also registers the configured image under `/nvxhost/images/` by its +content ID, `sha256:` (`ImageId`). Registration hashes the image once; +`ImageDigest::Expect` additionally requires a digest, and `ImageDigest::Trusted` records a digest +that the caller's own policy verified without reading the file. Later backends and processes +reuse a registration that the file still matches without reading it. Provision records the image +ID and the runtime digests. Start compares each file's seal (volume, file ID, length, and +last-write time, plus the change time on Linux) with the one taken when the file was hashed, +instead of hashing again, and fails with `backend_unavailable` if anything changed, so a start +costs about the guest's boot time. A seal detects replacement or modification, not a writer that +deliberately restores timestamps. On Windows, start also keeps writers out of the files until +OpenVMM has opened them. Register a changed image again with `register_image`: unchanged content +keeps its ID, so sandboxes that use it start again. `images` lists the registrations and whether +each file still matches, `verify_image` hashes one again as a diagnostic, and `unregister_image` +refuses while a provisioned sandbox uses the image. `with_content_verification(true)` hashes the +image and runtime files again before every start, also as a diagnostic. + +Start claims the sandbox's OpenVMM log before launching, and OpenVMM inherits the claim. If the +caller dies before it records OpenVMM's identity, the next operation waits up to +`start_timeout` for that OpenVMM to open its endpoint and then terminates it; once no process +holds the log, nothing of the interrupted launch runs, so the sandbox is usable again. + +As with the direct backend, executions on one sandbox share its single control connection: a concurrent exec waits up to `control_timeout` for the running one to finish. The backend copies the guest boot console to `NvxHostBackend::console_log_path` from the process that started or last used the sandbox. OpenVMM serves one console client at a time and holds @@ -153,7 +174,8 @@ directories and environments. Snapshot/restore, image selection, and richer gues operations are not part of this profile. See [`examples/nvxhost_lifecycle.rs`](examples/nvxhost_lifecycle.rs) for a run requiring `--openvmm`, `--kernel`, `--initrd`, `--image`, `--host-library`, `--host-sha256`, -`--state-root`, `--hypervisor`, and a command after `--`. The ignored +`--state-root`, `--hypervisor`, and a command after `--`; an optional `--image-sha256` +requires the image's digest when it is registered. The ignored `tests/nvxhost_guest.rs` exercises an actual WHP guest when the corresponding `NVXHOST_TEST_*` paths and approved DLL digest are set. diff --git a/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs b/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs index dddd56b51..d24c7bea9 100644 --- a/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs +++ b/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs @@ -5,7 +5,9 @@ use std::path::PathBuf; use std::process::ExitCode; use std::sync::Arc; -use aci_edge_sandboxes::openvmm::{Hypervisor, NvxHostBackend, NvxHostConfig, OpenVmmConfig}; +use aci_edge_sandboxes::openvmm::{ + Hypervisor, ImageDigest, NvxHostBackend, NvxHostConfig, OpenVmmConfig, +}; use aci_edge_sandboxes::{AciEdgeSandbox, ExecOutcome, ExecRequest, ProvisionRequest}; fn main() -> ExitCode { @@ -25,6 +27,7 @@ fn run() -> Result<(), String> { let mut image = None; let mut library = None; let mut digest = None; + let mut image_digest = ImageDigest::Compute; let mut root = None; let mut hypervisor = None; let mut command = None; @@ -40,7 +43,10 @@ fn run() -> Result<(), String> { "--initrd" => initrd = Some(value()?), "--image" => image = Some(value()?), "--host-library" => library = Some(value()?), - "--host-sha256" => digest = Some(parse_digest(&value()?)?), + "--host-sha256" => digest = Some(parse_digest(&option, &value()?)?), + "--image-sha256" => { + image_digest = ImageDigest::Expect(parse_digest(&option, &value()?)?); + } "--state-root" => root = Some(value()?), "--hypervisor" => hypervisor = Some(value()?.parse::().map_err(describe)?), "--" => { @@ -66,8 +72,10 @@ fn run() -> Result<(), String> { let config = OpenVmmConfig::new(openvmm, kernel, initrd, hypervisor, PathBuf::from(root)); let backend = Arc::new( - NvxHostBackend::new(NvxHostConfig::new(config, image, library, digest)) - .map_err(describe)?, + NvxHostBackend::new( + NvxHostConfig::new(config, image, library, digest).with_image_digest(image_digest), + ) + .map_err(describe)?, ); let client = AciEdgeSandbox::from_shared(backend.clone()); let id = client @@ -133,14 +141,16 @@ fn run() -> Result<(), String> { Ok(()) } -fn parse_digest(hex: &str) -> Result<[u8; 32], String> { +fn parse_digest(option: &str, hex: &str) -> Result<[u8; 32], String> { if hex.len() != 64 || !hex.is_ascii() { - return Err("--host-sha256 must contain exactly 64 hexadecimal digits".to_owned()); + return Err(format!( + "{option} must contain exactly 64 hexadecimal digits" + )); } let mut digest = [0u8; 32]; for (index, byte) in digest.iter_mut().enumerate() { *byte = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16) - .map_err(|_| "--host-sha256 must be hexadecimal")?; + .map_err(|_| format!("{option} must be hexadecimal"))?; } Ok(digest) } @@ -155,9 +165,10 @@ mod tests { #[test] fn digest_requires_exact_hex_bytes() { - assert_eq!(parse_digest(&"ab".repeat(32)).unwrap(), [0xab; 32]); - assert!(parse_digest("ab").is_err()); - assert!(parse_digest(&"é".repeat(32)).is_err()); - assert!(parse_digest(&"gg".repeat(32)).is_err()); + let parse = |hex: &str| parse_digest("--image-sha256", hex); + assert_eq!(parse(&"ab".repeat(32)).unwrap(), [0xab; 32]); + assert!(parse("ab").unwrap_err().starts_with("--image-sha256 ")); + assert!(parse(&"é".repeat(32)).is_err()); + assert!(parse(&"gg".repeat(32)).is_err()); } } diff --git a/aci_edge_sandboxes/src/openvmm/images.rs b/aci_edge_sandboxes/src/openvmm/images.rs new file mode 100644 index 000000000..c86891524 --- /dev/null +++ b/aci_edge_sandboxes/src/openvmm/images.rs @@ -0,0 +1,891 @@ +//! Registry of guest images, which keeps hashing off the sandbox start path. +//! +//! ```text +//! /nvxhost/images/ +//! .lock serializes adding, removing, and using registrations +//! .json path and seal of one registered image content +//! ``` +//! +//! An image is hashed once, when it is registered, or not at all when the caller supplies a +//! digest that its own policy already verified. The registration keeps the file's [`FileSeal`]; +//! provisioning and starting a sandbox compare the file's current seal with it and fail closed +//! when they differ, rather than hashing the image again. The backend checks its OpenVMM runtime +//! files the same way: it hashes them once, when it is created, and later compares seals. + +use std::fmt; +use std::fs::{self, File}; +use std::io::{self, BufReader, Seek}; +use std::path::{Path, PathBuf}; +use std::str::FromStr; + +use serde::{Deserialize, Deserializer, Serialize, Serializer}; +use sha2_runtime::{Digest, Sha256}; + +use super::platform::{self, FileSeal}; +use super::state::{self, LockGuard}; +use crate::error::{Error, Result}; + +const RECORD_FORMAT: u32 = 1; +const LOCK_NAME: &str = ".lock"; +const RECORD_SUFFIX: &str = ".json"; +const HASH_BUFFER_BYTES: usize = 1 << 20; + +/// Content identity of a registered guest image: the SHA-256 digest of the image file. +/// +/// An ID displays and parses as `sha256:` followed by 64 lowercase hexadecimal digits. +#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub struct ImageId([u8; 32]); + +impl ImageId { + const PREFIX: &'static str = "sha256:"; + + /// Returns the ID of image content with this SHA-256 digest. + pub const fn from_sha256(digest: [u8; 32]) -> Self { + Self(digest) + } + + /// Returns the SHA-256 digest of the image content. + pub const fn sha256(&self) -> &[u8; 32] { + &self.0 + } + + /// Parses an ID, returning + /// [`ErrorCode::MalformedRequest`](crate::ErrorCode::MalformedRequest) unless it is `sha256:` + /// followed by 64 lowercase hexadecimal digits. + pub fn parse(value: &str) -> Result { + value + .strip_prefix(Self::PREFIX) + .and_then(decode_digest) + .map(Self) + .ok_or_else(|| { + Error::malformed_request(format!( + "image ID {value:?} must be sha256: followed by 64 lowercase hexadecimal \ + digits" + )) + }) + } +} + +impl fmt::Display for ImageId { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(formatter, "{}{}", Self::PREFIX, encode_hex(&self.0)) + } +} + +impl fmt::Debug for ImageId { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(formatter, "ImageId({self})") + } +} + +impl FromStr for ImageId { + type Err = Error; + + fn from_str(value: &str) -> Result { + Self::parse(value) + } +} + +impl Serialize for ImageId { + fn serialize(&self, serializer: S) -> std::result::Result { + serializer.collect_str(self) + } +} + +impl<'de> Deserialize<'de> for ImageId { + fn deserialize>(deserializer: D) -> std::result::Result { + let value = String::deserialize(deserializer)?; + Self::parse(&value).map_err(|error| serde::de::Error::custom(error.message())) + } +} + +/// How the backend establishes the content digest of an image that it registers. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +#[non_exhaustive] +pub enum ImageDigest { + /// Hashes the file. + #[default] + Compute, + /// Hashes the file and requires this SHA-256 digest. + Expect([u8; 32]), + /// Records this SHA-256 digest without reading the file. + /// + /// Use it only for content that the caller's own policy has verified and protects from + /// writers, for example a file that an installer hashed before making it read-only. + Trusted([u8; 32]), +} + +/// A registered guest image. +#[derive(Debug, Clone, PartialEq, Eq)] +#[non_exhaustive] +pub struct RegisteredImage { + /// Content identity. + pub id: ImageId, + /// Absolute path that sandboxes attach the image from. + pub path: PathBuf, + /// Length in bytes at registration. + pub length: u64, + /// Whether the caller supplied the digest instead of the backend hashing the file. + pub trusted: bool, + /// Whether the file still matches its registration. A sandbox whose image changed does not + /// start until the image is registered again. + pub intact: bool, +} + +/// Persisted registration of one image content. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub(crate) struct ImageRecord { + format: u32, + pub(crate) id: ImageId, + pub(crate) path: PathBuf, + trusted: bool, + seal: FileSeal, +} + +impl ImageRecord { + pub(crate) fn describe(&self, intact: bool) -> RegisteredImage { + RegisteredImage { + id: self.id, + path: self.path.clone(), + length: self.seal.length, + trusted: self.trusted, + intact, + } + } + + fn subject(&self) -> String { + format!("image {} at {}", self.id, self.path.display()) + } + + /// Hashes the image through `file`, a handle from [`ImageStore::open_checked`], and compares + /// the digest with the image's ID. + pub(crate) fn verify_content(&self, file: &mut File) -> Result<()> { + verify_digest(file, &self.id.0, &self.subject()) + } +} + +/// Directory of image registrations shared by every backend with the same state root. +#[derive(Debug)] +pub(crate) struct ImageStore { + dir: PathBuf, +} + +impl ImageStore { + /// Opens the registry in `dir`, creating it and restricting it to the current user if needed. + pub(crate) fn open(dir: &Path) -> io::Result { + platform::create_private_dir(dir)?; + Ok(Self { + dir: dir.to_path_buf(), + }) + } + + /// Takes the registry lock, which serializes adding and removing registrations with recording + /// sandboxes that use them. + pub(crate) fn lock(&self) -> Result { + state::lock(&self.dir.join(LOCK_NAME)) + } + + fn record_path(&self, id: &ImageId) -> PathBuf { + self.dir + .join(format!("{}{RECORD_SUFFIX}", encode_hex(&id.0))) + } + + /// Loads the registration of `id`, if there is one. + pub(crate) fn get(&self, id: &ImageId) -> Result> { + let path = self.record_path(id); + match state::read_json::(&path)? { + Some(record) if record.format != RECORD_FORMAT || record.id != *id => { + Err(Error::backend_error(format!( + "image record {} does not describe {id}", + path.display() + ))) + } + record => Ok(record), + } + } + + /// Loads every registration, ordered by ID. + pub(crate) fn list(&self) -> Result> { + let list_error = || { + Error::backend_error(format!( + "cannot list the image registry {}", + self.dir.display() + )) + }; + let mut records = Vec::new(); + for entry in fs::read_dir(&self.dir).map_err(|error| list_error().with_source(error))? { + let name = entry + .map_err(|error| list_error().with_source(error))? + .file_name(); + let Some(digest) = name + .to_str() + .and_then(|name| name.strip_suffix(RECORD_SUFFIX)) + .and_then(decode_digest) + else { + continue; + }; + // A registration removed since the listing no longer exists. + if let Some(record) = self.get(&ImageId(digest))? { + records.push(record); + } + } + records.sort_by_key(|record| record.id); + Ok(records) + } + + /// Registers the image at the absolute `path`, reusing a registration that the file still + /// matches without reading the file. + /// + /// The caller must not hold the registry lock, which this takes to record a new + /// registration. + pub(crate) fn register(&self, path: &Path, digest: ImageDigest) -> Result { + let subject = format!("image {}", path.display()); + let mut file = open(path, true, &subject)?; + let seal = seal(&file, &subject)?; + if let Some(record) = self + .list()? + .into_iter() + .find(|record| record.path == path && record.seal == seal) + { + return match digest { + ImageDigest::Expect(expected) | ImageDigest::Trusted(expected) + if record.id.0 != expected => + { + Err(Error::backend_unavailable(format!( + "{subject} is registered as {}, not {}", + record.id, + ImageId(expected) + ))) + } + _ => Ok(record), + }; + } + let (id, trusted) = match digest { + ImageDigest::Trusted(expected) => (ImageId(expected), true), + ImageDigest::Compute | ImageDigest::Expect(_) => { + let id = ImageId(hash(&mut file, &subject)?); + if self::seal(&file, &subject)? != seal { + return Err(Error::backend_unavailable(format!( + "{subject} changed while it was hashed" + ))); + } + if let ImageDigest::Expect(expected) = digest + && id.0 != expected + { + return Err(Error::backend_unavailable(format!( + "{subject} is {id}, not the expected {}", + ImageId(expected) + ))); + } + (id, false) + } + }; + let record = ImageRecord { + format: RECORD_FORMAT, + id, + path: path.to_path_buf(), + trusted, + seal, + }; + let _registry = self.lock()?; + state::write_json(&self.record_path(&id), &record)?; + Ok(record) + } + + /// Opens a registered image and checks that it still matches its registration. + /// + /// With `deny_writers`, the handle keeps writers out on Windows until it is dropped. + pub(crate) fn open_checked(&self, record: &ImageRecord, deny_writers: bool) -> Result { + open_sealed( + &record.path, + &record.seal, + deny_writers, + &record.subject(), + "it was registered; register it again", + ) + } + + /// Hashes a registered image and compares the digest with its ID. + pub(crate) fn verify(&self, record: &ImageRecord) -> Result<()> { + record.verify_content(&mut self.open_checked(record, true)?) + } + + /// Removes the registration of `id`, returning whether it existed. + /// + /// The caller holds the registry lock and has checked that no sandbox uses the image. + pub(crate) fn remove(&self, id: &ImageId) -> Result { + let path = self.record_path(id); + match fs::remove_file(&path) { + Ok(()) => Ok(true), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(false), + Err(error) => Err( + Error::backend_error(format!("cannot remove {}", path.display())) + .with_source(error), + ), + } + } +} + +/// A runtime file that the backend hashed once, with the seal that it had then. +#[derive(Debug)] +pub(crate) struct VerifiedFile { + subject: String, + path: PathBuf, + sha256: [u8; 32], + seal: FileSeal, +} + +impl VerifiedFile { + /// Hashes the file at `path`, requiring the `approved` digest if there is one. + pub(crate) fn verify( + path: &Path, + description: &str, + approved: Option<[u8; 32]>, + ) -> Result { + let subject = format!("the {description} {}", path.display()); + let mut file = open(path, true, &subject)?; + let seal = seal(&file, &subject)?; + let sha256 = hash(&mut file, &subject)?; + if self::seal(&file, &subject)? != seal { + return Err(Error::backend_unavailable(format!( + "{subject} changed while it was hashed" + ))); + } + if approved.is_some_and(|approved| approved != sha256) { + return Err(Error::backend_unavailable(format!( + "{subject} does not match the approved SHA-256" + ))); + } + Ok(Self { + subject, + path: path.to_path_buf(), + sha256, + seal, + }) + } + + pub(crate) fn sha256(&self) -> &[u8; 32] { + &self.sha256 + } + + /// Opens the file, keeping writers out on Windows until the handle is dropped, and checks + /// that it has not changed since it was hashed. + pub(crate) fn open_checked(&self) -> Result { + open_sealed( + &self.path, + &self.seal, + true, + &self.subject, + "the backend verified it; create the backend again", + ) + } + + /// Hashes the file again through `file`, a handle from [`Self::open_checked`]. + pub(crate) fn verify_content(&self, file: &mut File) -> Result<()> { + verify_digest(file, &self.sha256, &self.subject) + } +} + +fn open(path: &Path, deny_writers: bool, subject: &str) -> Result { + platform::open_sealable(path, deny_writers).map_err(|error| { + Error::backend_unavailable(format!("cannot open {subject}: {error}")).with_source(error) + }) +} + +fn seal(file: &File, subject: &str) -> Result { + platform::file_seal(file).map_err(|error| { + Error::backend_unavailable(format!("cannot inspect {subject}: {error}")).with_source(error) + }) +} + +/// Opens `path` and checks that its seal is still `expected`. +fn open_sealed( + path: &Path, + expected: &FileSeal, + deny_writers: bool, + subject: &str, + since: &str, +) -> Result { + let file = open(path, deny_writers, subject)?; + if seal(&file, subject)? == *expected { + Ok(file) + } else { + Err(Error::backend_unavailable(format!( + "{subject} changed since {since}" + ))) + } +} + +fn verify_digest(file: &mut File, expected: &[u8; 32], subject: &str) -> Result<()> { + if hash(file, subject)? == *expected { + Ok(()) + } else { + Err(Error::backend_unavailable(format!( + "{subject} no longer matches its SHA-256" + ))) + } +} + +/// Hashes a whole file from its start. +fn hash(file: &mut File, subject: &str) -> Result<[u8; 32]> { + fn digest(file: &mut File) -> io::Result<[u8; 32]> { + file.rewind()?; + let mut hasher = Sha256::new(); + io::copy( + &mut BufReader::with_capacity(HASH_BUFFER_BYTES, file), + &mut hasher, + )?; + Ok(hasher.finalize().into()) + } + digest(file).map_err(|error| { + Error::backend_unavailable(format!("cannot hash {subject}: {error}")).with_source(error) + }) +} + +/// Returns `bytes` as lowercase hexadecimal digits. +pub(crate) fn encode_hex(bytes: &[u8]) -> String { + bytes.iter().map(|byte| format!("{byte:02x}")).collect() +} + +fn decode_digest(hex: &str) -> Option<[u8; 32]> { + if hex.len() != 64 + || !hex + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return None; + } + let mut digest = [0u8; 32]; + for (index, byte) in digest.iter_mut().enumerate() { + *byte = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16).ok()?; + } + Some(digest) +} + +#[cfg(test)] +mod tests { + use std::io::Write; + use std::time::{Duration, SystemTime}; + + use super::*; + use crate::ErrorCode; + + fn sha256(bytes: &[u8]) -> [u8; 32] { + Sha256::digest(bytes).into() + } + + fn registry(root: &Path) -> ImageStore { + ImageStore::open(&root.join("images")).unwrap() + } + + fn image(root: &Path, name: &str, content: &[u8]) -> PathBuf { + let path = root.join(name); + fs::write(&path, content).unwrap(); + path + } + + /// Moves the file's modification time without changing its content. + fn touch(path: &Path) { + let file = File::options().write(true).open(path).unwrap(); + let modified = file.metadata().unwrap().modified().unwrap(); + file.set_modified(modified + Duration::from_secs(2)) + .unwrap(); + } + + fn unavailable(result: Result, expected: &str) { + let error = result.unwrap_err(); + assert_eq!(error.code(), ErrorCode::BackendUnavailable, "{error}"); + assert!(error.message().contains(expected), "{error}"); + } + + #[test] + fn image_ids_display_parse_and_serialize_as_prefixed_digests() { + let id = ImageId::from_sha256([0xab; 32]); + let text = format!("sha256:{}", "ab".repeat(32)); + assert_eq!(id.to_string(), text); + assert_eq!(ImageId::parse(&text).unwrap(), id); + assert_eq!(text.parse::().unwrap(), id); + assert_eq!(id.sha256(), &[0xab; 32]); + assert_eq!(serde_json::to_string(&id).unwrap(), format!("\"{text}\"")); + assert_eq!( + serde_json::from_str::(&format!("\"{text}\"")).unwrap(), + id + ); + for malformed in [ + String::new(), + "ab".repeat(32), + format!("sha512:{}", "ab".repeat(32)), + format!("sha256:{}", "AB".repeat(32)), + format!("sha256:{}", "ab".repeat(31)), + format!("sha256:{}0", "ab".repeat(32)), + format!("sha256:{}+0", "ab".repeat(31)), + ] { + let error = ImageId::parse(&malformed).unwrap_err(); + assert_eq!(error.code(), ErrorCode::MalformedRequest, "{malformed:?}"); + assert!(serde_json::from_str::(&format!("{malformed:?}")).is_err()); + } + } + + #[test] + fn registration_hashes_once_and_is_reused_while_the_file_is_unchanged() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + assert_eq!(record.id, ImageId(sha256(b"image content"))); + assert_eq!( + record.describe(true), + RegisteredImage { + id: record.id, + path: path.clone(), + length: 13, + trusted: false, + intact: true, + } + ); + + // Another process sees the same registration, and the expected digest matches it. + let reopened = registry(root.path()); + assert_eq!(reopened.list().unwrap(), std::slice::from_ref(&record)); + assert_eq!(reopened.get(&record.id).unwrap(), Some(record.clone())); + assert_eq!( + reopened + .register(&path, ImageDigest::Expect(record.id.0)) + .unwrap(), + record + ); + drop(reopened.open_checked(&record, true).unwrap()); + reopened.verify(&record).unwrap(); + } + + #[test] + fn trusted_digests_are_recorded_without_hashing_and_then_reused() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + // The content does not have this digest, so a registration that hashed would differ. + let claimed = [7; 32]; + let record = store + .register(&path, ImageDigest::Trusted(claimed)) + .unwrap(); + assert_eq!(record.id, ImageId(claimed)); + assert!(record.describe(true).trusted); + assert_eq!(store.register(&path, ImageDigest::Compute).unwrap(), record); + assert_eq!( + store + .register(&path, ImageDigest::Trusted(claimed)) + .unwrap(), + record + ); + // The diagnostic hash exposes a trusted digest that the content does not have. + unavailable(store.verify(&record), "no longer matches its SHA-256"); + + for conflicting in [ + ImageDigest::Trusted(sha256(b"image content")), + ImageDigest::Expect(sha256(b"image content")), + ] { + unavailable( + store.register(&path, conflicting), + &format!("is registered as {}", record.id), + ); + } + assert_eq!(store.list().unwrap(), [record]); + } + + #[test] + fn an_unexpected_digest_registers_nothing() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + unavailable( + store.register(&path, ImageDigest::Expect([1; 32])), + &format!("not the expected {}", ImageId([1; 32])), + ); + assert!(store.list().unwrap().is_empty()); + unavailable( + store.register(&root.path().join("missing.vhd"), ImageDigest::Compute), + "cannot open image", + ); + unavailable(store.register(root.path(), ImageDigest::Compute), "image"); + assert!(store.list().unwrap().is_empty()); + } + + #[test] + fn a_changed_image_fails_closed_until_it_is_registered_again() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + + // Only the timestamp moves: the content, and thus the ID, stays the same. + touch(&path); + unavailable( + store.open_checked(&record, false), + "changed since it was registered; register it again", + ); + unavailable(store.verify(&record), "changed since it was registered"); + let refreshed = store.register(&path, ImageDigest::Compute).unwrap(); + assert_eq!(refreshed.id, record.id); + assert_ne!(refreshed.seal, record.seal); + assert_eq!(store.list().unwrap(), std::slice::from_ref(&refreshed)); + drop(store.open_checked(&refreshed, true).unwrap()); + + // New content gets a new ID; the old registration stays, but no longer matches. + File::options() + .append(true) + .open(&path) + .unwrap() + .write_all(b" changed") + .unwrap(); + unavailable(store.open_checked(&refreshed, true), "changed since"); + let changed = store.register(&path, ImageDigest::Compute).unwrap(); + assert_eq!(changed.id, ImageId(sha256(b"image content changed"))); + assert_eq!(store.list().unwrap().len(), 2); + drop(store.open_checked(&changed, false).unwrap()); + } + + #[test] + fn a_replaced_file_fails_closed_even_with_identical_content_and_times() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + let modified = fs::metadata(&path).unwrap().modified().unwrap(); + let replacement = image(root.path(), "replacement.vhd", b"image content"); + File::options() + .write(true) + .open(&replacement) + .unwrap() + .set_modified(modified) + .unwrap(); + fs::remove_file(&path).unwrap(); + fs::rename(&replacement, &path).unwrap(); + unavailable(store.open_checked(&record, false), "changed since"); + } + + #[test] + fn the_same_content_moves_to_its_latest_registered_location() { + let root = tempfile::tempdir().unwrap(); + let first = image(root.path(), "first.vhd", b"image content"); + let second = image(root.path(), "second.vhd", b"image content"); + let store = registry(root.path()); + let original = store.register(&first, ImageDigest::Compute).unwrap(); + let moved = store.register(&second, ImageDigest::Compute).unwrap(); + assert_eq!(moved.id, original.id); + assert_eq!(store.get(&original.id).unwrap().unwrap().path, second); + assert_eq!(store.list().unwrap(), [moved]); + } + + #[test] + fn records_are_private_and_unrelated_files_are_ignored() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + let dir = root.path().join("images"); + for name in [".lock", ".record.json.123.tmp", "notes.json", "README"] { + fs::write(dir.join(name), b"not a record").unwrap(); + } + assert_eq!(store.list().unwrap(), std::slice::from_ref(&record)); + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + let mode = |path: PathBuf| fs::metadata(path).unwrap().permissions().mode() & 0o777; + assert_eq!(mode(dir.clone()), 0o700); + assert_eq!(mode(store.record_path(&record.id)), 0o600); + } + + fs::write(store.record_path(&record.id), b"{ truncated").unwrap(); + assert_eq!(store.list().unwrap_err().code(), ErrorCode::BackendError); + let other = ImageId([9; 32]); + fs::write( + store.record_path(&other), + serde_json::to_vec(&record).unwrap(), + ) + .unwrap(); + fs::remove_file(store.record_path(&record.id)).unwrap(); + let error = store.get(&other).unwrap_err(); + assert!(error.message().contains("does not describe"), "{error}"); + } + + #[test] + fn removal_reports_whether_a_registration_existed() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + let _registry = store.lock().unwrap(); + assert!(store.remove(&record.id).unwrap()); + assert!(!store.remove(&record.id).unwrap()); + assert_eq!(store.get(&record.id).unwrap(), None); + assert!(path.exists()); + } + + #[test] + fn runtime_files_are_hashed_once_and_then_checked_by_seal() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "vmlinux", b"kernel"); + let verified = VerifiedFile::verify(&path, "guest kernel", None).unwrap(); + assert_eq!(verified.sha256(), &sha256(b"kernel")); + assert_eq!( + VerifiedFile::verify(&path, "guest kernel", Some(sha256(b"kernel"))) + .unwrap() + .sha256(), + &sha256(b"kernel") + ); + unavailable( + VerifiedFile::verify(&path, "guest kernel", Some([0; 32])), + "the guest kernel", + ); + let mut handle = verified.open_checked().unwrap(); + verified.verify_content(&mut handle).unwrap(); + drop(handle); + + touch(&path); + unavailable( + verified.open_checked(), + "changed since the backend verified it; create the backend again", + ); + } + + #[cfg(windows)] + #[test] + fn sealed_handles_keep_writers_out_and_writers_block_registration() { + const ERROR_SHARING_VIOLATION: i32 = 32; + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + + let held = store.open_checked(&record, true).unwrap(); + let error = File::options().write(true).open(&path).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(ERROR_SHARING_VIOLATION)); + assert!(fs::remove_file(&path).is_err()); + // Readers, including a second start of the same image, still share it. + drop(File::open(&path).unwrap()); + drop(store.open_checked(&record, true).unwrap()); + drop(held); + + let writer = File::options().write(true).open(&path).unwrap(); + unavailable( + store.register(&path, ImageDigest::Compute), + "cannot open image", + ); + unavailable(store.open_checked(&record, true), "cannot open image"); + // Listing does not deny writers, so it still inspects an image that is open for writing. + drop(store.open_checked(&record, false).unwrap()); + drop(writer); + } + + /// Restoring the last-write time after a write preserves the seal on Windows, so only content + /// verification detects the change. + #[cfg(windows)] + #[test] + fn content_verification_detects_a_change_that_preserves_the_seal() { + use std::os::windows::io::AsRawHandle; + + use windows_sys::Win32::Storage::FileSystem::{ + FILE_BASIC_INFO, FileBasicInfo, GetFileInformationByHandleEx, + SetFileInformationByHandle, + }; + + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + + let mut file = File::options().read(true).write(true).open(&path).unwrap(); + let handle = file.as_raw_handle(); + let mut times = FILE_BASIC_INFO::default(); + let size = std::mem::size_of::() as u32; + // SAFETY: the handle is open and the buffer is a FILE_BASIC_INFO of the size passed. + assert_ne!( + unsafe { + GetFileInformationByHandleEx(handle, FileBasicInfo, (&raw mut times).cast(), size) + }, + 0 + ); + file.write_all(b"IMAGE").unwrap(); + // SAFETY: as above; the handle has write access. + assert_ne!( + unsafe { + SetFileInformationByHandle(handle, FileBasicInfo, (&raw const times).cast(), size) + }, + 0 + ); + drop(file); + assert_eq!(fs::read(&path).unwrap(), b"IMAGE content"); + + drop(store.open_checked(&record, true).unwrap()); + unavailable(store.verify(&record), "no longer matches its SHA-256"); + } + + #[cfg(target_os = "linux")] + #[test] + fn special_files_are_refused_without_blocking() { + use std::ffi::CString; + use std::os::unix::ffi::OsStrExt; + + let root = tempfile::tempdir().unwrap(); + let fifo = root.path().join("image.fifo"); + let name = CString::new(fifo.as_os_str().as_bytes()).unwrap(); + // SAFETY: the path is a NUL-terminated string. + assert_eq!(unsafe { libc::mkfifo(name.as_ptr(), 0o600) }, 0); + let store = registry(root.path()); + unavailable( + store.register(&fifo, ImageDigest::Compute), + "does not name a regular file", + ); + assert!(store.list().unwrap().is_empty()); + } + + #[test] + fn seal_changes_are_visible_to_new_handles() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let sealed = + |path: &Path| platform::file_seal(&platform::open_sealable(path, false).unwrap()); + let first = sealed(&path).unwrap(); + assert_eq!(sealed(&path).unwrap(), first); + assert_eq!(first.length, 13); + let modified = SystemTime::now() + Duration::from_secs(60); + File::options() + .write(true) + .open(&path) + .unwrap() + .set_modified(modified) + .unwrap(); + let second = sealed(&path).unwrap(); + assert_ne!(second.modified, first.modified); + assert_eq!((&second.file, second.volume), (&first.file, first.volume)); + } + + /// Windows security components update a file's change time when they cache its hash in an + /// extended attribute, so Windows seals ignore metadata-only changes. Linux seals keep the + /// change time, which writers cannot restore. + #[test] + fn metadata_only_changes_alter_the_seal_only_where_the_change_time_is_sealed() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let sealed = |path: &Path| { + platform::file_seal(&platform::open_sealable(path, false).unwrap()).unwrap() + }; + let before = sealed(&path); + // Let a coarse change-time clock advance. + std::thread::sleep(Duration::from_millis(50)); + let mut permissions = fs::metadata(&path).unwrap().permissions(); + permissions.set_readonly(true); + fs::set_permissions(&path, permissions.clone()).unwrap(); + let after = sealed(&path); + assert_eq!(after.modified, before.modified); + assert_eq!(after == before, cfg!(windows), "{before:?} -> {after:?}"); + // A read-only file would outlive the temporary directory on Windows. + #[cfg(windows)] + { + #[allow(clippy::permissions_set_readonly_false)] + permissions.set_readonly(false); + fs::set_permissions(&path, permissions).unwrap(); + } + } +} diff --git a/aci_edge_sandboxes/src/openvmm/mod.rs b/aci_edge_sandboxes/src/openvmm/mod.rs index 3da4ffd37..db2386914 100644 --- a/aci_edge_sandboxes/src/openvmm/mod.rs +++ b/aci_edge_sandboxes/src/openvmm/mod.rs @@ -91,6 +91,8 @@ mod artifacts; mod config; mod contract; mod filesystem; +#[cfg(feature = "nvxhost")] +mod images; mod launch; #[cfg(feature = "nvxhost")] mod native; @@ -114,7 +116,9 @@ pub use self::artifacts::Artifacts; pub use self::config::{Hypervisor, OpenVmmConfig}; pub use self::filesystem::{guest_path, resolve_guest_path}; #[cfg(feature = "nvxhost")] -pub use self::native::{NvxHostBackend, NvxHostConfig}; +pub use self::images::{ImageDigest, ImageId, RegisteredImage}; +#[cfg(feature = "nvxhost")] +pub use self::native::{NvxHostBackend, NvxHostConfig, RuntimeDigests}; use self::protocol::{ CAPABILITY_LEN, ExitCategory, GuestFeatures, MAX_ARGUMENT_BYTES, MAX_CWD_BYTES, MAX_OUTPUT_BYTES, MAX_TIMEOUT_MS, Workload, WorkloadEnvironment, @@ -215,7 +219,8 @@ impl OpenVmmBackend { /// /// Lifecycle locks serialize starts, so a launch marker seen under the lock always belongs /// to an interrupted start. Recorded identity is authoritative even before an endpoint - /// exists; legacy markers can be recovered only when their endpoint identifies the child. + /// exists; other markers can be recovered only when their endpoint identifies the child, or + /// when a released claim on the OpenVMM log proves that nothing of the launch still runs. fn recover_launch(&self, sandbox_id: &SandboxId, launch: &LaunchRecord) -> Result<()> { let interrupted = |detail: &str| { Error::backend_error(format!( @@ -225,13 +230,8 @@ impl OpenVmmBackend { let identity = match &launch.process { Some(identity) => identity.clone(), None => { - let pid = platform::endpoint_server_pid(&launch.endpoint) - .map_err(|error| interrupted("cannot be recovered yet").with_source(error))?; - let Some(pid) = pid else { - return Err(interrupted( - "has no recorded process identity or observable endpoint; cannot verify \ - that OpenVMM exited, so the launch marker is retained", - )); + let Some(pid) = self.launched_server(sandbox_id, launch)? else { + return Ok(()); }; let Some(start_time) = platform::process_start_time(pid) .map_err(|error| interrupted("cannot be recovered yet").with_source(error))? @@ -255,6 +255,52 @@ impl OpenVmmBackend { } } + /// Finds the process that serves the endpoint of an interrupted start that recorded no + /// process identity, returning `None` once the launch provably has no running process. + /// + /// Without a claim on the OpenVMM log, an absent endpoint proves nothing. With one, OpenVMM + /// shares the claim, so a held claim means that OpenVMM has yet to open its endpoint or to + /// exit; either is awaited for up to the start timeout. + fn launched_server( + &self, + sandbox_id: &SandboxId, + launch: &LaunchRecord, + ) -> Result> { + let interrupted = |detail: &str| { + Error::backend_error(format!( + "an interrupted start of sandbox {sandbox_id} {detail}" + )) + }; + let deadline = Instant::now() + self.config.start_timeout; + loop { + let pid = platform::endpoint_server_pid(&launch.endpoint) + .map_err(|error| interrupted("cannot be recovered yet").with_source(error))?; + if pid.is_some() { + return Ok(pid); + } + if !launch.log_claimed { + return Err(interrupted( + "has no recorded process identity or observable endpoint; cannot verify that \ + OpenVMM exited, so the launch marker is retained", + )); + } + let log = self.store.log_path(sandbox_id); + if platform::launch_log_released(&log) + .map_err(|error| interrupted("cannot be recovered yet").with_source(error))? + { + return Ok(None); + } + if Instant::now() >= deadline { + return Err(interrupted(&format!( + "has no recorded process identity or observable endpoint, and its OpenVMM log \ + {} is missing or still held; the launch marker is retained", + log.display() + ))); + } + thread::sleep(Duration::from_millis(50)); + } + } + fn session_error(&self, sandbox_id: &SandboxId, error: SessionError) -> Error { match error { SessionError::ProcessExited => { @@ -587,6 +633,7 @@ impl Backend for OpenVmmBackend { format: STATE_FORMAT, endpoint: endpoint.clone(), process: None, + log_claimed: false, }, )?; let started = Instant::now(); @@ -632,6 +679,7 @@ impl Backend for OpenVmmBackend { format: STATE_FORMAT, endpoint: endpoint.clone(), process: Some(ProcessIdentity { pid, start_time }), + log_claimed: false, }; if let Err(error) = self .store diff --git a/aci_edge_sandboxes/src/openvmm/native.rs b/aci_edge_sandboxes/src/openvmm/native.rs index 674226a65..b01c0c40a 100644 --- a/aci_edge_sandboxes/src/openvmm/native.rs +++ b/aci_edge_sandboxes/src/openvmm/native.rs @@ -9,20 +9,23 @@ use std::thread; use std::time::{Duration, Instant}; use prost::Message; -use sha2_runtime::{Digest, Sha256}; use super::artifacts::absolute; use super::config::{OpenVmmConfig, validate_unix_socket_path}; +use super::images::{ + ImageDigest, ImageId, ImageRecord, ImageStore, RegisteredImage, VerifiedFile, encode_hex, +}; use super::platform::{self, Transport}; use super::process; use super::state::{ - BOOT_SOCKET_NAME, LaunchRecord, NATIVE_BACKEND_KEY, NativeArtifactRecord, ProcessIdentity, - RuntimeRecord, SOCKET_NAME, STATE_FORMAT, SandboxRecord, StateStore, remove_if_present, + BOOT_SOCKET_NAME, IMAGES_NAME, LaunchRecord, NATIVE_BACKEND_KEY, NativeArtifactRecord, + ProcessIdentity, RuntimeRecord, SOCKET_NAME, STATE_FORMAT, SandboxRecord, StateStore, + remove_if_present, }; use super::{OpenVmmBackend, RunState}; use crate::backend::{Backend, ExecControl, ExecIo}; use crate::capabilities::Capabilities; -use crate::error::{Error, Result}; +use crate::error::{Error, ErrorCode, Result}; use crate::exec::{Completion, ExecOutcome}; use crate::id::SandboxId; use crate::model::{ @@ -41,17 +44,35 @@ const CONSOLE_POLL: Duration = Duration::from_millis(250); /// Artifact paths and approved native-library digest for an image-backed guest. /// /// The image is a caller-prepared GPT disk, not a container image reference. This backend -/// creates no disks, filesystem mappings, or network devices. +/// creates no disks, filesystem mappings, or network devices. Construct it with +/// [`NvxHostConfig::new`] and its builder methods, which keep callers compatible as options +/// are added. #[derive(Debug, Clone)] +#[non_exhaustive] pub struct NvxHostConfig { /// OpenVMM and guest boot artifacts, hypervisor, state root, and deadlines. pub openvmm: OpenVmmConfig, /// GPT image attached read-only as the distro block device. + /// + /// The backend registers the image when it is created, and the sandboxes that it provisions + /// refer to the image by its [`ImageId`]. pub image: PathBuf, + /// How the backend establishes the content digest of `image` when it registers it. + pub image_digest: ImageDigest, /// Absolute path to a separately installed nvxhost DLL or shared library. pub library: PathBuf, /// SHA-256 approved by the caller's independent artifact policy. pub library_sha256: [u8; 32], + /// SHA-256 digests of the OpenVMM runtime files approved by the caller's policy, if any. + /// + /// The backend hashes the runtime files once, when it is created, and requires these digests + /// when they are set. + pub runtime_sha256: Option, + /// Hashes the image and the runtime files again before every start. + /// + /// Otherwise a start compares only the files' seals with those taken when they were hashed. + /// This diagnostic reads every file in full, which adds the hashing time to each start. + pub content_verification: bool, /// Requests verbose guest kernel diagnostics on the boot console. pub guest_debug: bool, } @@ -67,12 +88,36 @@ impl NvxHostConfig { Self { openvmm, image: image.into(), + image_digest: ImageDigest::Compute, library: library.into(), library_sha256, + runtime_sha256: None, + content_verification: false, guest_debug: false, } } + /// Sets how the backend establishes the digest of the image that it registers. + #[must_use] + pub fn with_image_digest(mut self, digest: ImageDigest) -> Self { + self.image_digest = digest; + self + } + + /// Requires the OpenVMM runtime files to have these approved digests. + #[must_use] + pub fn with_runtime_digests(mut self, digests: RuntimeDigests) -> Self { + self.runtime_sha256 = Some(digests); + self + } + + /// Hashes the image and the runtime files again before every start, as a diagnostic. + #[must_use] + pub fn with_content_verification(mut self, enabled: bool) -> Self { + self.content_verification = enabled; + self + } + /// Enables verbose guest boot diagnostics without changing the guest lifecycle. #[must_use] pub fn with_guest_debug(mut self, enabled: bool) -> Self { @@ -81,6 +126,17 @@ impl NvxHostConfig { } } +/// SHA-256 digests of the OpenVMM executable, guest kernel, and guest initramfs. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RuntimeDigests { + /// Digest of the OpenVMM executable. + pub openvmm: [u8; 32], + /// Digest of the guest kernel. + pub kernel: [u8; 32], + /// Digest of the guest initramfs. + pub initrd: [u8; 32], +} + /// Owns sandbox state and OpenVMM processes; the private native library owns only /// the launch arguments and authenticated guest channel. #[derive(Debug)] @@ -88,9 +144,65 @@ pub struct NvxHostBackend { config: NvxHostConfig, base: OpenVmmBackend, host: Arc, + images: ImageStore, + image: ImageId, + runtime: RuntimeFiles, console_pumps: Mutex>, } +/// The OpenVMM runtime files, hashed once, when the backend is created. +#[derive(Debug)] +struct RuntimeFiles { + openvmm: VerifiedFile, + kernel: VerifiedFile, + initrd: VerifiedFile, +} + +impl RuntimeFiles { + fn verify(config: &OpenVmmConfig, approved: Option) -> Result { + Ok(Self { + openvmm: VerifiedFile::verify( + &config.openvmm, + "OpenVMM executable", + approved.map(|digests| digests.openvmm), + )?, + kernel: VerifiedFile::verify( + &config.kernel, + "guest kernel", + approved.map(|digests| digests.kernel), + )?, + initrd: VerifiedFile::verify( + &config.initrd, + "guest initramfs", + approved.map(|digests| digests.initrd), + )?, + }) + } + + fn digests(&self) -> RuntimeDigests { + RuntimeDigests { + openvmm: *self.openvmm.sha256(), + kernel: *self.kernel.sha256(), + initrd: *self.initrd.sha256(), + } + } + + /// Opens every file, keeping writers out on Windows while the handles live, and checks that + /// none changed since it was hashed. With `rehash`, it also hashes every file again. + fn open_checked(&self, rehash: bool) -> Result> { + [&self.openvmm, &self.kernel, &self.initrd] + .into_iter() + .map(|verified| { + let mut file = verified.open_checked()?; + if rehash { + verified.verify_content(&mut file)?; + } + Ok(file) + }) + .collect() + } +} + /// Copies the boot console of one OpenVMM launch into the sandbox's console log. #[derive(Debug)] struct ConsolePump { @@ -199,7 +311,12 @@ struct ExecJob { } impl NvxHostBackend { - /// Opens the state root and verifies the explicitly supplied native asset. + /// Opens the state root, verifies the explicitly supplied native asset, hashes the OpenVMM + /// runtime files, and registers the configured image. + /// + /// Creating a backend reads every runtime file in full. It reads the image only if no + /// registration that the file still matches exists and `image_digest` is not + /// [`ImageDigest::Trusted`]. pub fn new(mut config: NvxHostConfig) -> Result { config.openvmm = config.openvmm.normalized()?; config.image = absolute(&config.image)?; @@ -215,6 +332,16 @@ impl NvxHostBackend { )) .with_source(error) })?; + let images_dir = store_root.join(IMAGES_NAME); + let images = ImageStore::open(&images_dir).map_err(|error| { + Error::backend_unavailable(format!( + "cannot open the image registry {}", + images_dir.display() + )) + .with_source(error) + })?; + let runtime = RuntimeFiles::verify(&config.openvmm, config.runtime_sha256)?; + let image = images.register(&config.image, config.image_digest)?.id; let base = OpenVmmBackend { config: config.openvmm.clone(), store, @@ -223,6 +350,9 @@ impl NvxHostBackend { config, base, host, + images, + image, + runtime, console_pumps: Mutex::new(HashMap::new()), }) } @@ -247,6 +377,81 @@ impl NvxHostBackend { self.base.store.outcome_path(sandbox_id) } + /// Returns the ID of the configured image, which the sandboxes that this backend provisions + /// boot. + pub fn image_id(&self) -> ImageId { + self.image + } + + /// Returns the digests of the OpenVMM runtime files, computed when the backend was created. + pub fn runtime_digests(&self) -> RuntimeDigests { + self.runtime.digests() + } + + /// Registers the image at `path` in the state root's image registry. + /// + /// A registration that the file still matches is reused without reading the file. Otherwise + /// the image is hashed, or recorded with an [`ImageDigest::Trusted`] digest, and its + /// registration replaces any earlier one of the same content. Registering a changed image + /// again lets the sandboxes that boot it start again if its content is unchanged. + pub fn register_image( + &self, + path: impl AsRef, + digest: ImageDigest, + ) -> Result { + let path = absolute(path.as_ref())?; + Ok(self.images.register(&path, digest)?.describe(true)) + } + + /// Lists the registered images and whether each file still matches its registration. + pub fn images(&self) -> Result> { + Ok(self + .images + .list()? + .into_iter() + .map(|record| { + let intact = self.images.open_checked(&record, false).is_ok(); + record.describe(intact) + }) + .collect()) + } + + /// Hashes a registered image and checks that it still has the content that its ID names. + /// + /// This diagnostic reads the whole image, whereas provisioning and starting compare only the + /// image's seal with its registration. + pub fn verify_image(&self, id: &ImageId) -> Result<()> { + let record = self + .images + .get(id)? + .ok_or_else(|| Error::backend_unavailable(format!("image {id} is not registered")))?; + self.images.verify(&record) + } + + /// Removes the registration of an image, returning whether it existed. The file stays. + /// + /// Fails with [`ErrorCode::PolicyValidation`] while a provisioned sandbox refers to the image. + /// Removing the registration of the configured image makes provisioning fail until the image + /// is registered again. + pub fn unregister_image(&self, id: &ImageId) -> Result { + let _registry = self.images.lock()?; + let image = id.to_string(); + for sandbox_id in self.base.store.sandbox_ids()? { + let record = match self.base.store.load(&sandbox_id) { + Ok(record) => record, + // The sandbox was deprovisioned after the listing. + Err(error) if error.code() == ErrorCode::StaleId => continue, + Err(error) => return Err(error), + }; + if record.native.is_some_and(|native| native.image == image) { + return Err(Error::policy_validation(format!( + "image {id} is used by sandbox {sandbox_id}; deprovision the sandbox first" + ))); + } + } + self.images.remove(id) + } + /// Collects the guest's bounded, non-follow log snapshot through ttrpc. pub fn guest_logs(&self, sandbox_id: &SandboxId) -> Result> { let (runtime, capability) = self.running(sandbox_id)?; @@ -383,38 +588,84 @@ impl NvxHostBackend { Ok(()) } - fn artifact_record(&self) -> Result { - Ok(NativeArtifactRecord { - image: self.config.image.clone(), - image_sha256: file_digest(&self.config.image)?, - kernel_sha256: file_digest(&self.config.openvmm.kernel)?, - initrd_sha256: file_digest(&self.config.openvmm.initrd)?, - library_sha256: encode_digest(&self.config.library_sha256), - }) + fn artifact_record(&self) -> NativeArtifactRecord { + let digests = self.runtime.digests(); + NativeArtifactRecord { + image: self.image.to_string(), + openvmm_sha256: encode_hex(&digests.openvmm), + kernel_sha256: encode_hex(&digests.kernel), + initrd_sha256: encode_hex(&digests.initrd), + library_sha256: encode_hex(&self.config.library_sha256), + } } - fn verify_record(&self, record: &SandboxRecord, for_launch: bool) -> Result<()> { + /// Returns a sandbox's artifact identity after checking that it uses this backend's host + /// library. + fn native_record<'a>(&self, record: &'a SandboxRecord) -> Result<&'a NativeArtifactRecord> { let recorded = record.native.as_ref().ok_or_else(|| { Error::backend_error("the native sandbox has no pinned artifact identity") })?; - if recorded.image != self.config.image - || recorded.library_sha256 != encode_digest(&self.config.library_sha256) - { + if recorded.library_sha256 != encode_hex(&self.config.library_sha256) { return Err(Error::backend_unavailable( - "the native sandbox's image or host library differs from its provisioned identity", + "the native sandbox was provisioned with a different host library", )); } - if for_launch && *recorded != self.artifact_record()? { + Ok(recorded) + } + + /// Checks the configured image and the runtime files against their seals. + fn check_configured_artifacts(&self) -> Result<()> { + let image = self.images.get(&self.image)?.ok_or_else(|| { + Error::backend_unavailable(format!( + "image {} is no longer registered; register it again", + self.image + )) + })?; + self.images.open_checked(&image, false)?; + self.runtime.open_checked(false)?; + Ok(()) + } + + /// Checks a sandbox's artifacts before it launches, without hashing them unless content + /// verification is enabled, and returns its image registration. + /// + /// The returned handles keep writers out of the files on Windows until they are dropped, after + /// OpenVMM has opened the files. + fn launch_artifacts(&self, record: &SandboxRecord) -> Result<(ImageRecord, Vec)> { + let recorded = self.native_record(record)?; + let digests = self.runtime.digests(); + if recorded.openvmm_sha256 != encode_hex(&digests.openvmm) + || recorded.kernel_sha256 != encode_hex(&digests.kernel) + || recorded.initrd_sha256 != encode_hex(&digests.initrd) + { return Err(Error::backend_unavailable( - "the native sandbox's boot artifacts changed since provision", + "the native sandbox was provisioned with different OpenVMM runtime files", )); } - Ok(()) + let id = ImageId::parse(&recorded.image).map_err(|_| { + Error::backend_error(format!( + "the native sandbox records a malformed image ID {:?}", + recorded.image + )) + })?; + let image = self.images.get(&id)?.ok_or_else(|| { + Error::backend_unavailable(format!( + "image {id} of the native sandbox is not registered; register it again" + )) + })?; + let rehash = self.config.content_verification; + let mut held = self.runtime.open_checked(rehash)?; + let mut file = self.images.open_checked(&image, true)?; + if rehash { + image.verify_content(&mut file)?; + } + held.push(file); + Ok((image, held)) } fn running(&self, sandbox_id: &SandboxId) -> Result<(RuntimeRecord, [u8; 32])> { let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; - self.verify_record(&record, false)?; + self.native_record(&record)?; match self.base.reconcile(sandbox_id)? { RunState::Provisioned => Err(Error::not_started(format!( "sandbox {sandbox_id} is not running" @@ -520,13 +771,7 @@ impl Backend for NvxHostBackend { fn probe(&self) -> Result<()> { self.base.probe()?; - if !self.config.image.is_file() { - return Err(Error::backend_unavailable(format!( - "NVX image not found: {}", - self.config.image.display() - ))); - } - Ok(()) + self.check_configured_artifacts() } fn validate_provision(&self, request: &ProvisionRequest) -> Result<()> { @@ -565,13 +810,17 @@ impl Backend for NvxHostBackend { fn provision(&self, request: &ProvisionRequest) -> Result { self.validate_provision(request)?; - self.probe()?; + self.base.probe()?; + // Unregistering an image waits for this lock, so the image stays registered until the + // sandbox that refers to it is recorded. + let _registry = self.images.lock()?; + self.check_configured_artifacts()?; let record = SandboxRecord { format: STATE_FORMAT, backend: BACKEND_KEY.to_owned(), network: None, filesystem: None, - native: Some(self.artifact_record()?), + native: Some(self.artifact_record()), memory_mib: request .microvm .provision @@ -600,8 +849,9 @@ impl Backend for NvxHostBackend { // A listener of a launch that exited on its own ends promptly. Retire it now so that the // new launch's listener connects before the guest writes its first boot output. let _ = self.finish_console_pump(sandbox_id); - self.verify_record(&record, true)?; - self.probe()?; + // `_held` keeps writers out of the checked files on Windows until OpenVMM has opened them. + let (image, _held) = self.launch_artifacts(&record)?; + self.base.probe()?; let mut capability = [0u8; 32]; while capability == [0; 32] { @@ -620,7 +870,7 @@ impl Backend for NvxHostBackend { let mut arguments = self.host.launch_arguments(&LaunchInputs { kernel: &self.config.openvmm.kernel, initrd: &self.config.openvmm.initrd, - image: &self.config.image, + image: &image.path, control: &endpoint, boot: &boot, hypervisor: self.config.openvmm.hypervisor.as_str(), @@ -634,12 +884,18 @@ impl Backend for NvxHostBackend { self.base.store.write_capability(sandbox_id, &capability)?; let log = self.base.store.create_log(sandbox_id)?; + // OpenVMM inherits the log and this claim, so if this process dies before it records + // OpenVMM's identity, recovery can still tell whether anything of the launch runs. + platform::claim_launch_log(&log).map_err(|error| { + Error::backend_error("cannot claim the OpenVMM log").with_source(error) + })?; self.base.store.write_launch( sandbox_id, &LaunchRecord { format: STATE_FORMAT, endpoint: endpoint.clone(), process: None, + log_claimed: true, }, )?; let started = Instant::now(); @@ -684,6 +940,7 @@ impl Backend for NvxHostBackend { format: STATE_FORMAT, endpoint, process: Some(ProcessIdentity { pid, start_time }), + log_claimed: true, }, ) .and_then(|()| self.base.store.write_runtime(sandbox_id, &runtime)) @@ -782,7 +1039,7 @@ impl Backend for NvxHostBackend { fn stop(&self, sandbox_id: &SandboxId) -> Result { let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; - self.verify_record(&record, false)?; + self.native_record(&record)?; let RunState::Running(runtime) = self.base.reconcile(sandbox_id)? else { return Err(Error::already_stopped(format!( "sandbox {sandbox_id} is not running" @@ -847,7 +1104,7 @@ impl Backend for NvxHostBackend { fn deprovision(&self, sandbox_id: &SandboxId) -> Result { let (guard, record) = self.base.store.lock_and_load(sandbox_id)?; - self.verify_record(&record, false)?; + self.native_record(&record)?; if let RunState::Running(_) = self.base.reconcile(sandbox_id)? { return Err(Error::already_started(format!( "sandbox {sandbox_id} is running; stop it before deprovisioning" @@ -878,21 +1135,6 @@ fn capabilities() -> Capabilities { capabilities } -fn encode_digest(bytes: &[u8; 32]) -> String { - bytes.iter().map(|byte| format!("{byte:02x}")).collect() -} - -fn file_digest(path: &Path) -> Result { - let mut file = File::open(path).map_err(|error| { - Error::backend_unavailable(format!("cannot open {}", path.display())).with_source(error) - })?; - let mut hasher = Sha256::new(); - io::copy(&mut file, &mut hasher).map_err(|error| { - Error::backend_unavailable(format!("cannot hash {}", path.display())).with_source(error) - })?; - Ok(format!("{:x}", hasher.finalize())) -} - fn finish_session(session: Session, result: Result, deadline: Option) -> Result { match (result, session.close(deadline)) { (Ok(value), Ok(())) => Ok(value), @@ -1330,4 +1572,124 @@ mod tests { ErrorCode::PolicyValidation ); } + + fn host_hypervisor() -> super::super::Hypervisor { + if cfg!(windows) { + super::super::Hypervisor::Whp + } else { + super::super::Hypervisor::Kvm + } + } + + #[test] + fn a_launch_log_claim_lasts_until_the_launcher_and_openvmm_have_exited() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("openvmm.log"); + assert!( + !platform::launch_log_released(&path).unwrap(), + "a missing log proves nothing" + ); + let log = File::create(&path).unwrap(); + platform::claim_launch_log(&log).unwrap(); + assert!(!platform::launch_log_released(&path).unwrap()); + + // A long-running child stands in for OpenVMM, which inherits the log as its output. + #[cfg(windows)] + let (program, arguments) = ( + PathBuf::from(std::env::var_os("SystemRoot").unwrap()).join(r"System32\PING.EXE"), + ["-n", "60", "127.0.0.1"].map(std::ffi::OsString::from), + ); + #[cfg(not(windows))] + let (program, arguments) = (PathBuf::from("sleep"), [std::ffi::OsString::from("60")]); + let config = OpenVmmConfig::new( + program, + "vmlinux", + "initramfs.cpio.gz", + host_hypervisor(), + directory.path(), + ); + let launched = + process::spawn(&config, &arguments, &[1; 32], log, directory.path()).unwrap(); + assert!( + !platform::launch_log_released(&path).unwrap(), + "the child's inherited copy no longer holds the claim" + ); + assert!(process::kill_child(launched)); + let deadline = Instant::now() + Duration::from_secs(10); + while !platform::launch_log_released(&path).unwrap() { + assert!(Instant::now() < deadline, "the claim outlived its holders"); + thread::sleep(Duration::from_millis(20)); + } + } + + #[test] + fn claimed_launch_markers_are_cleared_only_once_nothing_holds_the_log() { + let directory = tempfile::tempdir().unwrap(); + let mut config = OpenVmmConfig::new( + "openvmm", + "vmlinux", + "initramfs.cpio.gz", + host_hypervisor(), + directory.path(), + ); + config.start_timeout = Duration::from_millis(200); + let store = StateStore::open_for(&directory.path().join(BACKEND_KEY), BACKEND_KEY).unwrap(); + let base = OpenVmmBackend { config, store }; + let sandbox_id = SandboxId::generate().unwrap(); + let record = SandboxRecord { + format: STATE_FORMAT, + backend: BACKEND_KEY.to_owned(), + network: None, + filesystem: None, + native: None, + memory_mib: 256, + workload_uid: 65534, + workload_gid: 65534, + create_workload_account: false, + hostname: "nvx-sandbox".to_owned(), + }; + base.store.create(&sandbox_id, &record).unwrap(); + let endpoint = platform::control_endpoint(&base.store.socket_path(&sandbox_id)).unwrap(); + let marker = |log_claimed| LaunchRecord { + format: STATE_FORMAT, + endpoint: endpoint.clone(), + process: None, + log_claimed, + }; + let retained = |expected: &str| { + let Err(error) = base.reconcile(&sandbox_id) else { + panic!("an unproven launch marker was cleared"); + }; + assert_eq!(error.code(), ErrorCode::BackendError, "{error}"); + assert!(error.message().contains(expected), "{error}"); + assert!(base.store.launch(&sandbox_id).unwrap().is_some()); + }; + + // A claim whose log is missing proves nothing. + base.store.write_launch(&sandbox_id, &marker(true)).unwrap(); + retained("is missing or still held"); + + // Without a claim, an absent endpoint never proves that the launch exited. + drop(base.store.create_log(&sandbox_id).unwrap()); + base.store + .write_launch(&sandbox_id, &marker(false)) + .unwrap(); + retained("cannot verify that OpenVMM exited"); + + // A held claim is awaited for the start timeout, and the marker then retained. + let log = base.store.create_log(&sandbox_id).unwrap(); + platform::claim_launch_log(&log).unwrap(); + base.store.write_launch(&sandbox_id, &marker(true)).unwrap(); + let waited = Instant::now(); + retained("is missing or still held"); + assert!(waited.elapsed() >= Duration::from_millis(200)); + + // A released claim proves that nothing of the launch runs. + drop(log); + assert!(matches!( + base.reconcile(&sandbox_id).unwrap(), + RunState::Provisioned + )); + assert!(base.store.launch(&sandbox_id).unwrap().is_none()); + } } diff --git a/aci_edge_sandboxes/src/openvmm/platform/linux.rs b/aci_edge_sandboxes/src/openvmm/platform/linux.rs index 76d1449d6..522892fbc 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/linux.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/linux.rs @@ -245,6 +245,59 @@ pub(crate) fn file_identity(path: &Path) -> io::Result<(u64, u64)> { Ok((metadata.dev(), metadata.ino())) } +#[cfg(feature = "nvxhost")] +pub(crate) use seal::{file_seal, open_sealable}; + +#[cfg(feature = "nvxhost")] +mod seal { + use std::fs::{File, OpenOptions}; + use std::io; + use std::os::unix::fs::{MetadataExt, OpenOptionsExt}; + use std::path::Path; + + use crate::openvmm::platform::FileSeal; + + /// Device and inode numbers, size, and modification and change times in nanoseconds. + const SCHEME: &str = "unix-file-v1"; + + /// Opens a file to seal it. + /// + /// Linux has no mandatory share modes, so `deny_writers` has no effect: comparing seals + /// taken from the handle detects a writer instead. + pub(crate) fn open_sealable(path: &Path, _deny_writers: bool) -> io::Result { + // A FIFO would otherwise block the open until a writer appears. + OpenOptions::new() + .read(true) + .custom_flags(libc::O_NONBLOCK) + .open(path) + } + + /// Returns the seal of an open regular file. + pub(crate) fn file_seal(file: &File) -> io::Result { + let metadata = file.metadata()?; + if !metadata.is_file() { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "the path does not name a regular file", + )); + } + Ok(FileSeal { + scheme: SCHEME.to_owned(), + volume: metadata.dev(), + file: format!("{:x}", metadata.ino()), + length: metadata.size(), + modified: nanoseconds(metadata.mtime(), metadata.mtime_nsec()), + changed: Some(nanoseconds(metadata.ctime(), metadata.ctime_nsec())), + }) + } + + fn nanoseconds(seconds: i64, nanoseconds: i64) -> i64 { + seconds + .saturating_mul(1_000_000_000) + .saturating_add(nanoseconds) + } +} + /// Returns the control endpoint OpenVMM should listen on for a sandbox. pub(crate) fn control_endpoint(socket_path: &Path) -> io::Result { socket_path @@ -300,6 +353,37 @@ pub(crate) fn endpoint_server_pid(endpoint: &str) -> io::Result> { } } +/// Claims a fresh OpenVMM log for one launch. +/// +/// The claim is an exclusive lock of the log's open file description, which OpenVMM shares +/// through its inherited standard output and error, so the claim lasts until both the launching +/// process and OpenVMM have exited. +#[cfg(feature = "nvxhost")] +pub(crate) fn claim_launch_log(log: &fs::File) -> io::Result<()> { + log.try_lock().map_err(|error| match error { + fs::TryLockError::WouldBlock => io::Error::new( + io::ErrorKind::WouldBlock, + "another process holds the OpenVMM log", + ), + fs::TryLockError::Error(error) => error, + }) +} + +/// Returns whether no process holds the claim on an OpenVMM log, which proves that the launch +/// that claimed it has no running process. A missing log proves nothing. +pub(crate) fn launch_log_released(path: &Path) -> io::Result { + let log = match fs::File::open(path) { + Ok(log) => log, + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false), + Err(error) => return Err(error), + }; + match log.try_lock() { + Ok(()) => Ok(true), + Err(fs::TryLockError::WouldBlock) => Ok(false), + Err(fs::TryLockError::Error(error)) => Err(error), + } +} + fn connect_nonblocking(path: &str) -> io::Result { // SAFETY: socket has no memory-safety preconditions. let descriptor = unsafe { diff --git a/aci_edge_sandboxes/src/openvmm/platform/mod.rs b/aci_edge_sandboxes/src/openvmm/platform/mod.rs index e1694540b..86fdac258 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/mod.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/mod.rs @@ -6,6 +6,9 @@ use std::io; use std::time::Duration; +#[cfg(feature = "nvxhost")] +use serde::{Deserialize, Serialize}; + #[cfg(target_os = "linux")] mod linux; #[cfg(not(any(target_os = "linux", windows)))] @@ -30,3 +33,31 @@ pub(crate) trait Transport: Send { /// Writes all of `data`, waiting at most `timeout` for each write to complete. fn write_all(&mut self, data: &[u8], timeout: Option) -> io::Result<()>; } + +/// Identity of a regular file and of its last change, compared instead of hashing the file again. +/// +/// Replacing the file or writing its data changes the seal. A writer that restores the last-write +/// time afterwards can preserve it on Windows, and a write within the file system's timestamp +/// granularity can preserve it on Linux, so a seal detects accidental change rather than +/// deliberate tampering. On Linux, any metadata change also changes the seal, because the seal +/// includes the change time, which writers cannot restore. Windows seals omit the change time: +/// writers can set it, and the system updates it when security components cache a file's hash in +/// its extended attributes, which would fail unchanged files closed. +#[cfg(feature = "nvxhost")] +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub(crate) struct FileSeal { + /// Platform scheme of the remaining fields. + pub(crate) scheme: String, + /// Volume serial number or device number. + pub(crate) volume: u64, + /// File ID or inode number, in hexadecimal. + pub(crate) file: String, + /// Length in bytes. + pub(crate) length: u64, + /// Last data modification, in the scheme's time unit. + pub(crate) modified: i64, + /// Last data or metadata change, in the scheme's time unit, where the scheme seals it. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(crate) changed: Option, +} diff --git a/aci_edge_sandboxes/src/openvmm/platform/other.rs b/aci_edge_sandboxes/src/openvmm/platform/other.rs index 4af82037a..0e555d17f 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/other.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/other.rs @@ -29,6 +29,16 @@ pub(crate) fn file_identity(_path: &Path) -> io::Result<(u64, u64)> { Err(unsupported()) } +#[cfg(feature = "nvxhost")] +pub(crate) fn open_sealable(_path: &Path, _deny_writers: bool) -> io::Result { + Err(unsupported()) +} + +#[cfg(feature = "nvxhost")] +pub(crate) fn file_seal(_file: &fs::File) -> io::Result { + Err(unsupported()) +} + pub(crate) fn detach(_command: &mut Command, _breakaway_from_job: bool) {} pub(crate) fn probe_hypervisor(_hypervisor: Hypervisor) -> Result<(), String> { @@ -57,3 +67,12 @@ pub(crate) fn connect_endpoint( pub(crate) fn endpoint_server_pid(_endpoint: &str) -> io::Result> { Ok(None) } + +#[cfg(feature = "nvxhost")] +pub(crate) fn claim_launch_log(_log: &fs::File) -> io::Result<()> { + Err(unsupported()) +} + +pub(crate) fn launch_log_released(_path: &Path) -> io::Result { + Err(unsupported()) +} diff --git a/aci_edge_sandboxes/src/openvmm/platform/windows.rs b/aci_edge_sandboxes/src/openvmm/platform/windows.rs index e8b8e3a39..ad3b277f1 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/windows.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/windows.rs @@ -11,13 +11,13 @@ use std::{mem, ptr}; use windows_sys::Win32::Foundation::{ ERROR_ACCESS_DENIED, ERROR_BROKEN_PIPE, ERROR_FILE_NOT_FOUND, ERROR_INVALID_PARAMETER, ERROR_IO_PENDING, ERROR_MORE_DATA, ERROR_NO_DATA, ERROR_OPERATION_ABORTED, ERROR_PIPE_BUSY, - ERROR_PIPE_NOT_CONNECTED, FILETIME, FreeLibrary, GENERIC_READ, GENERIC_WRITE, GetLastError, - HANDLE, HANDLE_FLAG_INHERIT, INVALID_HANDLE_VALUE, SetHandleInformation, WAIT_OBJECT_0, - WAIT_TIMEOUT, + ERROR_PIPE_NOT_CONNECTED, ERROR_SHARING_VIOLATION, FILETIME, FreeLibrary, GENERIC_READ, + GENERIC_WRITE, GetLastError, HANDLE, HANDLE_FLAG_INHERIT, INVALID_HANDLE_VALUE, + SetHandleInformation, WAIT_OBJECT_0, WAIT_TIMEOUT, }; use windows_sys::Win32::Storage::FileSystem::{ BY_HANDLE_FILE_INFORMATION, CreateFileW, FILE_FLAG_BACKUP_SEMANTICS, FILE_FLAG_OVERLAPPED, - GetFileInformationByHandle, OPEN_EXISTING, ReadFile, SECURITY_IDENTIFICATION, + FILE_SHARE_READ, GetFileInformationByHandle, OPEN_EXISTING, ReadFile, SECURITY_IDENTIFICATION, SECURITY_SQOS_PRESENT, WriteFile, }; use windows_sys::Win32::System::IO::{CancelIoEx, GetOverlappedResult, OVERLAPPED}; @@ -448,6 +448,97 @@ pub(crate) fn file_identity(path: &Path) -> io::Result<(u64, u64)> { )) } +#[cfg(feature = "nvxhost")] +pub(crate) use seal::{file_seal, open_sealable}; + +#[cfg(feature = "nvxhost")] +mod seal { + use std::fs::{File, OpenOptions}; + use std::io; + use std::mem; + use std::os::windows::fs::OpenOptionsExt; + use std::os::windows::io::AsRawHandle; + use std::path::Path; + + use windows_sys::Win32::Foundation::HANDLE; + use windows_sys::Win32::Storage::FileSystem::{ + FILE_BASIC_INFO, FILE_ID_INFO, FILE_INFO_BY_HANDLE_CLASS, FILE_READ_ATTRIBUTES, + FILE_SHARE_DELETE, FILE_SHARE_READ, FILE_SHARE_WRITE, FILE_STANDARD_INFO, FileBasicInfo, + FileIdInfo, FileStandardInfo, GetFileInformationByHandleEx, + }; + + use crate::openvmm::platform::FileSeal; + + /// Volume serial number, 128-bit file ID, end of file, and last-write time in 100-nanosecond + /// units. + const SCHEME: &str = "windows-file-v1"; + + /// Opens a file to seal it. + /// + /// With `deny_writers`, the handle reads the file and keeps writers, deletion, and renaming + /// out until it is closed, and opening fails while another handle can write the file. + /// Otherwise the handle reads only attributes and shares everything. + pub(crate) fn open_sealable(path: &Path, deny_writers: bool) -> io::Result { + let mut options = OpenOptions::new(); + if deny_writers { + options.read(true).share_mode(FILE_SHARE_READ); + } else { + options + .access_mode(FILE_READ_ATTRIBUTES) + .share_mode(FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE); + } + options.open(path) + } + + /// Returns the seal of an open regular file. + pub(crate) fn file_seal(file: &File) -> io::Result { + let handle = file.as_raw_handle() as HANDLE; + let standard: FILE_STANDARD_INFO = information(handle, FileStandardInfo)?; + if standard.Directory || standard.DeletePending { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "the path does not name a regular file", + )); + } + let basic: FILE_BASIC_INFO = information(handle, FileBasicInfo)?; + let identity: FILE_ID_INFO = information(handle, FileIdInfo)?; + Ok(FileSeal { + scheme: SCHEME.to_owned(), + volume: identity.VolumeSerialNumber, + file: identity + .FileId + .Identifier + .iter() + .rev() + .map(|byte| format!("{byte:02x}")) + .collect(), + length: u64::try_from(standard.EndOfFile) + .map_err(|_| io::Error::other("the file reports a negative length"))?, + modified: basic.LastWriteTime, + changed: None, + }) + } + + fn information(handle: HANDLE, class: FILE_INFO_BY_HANDLE_CLASS) -> io::Result { + let mut value = T::default(); + // SAFETY: the handle is open, and the buffer is a writable `T` of the size passed, which + // each caller pairs with the structure that `class` returns. + let succeeded = unsafe { + GetFileInformationByHandleEx( + handle, + class, + (&raw mut value).cast(), + mem::size_of::() as u32, + ) + } != 0; + if succeeded { + Ok(value) + } else { + Err(io::Error::last_os_error()) + } + } +} + /// Returns a fresh named-pipe endpoint for OpenVMM to listen on. pub(crate) fn control_endpoint(_socket_path: &Path) -> io::Result { let mut bytes = [0u8; 16]; @@ -550,6 +641,31 @@ pub(crate) fn endpoint_server_pid(endpoint: &str) -> io::Result> { } } +/// Claims a fresh OpenVMM log for one launch. +/// +/// The log's writable handles are the claim: OpenVMM inherits them as its standard output and +/// error, so the claim lasts until both the launching process and OpenVMM have exited. +#[cfg(feature = "nvxhost")] +pub(crate) fn claim_launch_log(_log: &fs::File) -> io::Result<()> { + Ok(()) +} + +/// Returns whether no process holds the claim on an OpenVMM log, which proves that the launch +/// that claimed it has no running process. A missing log proves nothing. +pub(crate) fn launch_log_released(path: &Path) -> io::Result { + // Opening without write sharing fails while any handle can write the log. + match OpenOptions::new() + .read(true) + .share_mode(FILE_SHARE_READ) + .open(path) + { + Ok(_) => Ok(true), + Err(error) if error.raw_os_error() == Some(ERROR_SHARING_VIOLATION as i32) => Ok(false), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(false), + Err(error) => Err(error), + } +} + struct PipeTransport { pipe: OwnedHandle, event: OwnedHandle, diff --git a/aci_edge_sandboxes/src/openvmm/state.rs b/aci_edge_sandboxes/src/openvmm/state.rs index 7138a4533..8a60543ed 100644 --- a/aci_edge_sandboxes/src/openvmm/state.rs +++ b/aci_edge_sandboxes/src/openvmm/state.rs @@ -3,6 +3,7 @@ //! ```text //! / //! .locks/.lock serializes lifecycle transitions of one sandbox +//! images/ registered guest images (nvxhost backend only) //! / //! sandbox.json provisioned configuration //! launch.json endpoint and available process identity of a start in progress @@ -41,6 +42,9 @@ pub(crate) const NATIVE_BACKEND_KEY: &str = "nvxhost"; pub(crate) const SOCKET_NAME: &str = "control.sock"; /// Linux boot-console socket name. pub(crate) const BOOT_SOCKET_NAME: &str = "boot.sock"; +/// Directory of the nvxhost backend's image registry. +#[cfg(feature = "nvxhost")] +pub(crate) const IMAGES_NAME: &str = "images"; const RECORD_NAME: &str = "sandbox.json"; const LAUNCH_NAME: &str = "launch.json"; @@ -73,11 +77,15 @@ pub(crate) struct SandboxRecord { pub(crate) hostname: String, } +/// Artifacts that a native sandbox was provisioned with. +/// +/// The image is referenced by its registered content ID (`sha256:`), whose registration +/// supplies the path; the digests of the runtime files and host library are lowercase hexadecimal. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "camelCase", deny_unknown_fields)] pub(crate) struct NativeArtifactRecord { - pub(crate) image: PathBuf, - pub(crate) image_sha256: String, + pub(crate) image: String, + pub(crate) openvmm_sha256: String, pub(crate) kernel_sha256: String, pub(crate) initrd_sha256: String, pub(crate) library_sha256: String, @@ -96,7 +104,9 @@ pub(crate) struct RuntimeRecord { /// Marker written before OpenVMM is launched and replaced by the runtime record afterwards. /// /// Process identity is added as soon as the child is identified, before writing runtime state. -/// A marker without identity is never expired: its child may be alive without an endpoint. +/// A marker without identity is never expired: its child may be alive without an endpoint. A +/// marker that records a claim on the OpenVMM log is the exception, because OpenVMM inherits the +/// claim, so a released claim proves that nothing of the launch still runs. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "camelCase", deny_unknown_fields)] pub(crate) struct LaunchRecord { @@ -104,6 +114,9 @@ pub(crate) struct LaunchRecord { pub(crate) endpoint: String, #[serde(default, skip_serializing_if = "Option::is_none")] pub(crate) process: Option, + /// Whether the launching process claimed the OpenVMM log before writing this marker. + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + pub(crate) log_claimed: bool, } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] @@ -218,6 +231,23 @@ impl StateStore { let _ = fs::remove_file(self.lock_path(sandbox_id)); } + /// Lists the IDs of the sandboxes that have state directories. + #[cfg(feature = "nvxhost")] + pub(crate) fn sandbox_ids(&self) -> Result> { + let list_error = || format!("cannot list sandbox state {}", self.root.display()); + let mut ids = Vec::new(); + for entry in fs::read_dir(&self.root).map_err(io_error(list_error()))? { + let name = entry.map_err(io_error(list_error()))?.file_name(); + if let Some(id) = name + .to_str() + .and_then(|token| SandboxId::parse(&format!("{}:{token}", SandboxId::PREFIX)).ok()) + { + ids.push(id); + } + } + Ok(ids) + } + /// Creates the state directory of a new sandbox. pub(crate) fn create(&self, sandbox_id: &SandboxId, record: &SandboxRecord) -> Result<()> { if record.backend != self.backend { @@ -418,7 +448,8 @@ pub(crate) fn remove_if_present(path: &Path) -> Result<()> { } } -fn lock(path: &Path) -> Result { +/// Takes an exclusive advisory lock on `path`, creating the lock file if needed. +pub(crate) fn lock(path: &Path) -> Result { let file = OpenOptions::new() .read(true) .write(true) @@ -462,21 +493,24 @@ fn write_private(path: &Path, contents: &[u8]) -> Result<()> { written.map_err(io_error(format!("cannot write {}", path.display()))) } -fn write_json(path: &Path, value: &T) -> Result<()> { - let mut text = serde_json::to_vec_pretty(value) - .map_err(|error| Error::backend_error("cannot encode sandbox state").with_source(error))?; +/// Atomically replaces `path` with `value` as JSON, readable only by the current user. +pub(crate) fn write_json(path: &Path, value: &T) -> Result<()> { + let mut text = serde_json::to_vec_pretty(value).map_err(|error| { + Error::backend_error(format!("cannot encode {}", path.display())).with_source(error) + })?; text.push(b'\n'); write_private(path, &text) } -fn read_json(path: &Path) -> Result> { +/// Reads a JSON state file, returning `None` if it does not exist. +pub(crate) fn read_json(path: &Path) -> Result> { let text = match fs::read(path) { Ok(text) => text, Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None), Err(error) => return Err(io_error(format!("cannot read {}", path.display()))(error)), }; serde_json::from_slice(&text).map(Some).map_err(|error| { - Error::backend_error(format!("sandbox state {} is malformed", path.display())) + Error::backend_error(format!("state file {} is malformed", path.display())) .with_source(error) }) } @@ -532,8 +566,8 @@ mod tests { let mut native = record(); native.backend = NATIVE_BACKEND_KEY.to_owned(); native.native = Some(NativeArtifactRecord { - image: PathBuf::from("image.vhd"), - image_sha256: "1".repeat(64), + image: format!("sha256:{}", "1".repeat(64)), + openvmm_sha256: "5".repeat(64), kernel_sha256: "2".repeat(64), initrd_sha256: "3".repeat(64), library_sha256: "4".repeat(64), @@ -585,9 +619,19 @@ mod tests { format: STATE_FORMAT, endpoint: "endpoint".to_owned(), process: None, + log_claimed: false, }; store.write_launch(&id, &launch).unwrap(); assert_eq!(store.launch(&id).unwrap(), Some(launch.clone())); + // Markers without a claim keep their earlier encoding. + let text = fs::read_to_string(store.dir(&id).join(LAUNCH_NAME)).unwrap(); + assert!(!text.contains("logClaimed"), "{text}"); + let claimed = LaunchRecord { + log_claimed: true, + ..launch.clone() + }; + store.write_launch(&id, &claimed).unwrap(); + assert_eq!(store.launch(&id).unwrap(), Some(claimed)); let identified = LaunchRecord { process: Some(ProcessIdentity { pid: 1, diff --git a/aci_edge_sandboxes/tests/nvxhost_guest.rs b/aci_edge_sandboxes/tests/nvxhost_guest.rs index f759fa77b..2f99801d0 100644 --- a/aci_edge_sandboxes/tests/nvxhost_guest.rs +++ b/aci_edge_sandboxes/tests/nvxhost_guest.rs @@ -3,7 +3,7 @@ //! The tests need `NVXHOST_TEST_OPENVMM`, `NVXHOST_TEST_KERNEL`, `NVXHOST_TEST_INITRD`, //! `NVXHOST_TEST_IMAGE`, `NVXHOST_TEST_LIBRARY`, and the approved `NVXHOST_TEST_SHA256`. Run them //! one at a time so that the check for leftover OpenVMM processes is exact, and optimized, since -//! every provision and start hashes the image: +//! creating a backend hashes the runtime files and registering an image hashes the image: //! `cargo test --release --features nvxhost --test nvxhost_guest -- --ignored --test-threads=1`. #![cfg(feature = "nvxhost")] @@ -14,7 +14,9 @@ use std::sync::Arc; use std::thread; use std::time::{Duration, Instant}; -use aci_edge_sandboxes::openvmm::{Hypervisor, NvxHostBackend, NvxHostConfig, OpenVmmConfig}; +use aci_edge_sandboxes::openvmm::{ + Hypervisor, ImageDigest, ImageId, NvxHostBackend, NvxHostConfig, OpenVmmConfig, +}; use aci_edge_sandboxes::{ AciEdgeSandbox, Error, ErrorCode, ExecOutcome, ExecOutput, ExecRequest, ProvisionRequest, Result, SandboxId, StdinMode, StopResult, @@ -22,6 +24,8 @@ use aci_edge_sandboxes::{ const HELPER_STATE: &str = "NVXHOST_TEST_HELPER_STATE"; const HELPER_SANDBOX: &str = "NVXHOST_TEST_HELPER_SANDBOX"; +/// Hashing the test image took most of a second of every start before images were registered. +const MAX_START_OVERHEAD: Duration = Duration::from_millis(500); fn required(name: &str) -> PathBuf { PathBuf::from(std::env::var_os(name).unwrap_or_else(|| panic!("{name} must be set"))) @@ -38,37 +42,52 @@ fn approved_digest() -> [u8; 32] { digest } +/// Returns the test configuration for `image`, with sandbox state under `state`. +fn native_config(state: &Path, image: &Path) -> NvxHostConfig { + NvxHostConfig::new( + OpenVmmConfig::new( + required("NVXHOST_TEST_OPENVMM"), + required("NVXHOST_TEST_KERNEL"), + required("NVXHOST_TEST_INITRD"), + Hypervisor::Whp, + state, + ), + image, + required("NVXHOST_TEST_LIBRARY"), + approved_digest(), + ) +} + fn backend_with( state: &Path, guest_debug: bool, adjust: impl FnOnce(&mut OpenVmmConfig), ) -> Arc { - let mut config = OpenVmmConfig::new( - required("NVXHOST_TEST_OPENVMM"), - required("NVXHOST_TEST_KERNEL"), - required("NVXHOST_TEST_INITRD"), - Hypervisor::Whp, - state, - ); - adjust(&mut config); - Arc::new( - NvxHostBackend::new( - NvxHostConfig::new( - config, - required("NVXHOST_TEST_IMAGE"), - required("NVXHOST_TEST_LIBRARY"), - approved_digest(), - ) - .with_guest_debug(guest_debug), - ) - .unwrap(), - ) + let mut config = + native_config(state, &required("NVXHOST_TEST_IMAGE")).with_guest_debug(guest_debug); + adjust(&mut config.openvmm); + Arc::new(NvxHostBackend::new(config).unwrap()) } fn backend(state: &Path) -> Arc { backend_with(state, false, |_| {}) } +/// Copies the test image into `state`, so that a test can change the copy. +fn copied_image(state: &Path) -> PathBuf { + let image = state.join("image.vhd"); + std::fs::copy(required("NVXHOST_TEST_IMAGE"), &image).unwrap(); + image +} + +/// Moves a file's modification time without changing its content. +fn touch(path: &Path) { + let file = std::fs::File::options().write(true).open(path).unwrap(); + let modified = file.metadata().unwrap().modified().unwrap(); + file.set_modified(modified + Duration::from_secs(2)) + .unwrap(); +} + /// Lists the OpenVMM processes running on this host, failing if they cannot be listed. fn openvmm_processes() -> BTreeSet { let executable = required("NVXHOST_TEST_OPENVMM"); @@ -242,6 +261,33 @@ fn failure(result: Result) -> ErrorCode { } } +/// Fails unless `result` is a `BackendUnavailable` error whose message contains `expected`. +fn assert_unavailable(result: Result, expected: &str) { + match result { + Ok(_) => { + panic!("the operation unexpectedly succeeded instead of failing with {expected:?}") + } + Err(error) => { + assert_eq!(error.code(), ErrorCode::BackendUnavailable, "{error}"); + assert!(error.message().contains(expected), "{error}"); + } + } +} + +/// Starts a sandbox and returns the time that the call spent outside the guest boot. +fn start_overhead(client: &AciEdgeSandbox, id: &SandboxId) -> Duration { + let called = Instant::now(); + let started = client.start(id).unwrap(); + let total = called.elapsed(); + let boot = started + .metadata + .as_ref() + .and_then(|metadata| metadata.get("bootMilliseconds")) + .and_then(serde_json::Value::as_u64) + .unwrap(); + total.saturating_sub(Duration::from_millis(boot)) +} + fn start_in_helper(state: &Path, id: &SandboxId) -> Child { Command::new(std::env::current_exe().unwrap()) .args([ @@ -665,6 +711,13 @@ fn whp_parallel_sandboxes_and_repeated_restarts_stay_isolated() { } client.deprovision(&sandbox.id).unwrap(); eprintln!("(start call, boot) milliseconds across restarts: {starts:?}"); + assert!( + starts + .iter() + .all(|&(total, boot)| total.saturating_sub(u128::from(boot)) + < MAX_START_OVERHEAD.as_millis()), + "starts spent more than {MAX_START_OVERHEAD:?} outside the guest boot: {starts:?}" + ); assert_no_new_openvmm(&before); } @@ -709,3 +762,146 @@ fn whp_concurrent_commands_share_one_guest() { client.deprovision(&sandbox.id).unwrap(); assert_no_new_openvmm(&before); } + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_registered_images_start_without_hashing_and_fail_closed_after_a_change() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let image = copied_image(state.path()); + let created = Instant::now(); + let first = NvxHostBackend::new(native_config(state.path(), &image)).unwrap(); + let hashed = created.elapsed(); + let id = first.image_id(); + let created = Instant::now(); + let backend = Arc::new(NvxHostBackend::new(native_config(state.path(), &image)).unwrap()); + let reused = created.elapsed(); + eprintln!("backend creation: {hashed:?} registering the image, {reused:?} reusing it"); + assert_eq!(backend.image_id(), id); + assert_eq!(backend.runtime_digests(), first.runtime_digests()); + let images = backend.images().unwrap(); + assert_eq!(images.len(), 1, "{images:?}"); + assert_eq!( + ( + images[0].id, + &images[0].path, + images[0].intact, + images[0].trusted + ), + (id, &image, true, false) + ); + + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let overhead = start_overhead(&client, &sandbox.id); + eprintln!("the first start spent {overhead:?} outside the guest boot"); + assert_prints(&client, &sandbox.id, "registered"); + assert_graceful(client.stop(&sandbox.id)); + + // A changed seal fails closed, without hashing, until the image is registered again. + touch(&image); + assert_unavailable( + client.start(&sandbox.id), + "changed since it was registered; register it again", + ); + assert_unavailable(client.provision(&ProvisionRequest::new()), "changed since"); + assert!(!backend.images().unwrap()[0].intact); + assert_eq!( + failure(backend.unregister_image(&id)), + ErrorCode::PolicyValidation + ); + let registered = backend + .register_image(&image, ImageDigest::Compute) + .unwrap(); + assert_eq!((registered.id, registered.intact), (id, true)); + client.start(&sandbox.id).unwrap(); + assert_prints(&client, &sandbox.id, "reregistered"); + assert_graceful(client.stop(&sandbox.id)); + backend.verify_image(&id).unwrap(); + + client.deprovision(&sandbox.release()).unwrap(); + assert!(backend.unregister_image(&id).unwrap()); + assert!(!backend.unregister_image(&id).unwrap()); + assert!(backend.images().unwrap().is_empty()); + assert_unavailable( + client.provision(&ProvisionRequest::new()), + "no longer registered", + ); + assert!(image.is_file()); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_trusted_digests_and_content_verification_follow_their_contracts() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let image = copied_image(state.path()); + + // A trusted digest is recorded without hashing, even one that the content does not have. + let claimed = ImageId::from_sha256([7; 32]); + let trusting = NvxHostBackend::new( + native_config(state.path(), &image) + .with_image_digest(ImageDigest::Trusted(*claimed.sha256())), + ) + .unwrap(); + assert_eq!(trusting.image_id(), claimed); + assert!(trusting.images().unwrap()[0].trusted); + assert_unavailable( + trusting.verify_image(&claimed), + "no longer matches its SHA-256", + ); + + // Content verification hashes before every start, so it refuses the misattributed image. + let verifying = Arc::new( + NvxHostBackend::new(native_config(state.path(), &image).with_content_verification(true)) + .unwrap(), + ); + assert_eq!(verifying.image_id(), claimed); + let client = AciEdgeSandbox::from_shared(verifying.clone()); + let sandbox = provision(&client, &verifying); + assert_unavailable(client.start(&sandbox.id), "no longer matches its SHA-256"); + + // Runtime files are checked against approved digests and pinned per sandbox. + let digests = verifying.runtime_digests(); + NvxHostBackend::new(native_config(state.path(), &image).with_runtime_digests(digests)).unwrap(); + let mut unapproved = digests; + unapproved.initrd = [0; 32]; + assert_unavailable( + NvxHostBackend::new(native_config(state.path(), &image).with_runtime_digests(unapproved)), + "does not match the approved SHA-256", + ); + let initrd = state.path().join("initramfs.cpio.gz"); + let mut changed = std::fs::read(required("NVXHOST_TEST_INITRD")).unwrap(); + changed.push(0); + std::fs::write(&initrd, changed).unwrap(); + let mut config = native_config(state.path(), &image); + config.openvmm.initrd = initrd; + let other = AciEdgeSandbox::from_shared(Arc::new(NvxHostBackend::new(config).unwrap())); + assert_unavailable( + other.start(&sandbox.id), + "provisioned with different OpenVMM runtime files", + ); + client.deprovision(&sandbox.release()).unwrap(); + + // Once the image is registered by its actual digest, a verified start succeeds. + assert!(verifying.unregister_image(&claimed).unwrap()); + let actual = verifying + .register_image(&image, ImageDigest::Compute) + .unwrap() + .id; + assert_ne!(actual, claimed); + let verifying = Arc::new( + NvxHostBackend::new(native_config(state.path(), &image).with_content_verification(true)) + .unwrap(), + ); + assert_eq!(verifying.image_id(), actual); + let client = AciEdgeSandbox::from_shared(verifying.clone()); + let sandbox = provision(&client, &verifying); + let overhead = start_overhead(&client, &sandbox.id); + eprintln!("a start with content verification spent {overhead:?} outside the guest boot"); + assert_prints(&client, &sandbox.id, "verified"); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.release()).unwrap(); + assert_no_new_openvmm(&before); +}