diff --git a/aci_edge_sandboxes/Cargo.lock b/aci_edge_sandboxes/Cargo.lock index c8d486c48..486ede1b1 100644 --- a/aci_edge_sandboxes/Cargo.lock +++ b/aci_edge_sandboxes/Cargo.lock @@ -8,6 +8,8 @@ version = "0.1.0" dependencies = [ "getrandom", "libc", + "libloading", + "prost", "serde", "serde_json", "sha2", @@ -17,6 +19,12 @@ dependencies = [ "windows-sys", ] +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + [[package]] name = "bitflags" version = "2.13.2" @@ -73,6 +81,12 @@ dependencies = [ "crypto-common", ] +[[package]] +name = "either" +version = "1.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" + [[package]] name = "errno" version = "0.3.14" @@ -110,6 +124,15 @@ dependencies = [ "r-efi", ] +[[package]] +name = "itertools" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" +dependencies = [ + "either", +] + [[package]] name = "itoa" version = "1.0.18" @@ -122,6 +145,16 @@ version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" +[[package]] +name = "libloading" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d7c4b02199fee7c5d21a5ae7d8cfa79a6ef5bb2fc834d6e9058e89c825efdc55" +dependencies = [ + "cfg-if", + "windows-link", +] + [[package]] name = "linux-raw-sys" version = "0.12.1" @@ -155,6 +188,29 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "prost" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5" +dependencies = [ + "bytes", + "prost-derive", +] + +[[package]] +name = "prost-derive" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d" +dependencies = [ + "anyhow", + "itertools", + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "quote" version = "1.0.47" @@ -210,7 +266,7 @@ checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.6", ] [[package]] @@ -237,6 +293,17 @@ dependencies = [ "digest", ] +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "syn" version = "3.0.6" @@ -278,7 +345,7 @@ checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.6", ] [[package]] diff --git a/aci_edge_sandboxes/Cargo.toml b/aci_edge_sandboxes/Cargo.toml index 6cc84402a..14e661ae3 100644 --- a/aci_edge_sandboxes/Cargo.toml +++ b/aci_edge_sandboxes/Cargo.toml @@ -23,11 +23,16 @@ openvmm = [] bundled = ["openvmm", "dep:sha2"] # Tokio-based asynchronous wrappers around the synchronous core. async = ["dep:tokio"] +# Loads an explicitly supplied native host library and boots an image-backed ttrpc guest. +nvxhost = ["openvmm", "dep:libloading", "dep:prost", "dep:sha2-runtime"] # Test doubles: an in-memory `MockBackend` and the `aci-edge-sandboxes-fake-openvmm` binary. testing = [] [dependencies] getrandom = "0.4" +libloading = { version = "0.8", optional = true } +prost = { version = "0.13", optional = true } +sha2-runtime = { package = "sha2", version = "0.10", optional = true } serde = { version = "1.0.200", features = ["derive"] } serde_json = "1.0.120" thiserror = "2" @@ -65,6 +70,10 @@ doc = false name = "lifecycle" required-features = ["openvmm"] +[[example]] +name = "nvxhost_lifecycle" +required-features = ["nvxhost"] + [lints.rust] missing_docs = "warn" unsafe_op_in_unsafe_fn = "deny" diff --git a/aci_edge_sandboxes/README.md b/aci_edge_sandboxes/README.md index c2f69fc57..846e0e430 100644 --- a/aci_edge_sandboxes/README.md +++ b/aci_edge_sandboxes/README.md @@ -3,12 +3,15 @@ `aci_edge_sandboxes` is the Rust interface to the ACI Edge Sandboxes lifecycle. It exposes provision, start, exec, stop, and deprovision through a pluggable backend. The default backend drives the `openvmm` binary directly; no daemon or -Python tooling is involved at runtime. +Python tooling is involved at runtime. The optional `nvxhost` backend uses a +separately supplied native host library and a different, image-backed guest. -Each sandbox is a microVM that runs the NVX guest's Alpine Linux userland -directly from its initramfs. There are no image layers, scratch disks, or -container namespaces; workloads run as a non-root user in the guest itself, and -guest state lives in memory until the sandbox stops. +By default, each sandbox is a microVM that runs the NVX guest's Alpine Linux +userland directly from its initramfs. That backend has no image layers, scratch +disks, or container namespaces; workloads run as a non-root user in the guest +itself, and guest state lives in memory until the sandbox stops. The opt-in +native backend instead runs commands against a caller-supplied, read-only GPT +image with a RAM overlay. The Cargo package, library, and directory are named `aci_edge_sandboxes`. @@ -95,6 +98,8 @@ envelope (`version`, `phase`, and `containment`) belongs to the caller. - Backends in this crate: - `openvmm::OpenVmmBackend` (feature `openvmm`, default) drives the `openvmm` executable. + - `openvmm::NvxHostBackend` (feature `nvxhost`) drives OpenVMM using a + separately supplied native host library and image-backed guest. - `testing::MockBackend` (feature `testing`) implements the state machine in memory for consumers' unit tests. - `AsyncAciEdgeSandbox` (feature `async`) wraps `AciEdgeSandbox` for Tokio. Lifecycle calls run on @@ -104,6 +109,107 @@ To add a backend, implement `Backend`. Declare only the capabilities it can enforce, and report state-machine violations with the codes listed in the [error mapping](#error-mapping). +## Image-backed native host backend (opt-in) + +Enable `nvxhost` to use `AciEdgeSandbox::nvxhost(NvxHostConfig)`. This leaves the default +direct-OpenVMM backend and its Alpine guest unchanged. The caller supplies OpenVMM, a +compatible kernel and static edge-agent initramfs, a prepared GPT image, the absolute path +to `nvxhost.dll` (`libnvxhost.so` on Linux), and the independently approved SHA-256 of that +library. The Rust crate does **not** build, download, or publish the private library. +`NvxHostBackend::new` verifies the file digest and ABI version and requires the additive +`nvx_build_ramfs_launch_arguments` and `nvx_session_connect_verified` exports. A missing, +wrong-version, or wrong-digest library fails rather than falling back. + +The image is attached read-only in OpenVMM's distro block slot. The edge guest validates +its GPT and p2+ ext4 layers and uses a RAM-backed tmpfs upper/work for its overlay; it +does not interpret p1 OCI configuration or require a scratch disk. OpenVMM must support +the explicit `nvx_overlay_upper=ramfs` scratchless topology. On Windows, the backend +assigns `//./pipe/openvmm-microvm-` control and boot endpoints. It retains the +OpenVMM process/state, sends the 32-byte capability through stdin, checks the serving +process on the connected pipe/socket, and owns graceful-or-forced stop and deprovision. +State is kept separately under `/nvxhost/`; a running VM is never silently +converted to the direct backend. `NvxHostBackend::guest_logs` reads a bounded non-follow +guest-log snapshot before stop. + +The backend hashes OpenVMM, the kernel, and the initramfs once, when it is created +(`NvxHostBackend::runtime_digests`); `NvxHostConfig::with_runtime_digests` requires approved +digests. It also registers the configured image under `/nvxhost/images/` by its +content ID, `sha256:` (`ImageId`). Registration hashes the image once; +`ImageDigest::Expect` additionally requires a digest, and `ImageDigest::Trusted` records a digest +that the caller's own policy verified without reading the file. Later backends and processes +reuse a registration that the file still matches without reading it. Provision records the image +ID and the runtime digests. Start compares each file's seal (volume, file ID, length, and +last-write time, plus the change time on Linux) with the one taken when the file was hashed, +instead of hashing again, and fails with `backend_unavailable` if anything changed, so a start +costs about the guest's boot time. A seal detects replacement or modification, not a writer that +deliberately restores timestamps. On Windows, start also keeps writers out of the files until +OpenVMM has opened them. Register a changed image again with `register_image`: unchanged content +keeps its ID, so sandboxes that use it start again. `images` lists the registrations and whether +each file still matches, `verify_image` hashes one again as a diagnostic, and `unregister_image` +refuses while a provisioned sandbox uses the image. `with_content_verification(true)` hashes the +image and runtime files again before every start, also as a diagnostic. + +Start claims the sandbox's OpenVMM log before launching, and OpenVMM inherits the claim. If the +caller dies before it records OpenVMM's identity, the next operation waits up to +`start_timeout` for that OpenVMM to open its endpoint and then terminates it; once no process +holds the log, nothing of the interrupted launch runs, so the sandbox is usable again. + +As with the direct backend, executions on one sandbox share its single control connection: a +concurrent exec waits up to `control_timeout` for the running one to finish. The backend +copies the guest boot console to `NvxHostBackend::console_log_path` from the process that +started or last used the sandbox. OpenVMM serves one console client at a time and holds +guest output while none is connected, so a guest that writes enough console output stalls +until a listener reconnects: keep that process running, or use the sandbox from its +successor, whose listener waits to take the capture over. Start reports `bootMilliseconds` +and `guestBuildId`; stop reports `forced` and, after a failed graceful shutdown, +`gracefulError`. A capture failure never fails stop or deprovision: both report it as +`consoleError`. + +This first backend supports provision/start/exec/stop/deprovision with shell commands or +argv and a fixed caller-provided image. Positive execution timeouts must be whole seconds, +matching the guest RPC's precision; finer-grained timeouts fail validation rather than +silently extending execution. It explicitly rejects host file mappings, network +configuration beyond deny-all, piped stdin, execution cancellation, custom working +directories and environments. Snapshot/restore, image selection, and richer guest +operations are not part of this profile. See +[`examples/nvxhost_lifecycle.rs`](examples/nvxhost_lifecycle.rs) for a run requiring +`--openvmm`, `--kernel`, `--initrd`, `--image`, `--host-library`, `--host-sha256`, +`--state-root`, `--hypervisor`, and a command after `--`; an optional `--image-sha256` +requires the image's digest when it is registered. The ignored +`tests/nvxhost_guest.rs` exercises an actual WHP guest when the corresponding +`NVXHOST_TEST_*` paths and approved DLL digest are set. + +For example, from `aci_edge_sandboxes` on a Windows WHP host, set the following +paths to compatible, separately built artifacts and a caller-prepared GPT disk +with p2+ ext4 layers (not a container image reference). Use the **edge** guest +initramfs, not the default Alpine or standard container guest initramfs: + +```powershell +$openvmm = 'C:\path\to\openvmm.exe' +$kernel = 'C:\path\to\vmlinux' +$initrd = 'C:\path\to\nvx-edge-initramfs.cpio.gz' +$image = 'C:\path\to\layers.gpt' +$hostLibrary = 'C:\path\to\nvxhost.dll' +$approvedHostSha256 = '<64 hexadecimal digits from an independent trust policy>' +$stateRoot = 'C:\nvx-edge-state' + +cargo run --release --locked --features nvxhost --example nvxhost_lifecycle -- ` + --openvmm $openvmm --kernel $kernel --initrd $initrd --image $image ` + --host-library $hostLibrary --host-sha256 $approvedHostSha256 ` + --state-root $stateRoot --hypervisor whp -- 'printf READY' +``` + +Build the native library and the static edge initramfs separately from their +matching private sources; this example neither fetches nor builds them. Use the +pinned OpenVMM, which includes the scratchless RAM-overlay topology, and a kernel +compatible with that OpenVMM and guest revision. On Linux, supply a matching +`libnvxhost.so` and OpenVMM build and select `--hypervisor mshv`; the Linux edge +lifecycle has not yet been verified end to end. + +The caller must independently approve and protect the native asset. Checking a caller-supplied +digest does not make a writable path or a self-declared digest trustworthy; use this profile +only with an immutable, externally authorized library installation. + ## OpenVMM backend ### Configuration diff --git a/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs b/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs new file mode 100644 index 000000000..d24c7bea9 --- /dev/null +++ b/aci_edge_sandboxes/examples/nvxhost_lifecycle.rs @@ -0,0 +1,174 @@ +//! Runs the image-backed guest lifecycle with a separately supplied native host library. + +use std::io::Write; +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::Arc; + +use aci_edge_sandboxes::openvmm::{ + Hypervisor, ImageDigest, NvxHostBackend, NvxHostConfig, OpenVmmConfig, +}; +use aci_edge_sandboxes::{AciEdgeSandbox, ExecOutcome, ExecRequest, ProvisionRequest}; + +fn main() -> ExitCode { + match run() { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("error: {error}"); + ExitCode::FAILURE + } + } +} + +fn run() -> Result<(), String> { + let mut openvmm = None; + let mut kernel = None; + let mut initrd = None; + let mut image = None; + let mut library = None; + let mut digest = None; + let mut image_digest = ImageDigest::Compute; + let mut root = None; + let mut hypervisor = None; + let mut command = None; + let mut args = std::env::args().skip(1); + while let Some(option) = args.next() { + let mut value = || { + args.next() + .ok_or_else(|| format!("{option} requires a value")) + }; + match option.as_str() { + "--openvmm" => openvmm = Some(value()?), + "--kernel" => kernel = Some(value()?), + "--initrd" => initrd = Some(value()?), + "--image" => image = Some(value()?), + "--host-library" => library = Some(value()?), + "--host-sha256" => digest = Some(parse_digest(&option, &value()?)?), + "--image-sha256" => { + image_digest = ImageDigest::Expect(parse_digest(&option, &value()?)?); + } + "--state-root" => root = Some(value()?), + "--hypervisor" => hypervisor = Some(value()?.parse::().map_err(describe)?), + "--" => { + command = Some(args.by_ref().collect::>().join(" ")); + break; + } + other => return Err(format!("unexpected option {other}")), + } + } + let required = + |name: &str, value: Option| value.ok_or_else(|| format!("missing required {name}")); + let openvmm = required("--openvmm", openvmm)?; + let kernel = required("--kernel", kernel)?; + let initrd = required("--initrd", initrd)?; + let image = required("--image", image)?; + let library = required("--host-library", library)?; + let root = required("--state-root", root)?; + let digest = digest.ok_or("missing required --host-sha256")?; + let hypervisor = hypervisor.ok_or("missing required --hypervisor")?; + let command = command + .filter(|command| !command.is_empty()) + .ok_or("pass the guest command after --")?; + + let config = OpenVmmConfig::new(openvmm, kernel, initrd, hypervisor, PathBuf::from(root)); + let backend = Arc::new( + NvxHostBackend::new( + NvxHostConfig::new(config, image, library, digest).with_image_digest(image_digest), + ) + .map_err(describe)?, + ); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let id = client + .provision(&ProvisionRequest::new()) + .map_err(describe)? + .sandbox_id; + let started = client.start(&id); + let succeeded = started.is_ok(); + let executed = started.and_then(|_| { + let output = client + .exec(&id, &ExecRequest::command_line(command))? + .wait_with_output()?; + let logs = backend.guest_logs(&id)?; + if logs.is_empty() || !logs.windows(8).any(|window| window == b"execute:") { + return Err(aci_edge_sandboxes::Error::new( + aci_edge_sandboxes::ErrorCode::BackendError, + "the guest did not return its execution log", + )); + } + Ok(output) + }); + let stopped = succeeded.then(|| client.stop(&id)); + let deprovisioned = if stopped.as_ref().is_none_or(Result::is_ok) { + Some(client.deprovision(&id)) + } else { + None + }; + let mut failures = Vec::new(); + if let Err(error) = &executed { + failures.push(format!("guest lifecycle: {error}")); + } + if let Some(Err(error)) = &stopped { + failures.push(format!("stop: {error}")); + } + if let Some(Err(error)) = &deprovisioned { + failures.push(format!("deprovision: {error}")); + } + if !failures.is_empty() { + return Err(failures.join("; ")); + } + let stopped = stopped + .expect("started guests always have a stop result") + .map_err(describe)?; + if stopped + .metadata + .as_ref() + .and_then(|metadata| metadata.get("forced")) + .and_then(serde_json::Value::as_bool) + != Some(false) + { + return Err("the guest did not shut down gracefully".to_owned()); + } + let output = executed.map_err(describe)?; + std::io::stdout() + .write_all(&output.stdout) + .map_err(|error| error.to_string())?; + std::io::stderr() + .write_all(&output.stderr) + .map_err(|error| error.to_string())?; + if output.outcome != ExecOutcome::Exited(0) { + return Err(format!("guest command {}", output.outcome)); + } + Ok(()) +} + +fn parse_digest(option: &str, hex: &str) -> Result<[u8; 32], String> { + if hex.len() != 64 || !hex.is_ascii() { + return Err(format!( + "{option} must contain exactly 64 hexadecimal digits" + )); + } + let mut digest = [0u8; 32]; + for (index, byte) in digest.iter_mut().enumerate() { + *byte = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16) + .map_err(|_| format!("{option} must be hexadecimal"))?; + } + Ok(digest) +} + +fn describe(error: aci_edge_sandboxes::Error) -> String { + error.to_string() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn digest_requires_exact_hex_bytes() { + let parse = |hex: &str| parse_digest("--image-sha256", hex); + assert_eq!(parse(&"ab".repeat(32)).unwrap(), [0xab; 32]); + assert!(parse("ab").unwrap_err().starts_with("--image-sha256 ")); + assert!(parse(&"é".repeat(32)).is_err()); + assert!(parse(&"gg".repeat(32)).is_err()); + } +} diff --git a/aci_edge_sandboxes/src/client.rs b/aci_edge_sandboxes/src/client.rs index 8a7c31d47..0d631ee7b 100644 --- a/aci_edge_sandboxes/src/client.rs +++ b/aci_edge_sandboxes/src/client.rs @@ -43,6 +43,12 @@ impl AciEdgeSandbox { Ok(Self::new(crate::openvmm::OpenVmmBackend::new(config)?)) } + /// Creates an image-backed sandbox through an explicitly supplied nvxhost library. + #[cfg(feature = "nvxhost")] + pub fn nvxhost(config: crate::openvmm::NvxHostConfig) -> Result { + Ok(Self::new(crate::openvmm::NvxHostBackend::new(config)?)) + } + /// Returns the backend. pub fn backend(&self) -> &dyn Backend { self.backend.as_ref() diff --git a/aci_edge_sandboxes/src/lib.rs b/aci_edge_sandboxes/src/lib.rs index 1a2dc6954..c423d6196 100644 --- a/aci_edge_sandboxes/src/lib.rs +++ b/aci_edge_sandboxes/src/lib.rs @@ -58,6 +58,8 @@ mod exec; mod id; mod input; mod model; +#[cfg(feature = "nvxhost")] +mod nvxhost; mod stream; mod validate; diff --git a/aci_edge_sandboxes/src/nvxhost.rs b/aci_edge_sandboxes/src/nvxhost.rs new file mode 100644 index 000000000..e3c0593de --- /dev/null +++ b/aci_edge_sandboxes/src/nvxhost.rs @@ -0,0 +1,909 @@ +//! Checked dynamic binding for the privately supplied nvxhost C ABI. + +#[cfg(not(target_pointer_width = "64"))] +compile_error!("the nvxhost ABI is supported only on 64-bit hosts"); + +use std::ffi::{OsString, c_void}; +use std::fs; +use std::mem::size_of; +use std::path::Path; +use std::ptr; +use std::slice; +use std::sync::Arc; +use std::thread; +use std::time::{Duration, Instant}; + +use libloading::Library; +use prost::Message; +use sha2_runtime::{Digest, Sha256}; + +use crate::error::{Error, Result}; + +const ABI_VERSION: u32 = 1; +const STATUS_OK: i32 = 0; +const STATUS_TIMEOUT: i32 = 2; +const RESULT_OK: i32 = 0; +const RESULT_ERROR: i32 = 1; +const COMPLETION_CONNECT: u32 = 1; +const COMPLETION_SEND: u32 = 2; +const COMPLETION_RECV: u32 = 3; +const COMPLETION_DISPOSE: u32 = 6; +const ERROR_CONNECT: u32 = 12; +const CALL_UNARY: u32 = 0; +const CALL_SERVER_STREAM: u32 = 1; +const FRAME_RESPONSE: u8 = 2; +const FRAME_DATA: u8 = 3; +const FLAG_REMOTE_CLOSED: u8 = 1; +const FLAG_NO_DATA: u8 = 4; +const LAUNCH_GUEST_DEBUG: u32 = 1; +const LAUNCH_HAS_MEMORY: u32 = 4; + +#[repr(C)] +#[derive(Clone, Copy, Default)] +struct NvxStr { + data: *const u8, + length: usize, + present: u32, + reserved: u32, +} + +impl NvxStr { + fn present(text: &str) -> Self { + Self { + data: text.as_ptr(), + length: text.len(), + present: 1, + reserved: 0, + } + } +} + +#[repr(C)] +#[derive(Default)] +struct NvxLaunchRequest { + struct_size: u32, + flags: u32, + cpus: i32, + memory_mb: i32, + kernel_path: NvxStr, + initrd_path: NvxStr, + control_socket_path: NvxStr, + boot_console_socket_path: NvxStr, + hypervisor: NvxStr, + vhd_path: NvxStr, + scratch_vhd_path: NvxStr, + isolation_profile: NvxStr, + startup_mode: NvxStr, +} + +#[repr(C)] +struct NvxCompletion { + struct_size: u32, + kind: u32, + token: u64, + result: i32, + frame_type: u8, + frame_flags: u8, + reserved: u16, + handle: u64, + value: u64, + data: *const u8, + data_length: usize, + error: *mut c_void, +} + +#[repr(C)] +#[derive(Default)] +struct NvxErrorEntry { + struct_size: u32, + kind: u32, + reason: u32, + has_os_error: u32, + os_error: i32, + reserved: u32, + identity: u64, + external_token: u64, + message: *const u8, + message_length: usize, + param: *const u8, + param_length: usize, + data: *const u8, + data_length: usize, +} + +const _: () = { + assert!(size_of::() == 24); + assert!(size_of::() == 232); + assert!(size_of::() == 64); + assert!(size_of::() == 88); + assert!(std::mem::offset_of!(NvxCompletion, error) == 56); + assert!(std::mem::offset_of!(NvxLaunchRequest, hypervisor) == 112); +}; + +type GetVersion = unsafe extern "C" fn() -> u32; +type RuntimeCreate = unsafe extern "C" fn(u32, *mut u64, *mut *mut c_void) -> i32; +type RuntimeNext = unsafe extern "C" fn(u64, u32, *mut *mut NvxCompletion) -> i32; +type CompletionRelease = unsafe extern "C" fn(*mut NvxCompletion); +type OperationCancel = unsafe extern "C" fn(u64, u64) -> i32; +type RuntimeFree = unsafe extern "C" fn(u64) -> i32; +type ErrorCount = unsafe extern "C" fn(*const c_void) -> u32; +type ErrorGet = unsafe extern "C" fn(*const c_void, u32, *mut NvxErrorEntry) -> i32; +type ErrorFree = unsafe extern "C" fn(*mut c_void); +type SessionConnectVerified = + unsafe extern "C" fn(u64, *const u8, usize, *const u8, usize, u32, u64) -> i32; +type SessionDispose = unsafe extern "C" fn(u64, u64) -> i32; +type SessionRelease = unsafe extern "C" fn(u64) -> i32; +type CallOpen = + unsafe extern "C" fn(u64, *const u8, usize, u32, *mut u64, *mut u32, *mut *mut c_void) -> i32; +type CallSendRequest = unsafe extern "C" fn(u64, *const u8, usize, u8, i64, u32, u64) -> i32; +type CallRecv = unsafe extern "C" fn(u64, u64) -> i32; +type CallTryComplete = unsafe extern "C" fn(u64) -> i32; +type CallRelease = unsafe extern "C" fn(u64) -> i32; +type BuildRamfsArguments = unsafe extern "C" fn( + *const NvxLaunchRequest, + *mut *mut u8, + *mut usize, + *mut *mut c_void, +) -> i32; +type BufferFree = unsafe extern "C" fn(*mut u8, usize); + +struct Exports { + runtime_create: RuntimeCreate, + runtime_next: RuntimeNext, + completion_release: CompletionRelease, + operation_cancel: OperationCancel, + runtime_free: RuntimeFree, + error_count: ErrorCount, + error_get: ErrorGet, + error_free: ErrorFree, + session_connect_verified: SessionConnectVerified, + session_dispose: SessionDispose, + session_release: SessionRelease, + call_open: CallOpen, + call_send_request: CallSendRequest, + call_recv: CallRecv, + call_try_complete: CallTryComplete, + call_release: CallRelease, + build_ramfs_arguments: BuildRamfsArguments, + buffer_free: BufferFree, +} + +pub(crate) struct HostLibrary { + _library: Library, + exports: Exports, +} + +impl std::fmt::Debug for HostLibrary { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter + .debug_struct("HostLibrary") + .finish_non_exhaustive() + } +} + +fn resolve(library: &Library, name: &[u8]) -> Result { + // Function pointers remain valid because HostLibrary retains the loaded library. + unsafe { library.get::(name) } + .map(|symbol| *symbol) + .map_err(|error| { + Error::backend_unavailable(format!( + "nvxhost is missing required export {}", + String::from_utf8_lossy(name).trim_end_matches('\0') + )) + .with_source(error) + }) +} + +impl HostLibrary { + pub(crate) fn load(path: &Path, expected_sha256: &[u8; 32]) -> Result> { + if !path.is_absolute() { + return Err(Error::backend_unavailable( + "the nvxhost library path must be absolute", + )); + } + let path = fs::canonicalize(path).map_err(|error| { + Error::backend_unavailable(format!("cannot locate nvxhost at {}", path.display())) + .with_source(error) + })?; + let bytes = fs::read(&path).map_err(|error| { + Error::backend_unavailable(format!("cannot verify nvxhost at {}", path.display())) + .with_source(error) + })?; + let actual: [u8; 32] = Sha256::digest(&bytes).into(); + if &actual != expected_sha256 { + return Err(Error::backend_unavailable(format!( + "nvxhost at {} does not match the approved SHA-256", + path.display() + ))); + } + let library = unsafe { Library::new(&path) }.map_err(|error| { + Error::backend_unavailable(format!("cannot load nvxhost from {}", path.display())) + .with_source(error) + })?; + let version: GetVersion = resolve(&library, b"nvx_get_api_version\0")?; + let actual_version = unsafe { version() }; + if actual_version != ABI_VERSION { + return Err(Error::backend_unavailable(format!( + "nvxhost ABI is {actual_version}, expected {ABI_VERSION}" + ))); + } + let exports = Exports { + runtime_create: resolve(&library, b"nvx_runtime_create\0")?, + runtime_next: resolve(&library, b"nvx_runtime_next_completion\0")?, + completion_release: resolve(&library, b"nvx_completion_release\0")?, + operation_cancel: resolve(&library, b"nvx_operation_cancel\0")?, + runtime_free: resolve(&library, b"nvx_runtime_free\0")?, + error_count: resolve(&library, b"nvx_error_count\0")?, + error_get: resolve(&library, b"nvx_error_get\0")?, + error_free: resolve(&library, b"nvx_error_free\0")?, + session_connect_verified: resolve(&library, b"nvx_session_connect_verified\0")?, + session_dispose: resolve(&library, b"nvx_session_dispose\0")?, + session_release: resolve(&library, b"nvx_session_release\0")?, + call_open: resolve(&library, b"nvx_call_open\0")?, + call_send_request: resolve(&library, b"nvx_call_send_request\0")?, + call_recv: resolve(&library, b"nvx_call_recv\0")?, + call_try_complete: resolve(&library, b"nvx_call_try_complete\0")?, + call_release: resolve(&library, b"nvx_call_release\0")?, + build_ramfs_arguments: resolve(&library, b"nvx_build_ramfs_launch_arguments\0")?, + buffer_free: resolve(&library, b"nvx_buffer_free\0")?, + }; + Ok(Arc::new(Self { + _library: library, + exports, + })) + } + + pub(crate) fn launch_arguments(&self, inputs: &LaunchInputs<'_>) -> Result> { + fn path_text(path: &Path) -> Result<&str> { + path.to_str().ok_or_else(|| { + Error::backend_unavailable(format!( + "the nvxhost ABI requires a UTF-8 artifact path: {}", + path.display() + )) + }) + } + let kernel = path_text(inputs.kernel)?; + let initrd = path_text(inputs.initrd)?; + let image = path_text(inputs.image)?; + let memory_mb = i32::try_from(inputs.memory_mb) + .map_err(|_| Error::policy_validation("guest memory exceeds the nvxhost ABI range"))?; + let request = NvxLaunchRequest { + struct_size: size_of::() as u32, + flags: LAUNCH_HAS_MEMORY + | if inputs.guest_debug { + LAUNCH_GUEST_DEBUG + } else { + 0 + }, + memory_mb, + kernel_path: NvxStr::present(kernel), + initrd_path: NvxStr::present(initrd), + control_socket_path: NvxStr::present(inputs.control), + boot_console_socket_path: NvxStr::present(inputs.boot), + hypervisor: NvxStr::present(inputs.hypervisor), + vhd_path: NvxStr::present(image), + ..NvxLaunchRequest::default() + }; + let mut buffer = ptr::null_mut(); + let mut length = 0; + let mut error = ptr::null_mut(); + let status = unsafe { + (self.exports.build_ramfs_arguments)(&request, &mut buffer, &mut length, &mut error) + }; + self.check_status("building OpenVMM arguments", status, error)?; + if buffer.is_null() || length == 0 { + return Err(Error::backend_error( + "nvxhost returned no OpenVMM launch arguments", + )); + } + let encoded = unsafe { slice::from_raw_parts(buffer, length).to_vec() }; + unsafe { (self.exports.buffer_free)(buffer, length) }; + if encoded.last() != Some(&0) { + return Err(Error::backend_error( + "nvxhost returned an unterminated argument vector", + )); + } + encoded[..encoded.len() - 1] + .split(|&byte| byte == 0) + .map(|bytes| { + if bytes.is_empty() { + return Err(Error::backend_error( + "nvxhost returned an empty OpenVMM argument", + )); + } + String::from_utf8(bytes.to_vec()) + .map(OsString::from) + .map_err(|error| { + Error::backend_error("nvxhost returned a non-UTF-8 OpenVMM argument") + .with_source(error) + }) + }) + .collect() + } + + fn failure(&self, error: *mut c_void) -> NativeFailure { + if error.is_null() || unsafe { (self.exports.error_count)(error) } == 0 { + return NativeFailure { + kind: 0, + os_error: None, + message: "nvxhost returned an error without details".to_owned(), + }; + } + let mut entry = NvxErrorEntry::default(); + if unsafe { (self.exports.error_get)(error, 0, &mut entry) } != STATUS_OK + || entry.struct_size as usize != size_of::() + { + return NativeFailure { + kind: 0, + os_error: None, + message: "nvxhost returned an incompatible error layout".to_owned(), + }; + } + let message = if entry.message_length == 0 { + "nvxhost returned an empty error message".to_owned() + } else if entry.message.is_null() { + "nvxhost returned an invalid error message pointer".to_owned() + } else { + String::from_utf8_lossy(unsafe { + slice::from_raw_parts(entry.message, entry.message_length) + }) + .into_owned() + }; + NativeFailure { + kind: entry.kind, + os_error: (entry.has_os_error != 0).then_some(entry.os_error), + message, + } + } + + fn check_status(&self, action: &str, status: i32, error: *mut c_void) -> Result<()> { + if status == STATUS_OK && error.is_null() { + return Ok(()); + } + let failure = self.failure(error); + if !error.is_null() { + unsafe { (self.exports.error_free)(error) }; + } + Err(failure.as_error(action, status)) + } +} + +pub(crate) struct LaunchInputs<'a> { + pub(crate) kernel: &'a Path, + pub(crate) initrd: &'a Path, + pub(crate) image: &'a Path, + pub(crate) control: &'a str, + pub(crate) boot: &'a str, + pub(crate) hypervisor: &'a str, + pub(crate) memory_mb: u32, + pub(crate) guest_debug: bool, +} + +#[derive(Debug)] +struct NativeFailure { + kind: u32, + os_error: Option, + message: String, +} + +impl NativeFailure { + fn as_error(&self, action: &str, status: i32) -> Error { + Error::backend_error(format!( + "nvxhost {action} failed (status {status}, kind {}): {}", + self.kind, self.message + )) + } + + fn endpoint_unavailable(&self) -> bool { + self.kind == ERROR_CONNECT + && match self.os_error { + #[cfg(windows)] + Some(2 | 231) => true, + #[cfg(target_os = "linux")] + Some(2 | 111) => true, + _ => false, + } + } +} + +struct Completed { + kind: u32, + token: u64, + result: i32, + frame_type: u8, + frame_flags: u8, + handle: u64, + data: Vec, + error: Option, +} + +#[derive(Clone, PartialEq, Message)] +struct TtrpcStatus { + #[prost(int32, tag = "1")] + code: i32, + #[prost(string, tag = "2")] + message: String, +} + +#[derive(Clone, PartialEq, Message)] +struct TtrpcResponse { + #[prost(message, optional, tag = "1")] + status: Option, + #[prost(bytes, tag = "2")] + payload: Vec, +} + +fn response_payload(method: &str, bytes: &[u8]) -> Result> { + let response = TtrpcResponse::decode(bytes).map_err(|error| { + Error::backend_error(format!("{method} returned an invalid ttrpc response")) + .with_source(error) + })?; + if let Some(status) = response.status + && status.code != 0 + { + return Err(Error::backend_error(format!( + "{method} failed with guest status {}: {}", + status.code, status.message + ))); + } + Ok(response.payload) +} + +pub(crate) struct Session { + host: Arc, + runtime: u64, + session: u64, + next_token: u64, +} + +fn disposal_deadline(operation: Option, now: Instant) -> Instant { + let maximum = now + Duration::from_secs(5); + operation.map_or(maximum, |deadline| deadline.min(maximum)) +} + +impl Session { + pub(crate) fn connect_verified( + host: Arc, + endpoint: &str, + capability: &[u8; 32], + expected_pid: u32, + deadline: Instant, + is_running: impl Fn() -> Result, + ) -> Result { + let mut runtime = 0; + let mut error = ptr::null_mut(); + let status = unsafe { (host.exports.runtime_create)(2, &mut runtime, &mut error) }; + host.check_status("starting a native runtime", status, error)?; + if runtime == 0 { + return Err(Error::backend_error( + "nvxhost returned an invalid runtime handle", + )); + } + let mut client = Self { + host, + runtime, + session: 0, + next_token: 1, + }; + loop { + let token = client.token()?; + let status = unsafe { + (client.host.exports.session_connect_verified)( + client.runtime, + endpoint.as_ptr(), + endpoint.len(), + capability.as_ptr(), + capability.len(), + expected_pid, + token, + ) + }; + client + .host + .check_status("connecting to OpenVMM", status, ptr::null_mut())?; + let completed = client.next(COMPLETION_CONNECT, token, Some(deadline))?; + if completed.result == RESULT_OK { + if completed.handle == 0 || completed.data.len() != 16 { + if completed.handle != 0 { + unsafe { (client.host.exports.session_release)(completed.handle) }; + } + return Err(Error::backend_error( + "nvxhost returned an invalid broker session identity", + )); + } + client.session = completed.handle; + return Ok(client); + } + if completed.result != RESULT_ERROR { + return Err(Error::backend_error(format!( + "nvxhost returned unexpected connection result {}", + completed.result + ))); + } + let failure = completed.error.ok_or_else(|| { + Error::backend_error("nvxhost connection failed without error details") + })?; + if failure.endpoint_unavailable() && Instant::now() < deadline && is_running()? { + thread::sleep(Duration::from_millis(25)); + continue; + } + return Err(failure.as_error("connecting to OpenVMM", completed.result)); + } + } + + fn token(&mut self) -> Result { + let token = self.next_token; + self.next_token = token + .checked_add(1) + .ok_or_else(|| Error::backend_error("nvxhost completion tokens have been exhausted"))?; + Ok(token) + } + + fn next(&self, kind: u32, token: u64, deadline: Option) -> Result { + let mut completion = ptr::null_mut(); + let status = loop { + let timeout_ms = match deadline { + Some(deadline) => { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + unsafe { (self.host.exports.operation_cancel)(self.runtime, token) }; + return Err(Error::backend_error("nvxhost operation timed out")); + } + u32::try_from(remaining.as_millis()) + .unwrap_or(u32::MAX) + .max(1) + } + None => 1_000, + }; + let status = unsafe { + (self.host.exports.runtime_next)(self.runtime, timeout_ms, &mut completion) + }; + if status == STATUS_TIMEOUT && deadline.is_none() { + continue; + } + break status; + }; + if status == STATUS_TIMEOUT { + unsafe { (self.host.exports.operation_cancel)(self.runtime, token) }; + return Err(Error::backend_error("nvxhost operation timed out")); + } + self.host + .check_status("waiting for a native completion", status, ptr::null_mut())?; + if completion.is_null() { + return Err(Error::backend_error( + "nvxhost returned a null completion pointer", + )); + } + let result = unsafe { + let frame = &*completion; + if frame.struct_size as usize != size_of::() { + Err(Error::backend_error( + "nvxhost returned an incompatible completion layout", + )) + } else if frame.data_length != 0 && frame.data.is_null() { + Err(Error::backend_error( + "nvxhost returned an invalid completion payload pointer", + )) + } else { + Ok(Completed { + kind: frame.kind, + token: frame.token, + result: frame.result, + frame_type: frame.frame_type, + frame_flags: frame.frame_flags, + handle: frame.handle, + data: if frame.data_length == 0 { + Vec::new() + } else { + slice::from_raw_parts(frame.data, frame.data_length).to_vec() + }, + error: (!frame.error.is_null()).then(|| self.host.failure(frame.error)), + }) + } + }; + unsafe { (self.host.exports.completion_release)(completion) }; + let result = result?; + if result.kind != kind || result.token != token { + if result.kind == COMPLETION_CONNECT && result.handle != 0 { + unsafe { (self.host.exports.session_release)(result.handle) }; + } + return Err(Error::backend_error( + "nvxhost returned a completion for a different operation", + )); + } + Ok(result) + } + + fn open_call(&self, method: &str, kind: u32) -> Result { + let mut call = 0; + let mut stream_id = 0; + let mut error = ptr::null_mut(); + let status = unsafe { + (self.host.exports.call_open)( + self.session, + method.as_ptr(), + method.len(), + kind, + &mut call, + &mut stream_id, + &mut error, + ) + }; + self.host + .check_status("opening a guest RPC", status, error)?; + if call == 0 || stream_id == 0 || stream_id & 1 == 0 { + if call != 0 { + unsafe { (self.host.exports.call_release)(call) }; + } + return Err(Error::backend_error( + "nvxhost returned an invalid RPC stream", + )); + } + Ok(Call { + host: Arc::clone(&self.host), + handle: call, + }) + } + + fn send(&mut self, call: &Call, payload: &[u8], deadline: Option) -> Result<()> { + let timeout_nanos = match deadline { + Some(deadline) => { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + return Err(Error::backend_error( + "guest RPC deadline expired before dispatch", + )); + } + i64::try_from(remaining.as_nanos()) + .map_err(|_| Error::backend_error("guest RPC deadline is too long"))? + } + None => 0, + }; + let token = self.token()?; + let status = unsafe { + (self.host.exports.call_send_request)( + call.handle, + payload.as_ptr(), + payload.len(), + FLAG_REMOTE_CLOSED, + timeout_nanos, + 0, + token, + ) + }; + self.host + .check_status("sending a guest RPC", status, ptr::null_mut())?; + let sent = self.next(COMPLETION_SEND, token, deadline)?; + self.completed_ok("sending a guest RPC", &sent) + } + + fn recv(&mut self, call: &Call, deadline: Option) -> Result { + let token = self.token()?; + let status = unsafe { (self.host.exports.call_recv)(call.handle, token) }; + self.host + .check_status("receiving a guest RPC", status, ptr::null_mut())?; + let frame = self.next(COMPLETION_RECV, token, deadline)?; + self.completed_ok("receiving a guest RPC", &frame)?; + Ok(frame) + } + + fn completed_ok(&self, action: &str, completed: &Completed) -> Result<()> { + if completed.result == RESULT_OK { + return Ok(()); + } + if completed.result != RESULT_ERROR { + return Err(Error::backend_error(format!( + "nvxhost {action} returned unexpected result {}", + completed.result + ))); + } + Err(completed.error.as_ref().map_or_else( + || Error::backend_error(format!("nvxhost {action} returned {}", completed.result)), + |failure| failure.as_error(action, completed.result), + )) + } + + pub(crate) fn unary( + &mut self, + method: &str, + payload: &[u8], + timeout: Option, + ) -> Result> { + let deadline = timeout + .map(|timeout| { + Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("guest RPC timeout is too long")) + }) + .transpose()?; + let call = self.open_call(method, CALL_UNARY)?; + self.send(&call, payload, deadline)?; + let response = self.recv(&call, deadline)?; + if response.frame_type != FRAME_RESPONSE { + return Err(Error::backend_error(format!( + "{method} returned unexpected ttrpc frame type {}", + response.frame_type + ))); + } + if unsafe { (self.host.exports.call_try_complete)(call.handle) } != 1 { + return Err(Error::backend_error(format!( + "{method} did not complete its unary stream" + ))); + } + response_payload(method, &response.data) + } + + pub(crate) fn server_stream( + &mut self, + method: &str, + payload: &[u8], + timeout: Duration, + mut on_message: impl FnMut(&[u8]) -> Result<()>, + ) -> Result<()> { + let deadline = Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("guest RPC timeout is too long"))?; + let call = self.open_call(method, CALL_SERVER_STREAM)?; + self.send(&call, payload, Some(deadline))?; + loop { + let frame = self.recv(&call, Some(deadline))?; + match frame.frame_type { + FRAME_DATA => { + if frame.frame_flags & FLAG_NO_DATA == 0 { + on_message(&frame.data)?; + } + if frame.frame_flags & FLAG_REMOTE_CLOSED != 0 { + break; + } + } + FRAME_RESPONSE => { + let final_payload = response_payload(method, &frame.data)?; + if !final_payload.is_empty() { + on_message(&final_payload)?; + } + break; + } + other => { + return Err(Error::backend_error(format!( + "{method} returned unexpected ttrpc frame type {other}" + ))); + } + } + } + if unsafe { (self.host.exports.call_try_complete)(call.handle) } != 1 { + return Err(Error::backend_error(format!( + "{method} did not complete its server stream" + ))); + } + Ok(()) + } + + pub(crate) fn close(mut self, deadline: Option) -> Result<()> { + let now = Instant::now(); + let deadline = disposal_deadline(deadline, now); + if now >= deadline { + return Err(Error::backend_error( + "guest session disposal exceeded the operation deadline", + )); + } + let token = self.token()?; + let status = unsafe { (self.host.exports.session_dispose)(self.session, token) }; + self.host + .check_status("closing a guest session", status, ptr::null_mut())?; + let disposed = self.next(COMPLETION_DISPOSE, token, Some(deadline))?; + self.completed_ok("closing a guest session", &disposed)?; + let status = unsafe { (self.host.exports.session_release)(self.session) }; + self.host + .check_status("releasing a guest session", status, ptr::null_mut())?; + self.session = 0; + Ok(()) + } +} + +impl Drop for Session { + fn drop(&mut self) { + if self.session != 0 { + let status = unsafe { (self.host.exports.session_release)(self.session) }; + if status != STATUS_OK { + eprintln!("nvxhost failed to release guest session: {status}"); + } + } + if self.runtime != 0 { + let status = unsafe { (self.host.exports.runtime_free)(self.runtime) }; + if status != STATUS_OK { + eprintln!("nvxhost failed to release native runtime: {status}"); + } + } + } +} + +struct Call { + host: Arc, + handle: u64, +} + +impl Drop for Call { + fn drop(&mut self) { + let status = unsafe { (self.host.exports.call_release)(self.handle) }; + if status != STATUS_OK { + eprintln!("nvxhost failed to release guest RPC: {status}"); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn disposal_never_extends_a_short_operation_deadline() { + let now = Instant::now(); + let short = now + Duration::from_millis(100); + assert_eq!(disposal_deadline(Some(short), now), short); + let maximum = now + Duration::from_secs(5); + assert_eq!( + disposal_deadline(Some(now + Duration::from_secs(60)), now), + maximum + ); + assert_eq!(disposal_deadline(None, now), maximum); + let expired = now - Duration::from_millis(1); + assert_eq!(disposal_deadline(Some(expired), now), expired); + } + + #[test] + fn ttrpc_response_decodes_the_existing_wire_vector() { + let bytes = [0x0a, 0x00, 0x12, 0x06, 0x7a, 0x04, b'r', b'u', b's', b't']; + assert_eq!( + response_payload("GetGuestInfo", &bytes).unwrap(), + b"z\x04rust" + ); + } + + #[test] + fn missing_library_fails_explicitly() { + let path = std::env::temp_dir().join("nvxhost-missing-file.dll"); + let error = HostLibrary::load(&path, &[0; 32]).unwrap_err(); + assert!(error.message().contains("cannot locate nvxhost")); + } + + #[test] + #[ignore = "set NVXHOST_TEST_LIBRARY to a separately built nvxhost DLL or shared library"] + fn privately_built_library_exposes_the_ramfs_launch_abi() { + let path = std::env::var_os("NVXHOST_TEST_LIBRARY") + .map(std::path::PathBuf::from) + .expect("NVXHOST_TEST_LIBRARY must name a private nvxhost build"); + let bytes = fs::read(&path).unwrap(); + let digest: [u8; 32] = Sha256::digest(bytes).into(); + let host = HostLibrary::load(&path, &digest).unwrap(); + let error = HostLibrary::load(&path, &[0; 32]).unwrap_err(); + assert!( + error + .message() + .contains("does not match the approved SHA-256") + ); + let args = host + .launch_arguments(&LaunchInputs { + kernel: Path::new(r"C:\test\vmlinux"), + initrd: Path::new(r"C:\test\edge-initramfs.cpio.gz"), + image: Path::new(r"C:\test\image.gpt"), + control: "//./pipe/openvmm-microvm-edge-control", + boot: "//./pipe/openvmm-microvm-edge-boot", + hypervisor: "whp", + memory_mb: 256, + guest_debug: false, + }) + .unwrap(); + let args: Vec<_> = args + .iter() + .map(|argument| argument.to_str().unwrap()) + .collect(); + assert!(args.contains(&"--microvm-control-auth-stdin")); + assert!(args.contains(&"distro:file:C:\\test\\image.gpt,ro")); + assert_eq!( + args.iter() + .filter(|&&argument| argument == "--microvm-sandbox-block") + .count(), + 1 + ); + } +} diff --git a/aci_edge_sandboxes/src/openvmm/config.rs b/aci_edge_sandboxes/src/openvmm/config.rs index 5cbd5253b..61536dc87 100644 --- a/aci_edge_sandboxes/src/openvmm/config.rs +++ b/aci_edge_sandboxes/src/openvmm/config.rs @@ -349,23 +349,29 @@ impl OpenVmmConfig { )); } } - if cfg!(unix) { - // sockaddr_un holds 108 bytes including the terminating NUL. - let socket = self - .state_root - .join("0".repeat(32)) - .join(super::state::SOCKET_NAME); - if socket.as_os_str().len() >= 108 { - return invalid(format!( - "state_root {} is too long for a Unix control socket path", - self.state_root.display() - )); - } - } + validate_unix_socket_path(&self.state_root, super::state::SOCKET_NAME, "control")?; Ok(()) } } +pub(crate) fn validate_unix_socket_path( + state_root: &Path, + socket_name: &str, + purpose: &str, +) -> Result<()> { + if cfg!(unix) { + // sockaddr_un holds 108 bytes including the terminating NUL. + let socket = state_root.join("0".repeat(32)).join(socket_name); + if socket.as_os_str().len() >= 108 { + return Err(Error::backend_unavailable(format!( + "state_root {} is too long for a Unix {purpose} socket path", + state_root.display() + ))); + } + } + Ok(()) +} + fn valid_hostname(hostname: &str) -> bool { let bytes = hostname.as_bytes(); !bytes.is_empty() diff --git a/aci_edge_sandboxes/src/openvmm/images.rs b/aci_edge_sandboxes/src/openvmm/images.rs new file mode 100644 index 000000000..c86891524 --- /dev/null +++ b/aci_edge_sandboxes/src/openvmm/images.rs @@ -0,0 +1,891 @@ +//! Registry of guest images, which keeps hashing off the sandbox start path. +//! +//! ```text +//! /nvxhost/images/ +//! .lock serializes adding, removing, and using registrations +//! .json path and seal of one registered image content +//! ``` +//! +//! An image is hashed once, when it is registered, or not at all when the caller supplies a +//! digest that its own policy already verified. The registration keeps the file's [`FileSeal`]; +//! provisioning and starting a sandbox compare the file's current seal with it and fail closed +//! when they differ, rather than hashing the image again. The backend checks its OpenVMM runtime +//! files the same way: it hashes them once, when it is created, and later compares seals. + +use std::fmt; +use std::fs::{self, File}; +use std::io::{self, BufReader, Seek}; +use std::path::{Path, PathBuf}; +use std::str::FromStr; + +use serde::{Deserialize, Deserializer, Serialize, Serializer}; +use sha2_runtime::{Digest, Sha256}; + +use super::platform::{self, FileSeal}; +use super::state::{self, LockGuard}; +use crate::error::{Error, Result}; + +const RECORD_FORMAT: u32 = 1; +const LOCK_NAME: &str = ".lock"; +const RECORD_SUFFIX: &str = ".json"; +const HASH_BUFFER_BYTES: usize = 1 << 20; + +/// Content identity of a registered guest image: the SHA-256 digest of the image file. +/// +/// An ID displays and parses as `sha256:` followed by 64 lowercase hexadecimal digits. +#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub struct ImageId([u8; 32]); + +impl ImageId { + const PREFIX: &'static str = "sha256:"; + + /// Returns the ID of image content with this SHA-256 digest. + pub const fn from_sha256(digest: [u8; 32]) -> Self { + Self(digest) + } + + /// Returns the SHA-256 digest of the image content. + pub const fn sha256(&self) -> &[u8; 32] { + &self.0 + } + + /// Parses an ID, returning + /// [`ErrorCode::MalformedRequest`](crate::ErrorCode::MalformedRequest) unless it is `sha256:` + /// followed by 64 lowercase hexadecimal digits. + pub fn parse(value: &str) -> Result { + value + .strip_prefix(Self::PREFIX) + .and_then(decode_digest) + .map(Self) + .ok_or_else(|| { + Error::malformed_request(format!( + "image ID {value:?} must be sha256: followed by 64 lowercase hexadecimal \ + digits" + )) + }) + } +} + +impl fmt::Display for ImageId { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(formatter, "{}{}", Self::PREFIX, encode_hex(&self.0)) + } +} + +impl fmt::Debug for ImageId { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(formatter, "ImageId({self})") + } +} + +impl FromStr for ImageId { + type Err = Error; + + fn from_str(value: &str) -> Result { + Self::parse(value) + } +} + +impl Serialize for ImageId { + fn serialize(&self, serializer: S) -> std::result::Result { + serializer.collect_str(self) + } +} + +impl<'de> Deserialize<'de> for ImageId { + fn deserialize>(deserializer: D) -> std::result::Result { + let value = String::deserialize(deserializer)?; + Self::parse(&value).map_err(|error| serde::de::Error::custom(error.message())) + } +} + +/// How the backend establishes the content digest of an image that it registers. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +#[non_exhaustive] +pub enum ImageDigest { + /// Hashes the file. + #[default] + Compute, + /// Hashes the file and requires this SHA-256 digest. + Expect([u8; 32]), + /// Records this SHA-256 digest without reading the file. + /// + /// Use it only for content that the caller's own policy has verified and protects from + /// writers, for example a file that an installer hashed before making it read-only. + Trusted([u8; 32]), +} + +/// A registered guest image. +#[derive(Debug, Clone, PartialEq, Eq)] +#[non_exhaustive] +pub struct RegisteredImage { + /// Content identity. + pub id: ImageId, + /// Absolute path that sandboxes attach the image from. + pub path: PathBuf, + /// Length in bytes at registration. + pub length: u64, + /// Whether the caller supplied the digest instead of the backend hashing the file. + pub trusted: bool, + /// Whether the file still matches its registration. A sandbox whose image changed does not + /// start until the image is registered again. + pub intact: bool, +} + +/// Persisted registration of one image content. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub(crate) struct ImageRecord { + format: u32, + pub(crate) id: ImageId, + pub(crate) path: PathBuf, + trusted: bool, + seal: FileSeal, +} + +impl ImageRecord { + pub(crate) fn describe(&self, intact: bool) -> RegisteredImage { + RegisteredImage { + id: self.id, + path: self.path.clone(), + length: self.seal.length, + trusted: self.trusted, + intact, + } + } + + fn subject(&self) -> String { + format!("image {} at {}", self.id, self.path.display()) + } + + /// Hashes the image through `file`, a handle from [`ImageStore::open_checked`], and compares + /// the digest with the image's ID. + pub(crate) fn verify_content(&self, file: &mut File) -> Result<()> { + verify_digest(file, &self.id.0, &self.subject()) + } +} + +/// Directory of image registrations shared by every backend with the same state root. +#[derive(Debug)] +pub(crate) struct ImageStore { + dir: PathBuf, +} + +impl ImageStore { + /// Opens the registry in `dir`, creating it and restricting it to the current user if needed. + pub(crate) fn open(dir: &Path) -> io::Result { + platform::create_private_dir(dir)?; + Ok(Self { + dir: dir.to_path_buf(), + }) + } + + /// Takes the registry lock, which serializes adding and removing registrations with recording + /// sandboxes that use them. + pub(crate) fn lock(&self) -> Result { + state::lock(&self.dir.join(LOCK_NAME)) + } + + fn record_path(&self, id: &ImageId) -> PathBuf { + self.dir + .join(format!("{}{RECORD_SUFFIX}", encode_hex(&id.0))) + } + + /// Loads the registration of `id`, if there is one. + pub(crate) fn get(&self, id: &ImageId) -> Result> { + let path = self.record_path(id); + match state::read_json::(&path)? { + Some(record) if record.format != RECORD_FORMAT || record.id != *id => { + Err(Error::backend_error(format!( + "image record {} does not describe {id}", + path.display() + ))) + } + record => Ok(record), + } + } + + /// Loads every registration, ordered by ID. + pub(crate) fn list(&self) -> Result> { + let list_error = || { + Error::backend_error(format!( + "cannot list the image registry {}", + self.dir.display() + )) + }; + let mut records = Vec::new(); + for entry in fs::read_dir(&self.dir).map_err(|error| list_error().with_source(error))? { + let name = entry + .map_err(|error| list_error().with_source(error))? + .file_name(); + let Some(digest) = name + .to_str() + .and_then(|name| name.strip_suffix(RECORD_SUFFIX)) + .and_then(decode_digest) + else { + continue; + }; + // A registration removed since the listing no longer exists. + if let Some(record) = self.get(&ImageId(digest))? { + records.push(record); + } + } + records.sort_by_key(|record| record.id); + Ok(records) + } + + /// Registers the image at the absolute `path`, reusing a registration that the file still + /// matches without reading the file. + /// + /// The caller must not hold the registry lock, which this takes to record a new + /// registration. + pub(crate) fn register(&self, path: &Path, digest: ImageDigest) -> Result { + let subject = format!("image {}", path.display()); + let mut file = open(path, true, &subject)?; + let seal = seal(&file, &subject)?; + if let Some(record) = self + .list()? + .into_iter() + .find(|record| record.path == path && record.seal == seal) + { + return match digest { + ImageDigest::Expect(expected) | ImageDigest::Trusted(expected) + if record.id.0 != expected => + { + Err(Error::backend_unavailable(format!( + "{subject} is registered as {}, not {}", + record.id, + ImageId(expected) + ))) + } + _ => Ok(record), + }; + } + let (id, trusted) = match digest { + ImageDigest::Trusted(expected) => (ImageId(expected), true), + ImageDigest::Compute | ImageDigest::Expect(_) => { + let id = ImageId(hash(&mut file, &subject)?); + if self::seal(&file, &subject)? != seal { + return Err(Error::backend_unavailable(format!( + "{subject} changed while it was hashed" + ))); + } + if let ImageDigest::Expect(expected) = digest + && id.0 != expected + { + return Err(Error::backend_unavailable(format!( + "{subject} is {id}, not the expected {}", + ImageId(expected) + ))); + } + (id, false) + } + }; + let record = ImageRecord { + format: RECORD_FORMAT, + id, + path: path.to_path_buf(), + trusted, + seal, + }; + let _registry = self.lock()?; + state::write_json(&self.record_path(&id), &record)?; + Ok(record) + } + + /// Opens a registered image and checks that it still matches its registration. + /// + /// With `deny_writers`, the handle keeps writers out on Windows until it is dropped. + pub(crate) fn open_checked(&self, record: &ImageRecord, deny_writers: bool) -> Result { + open_sealed( + &record.path, + &record.seal, + deny_writers, + &record.subject(), + "it was registered; register it again", + ) + } + + /// Hashes a registered image and compares the digest with its ID. + pub(crate) fn verify(&self, record: &ImageRecord) -> Result<()> { + record.verify_content(&mut self.open_checked(record, true)?) + } + + /// Removes the registration of `id`, returning whether it existed. + /// + /// The caller holds the registry lock and has checked that no sandbox uses the image. + pub(crate) fn remove(&self, id: &ImageId) -> Result { + let path = self.record_path(id); + match fs::remove_file(&path) { + Ok(()) => Ok(true), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(false), + Err(error) => Err( + Error::backend_error(format!("cannot remove {}", path.display())) + .with_source(error), + ), + } + } +} + +/// A runtime file that the backend hashed once, with the seal that it had then. +#[derive(Debug)] +pub(crate) struct VerifiedFile { + subject: String, + path: PathBuf, + sha256: [u8; 32], + seal: FileSeal, +} + +impl VerifiedFile { + /// Hashes the file at `path`, requiring the `approved` digest if there is one. + pub(crate) fn verify( + path: &Path, + description: &str, + approved: Option<[u8; 32]>, + ) -> Result { + let subject = format!("the {description} {}", path.display()); + let mut file = open(path, true, &subject)?; + let seal = seal(&file, &subject)?; + let sha256 = hash(&mut file, &subject)?; + if self::seal(&file, &subject)? != seal { + return Err(Error::backend_unavailable(format!( + "{subject} changed while it was hashed" + ))); + } + if approved.is_some_and(|approved| approved != sha256) { + return Err(Error::backend_unavailable(format!( + "{subject} does not match the approved SHA-256" + ))); + } + Ok(Self { + subject, + path: path.to_path_buf(), + sha256, + seal, + }) + } + + pub(crate) fn sha256(&self) -> &[u8; 32] { + &self.sha256 + } + + /// Opens the file, keeping writers out on Windows until the handle is dropped, and checks + /// that it has not changed since it was hashed. + pub(crate) fn open_checked(&self) -> Result { + open_sealed( + &self.path, + &self.seal, + true, + &self.subject, + "the backend verified it; create the backend again", + ) + } + + /// Hashes the file again through `file`, a handle from [`Self::open_checked`]. + pub(crate) fn verify_content(&self, file: &mut File) -> Result<()> { + verify_digest(file, &self.sha256, &self.subject) + } +} + +fn open(path: &Path, deny_writers: bool, subject: &str) -> Result { + platform::open_sealable(path, deny_writers).map_err(|error| { + Error::backend_unavailable(format!("cannot open {subject}: {error}")).with_source(error) + }) +} + +fn seal(file: &File, subject: &str) -> Result { + platform::file_seal(file).map_err(|error| { + Error::backend_unavailable(format!("cannot inspect {subject}: {error}")).with_source(error) + }) +} + +/// Opens `path` and checks that its seal is still `expected`. +fn open_sealed( + path: &Path, + expected: &FileSeal, + deny_writers: bool, + subject: &str, + since: &str, +) -> Result { + let file = open(path, deny_writers, subject)?; + if seal(&file, subject)? == *expected { + Ok(file) + } else { + Err(Error::backend_unavailable(format!( + "{subject} changed since {since}" + ))) + } +} + +fn verify_digest(file: &mut File, expected: &[u8; 32], subject: &str) -> Result<()> { + if hash(file, subject)? == *expected { + Ok(()) + } else { + Err(Error::backend_unavailable(format!( + "{subject} no longer matches its SHA-256" + ))) + } +} + +/// Hashes a whole file from its start. +fn hash(file: &mut File, subject: &str) -> Result<[u8; 32]> { + fn digest(file: &mut File) -> io::Result<[u8; 32]> { + file.rewind()?; + let mut hasher = Sha256::new(); + io::copy( + &mut BufReader::with_capacity(HASH_BUFFER_BYTES, file), + &mut hasher, + )?; + Ok(hasher.finalize().into()) + } + digest(file).map_err(|error| { + Error::backend_unavailable(format!("cannot hash {subject}: {error}")).with_source(error) + }) +} + +/// Returns `bytes` as lowercase hexadecimal digits. +pub(crate) fn encode_hex(bytes: &[u8]) -> String { + bytes.iter().map(|byte| format!("{byte:02x}")).collect() +} + +fn decode_digest(hex: &str) -> Option<[u8; 32]> { + if hex.len() != 64 + || !hex + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return None; + } + let mut digest = [0u8; 32]; + for (index, byte) in digest.iter_mut().enumerate() { + *byte = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16).ok()?; + } + Some(digest) +} + +#[cfg(test)] +mod tests { + use std::io::Write; + use std::time::{Duration, SystemTime}; + + use super::*; + use crate::ErrorCode; + + fn sha256(bytes: &[u8]) -> [u8; 32] { + Sha256::digest(bytes).into() + } + + fn registry(root: &Path) -> ImageStore { + ImageStore::open(&root.join("images")).unwrap() + } + + fn image(root: &Path, name: &str, content: &[u8]) -> PathBuf { + let path = root.join(name); + fs::write(&path, content).unwrap(); + path + } + + /// Moves the file's modification time without changing its content. + fn touch(path: &Path) { + let file = File::options().write(true).open(path).unwrap(); + let modified = file.metadata().unwrap().modified().unwrap(); + file.set_modified(modified + Duration::from_secs(2)) + .unwrap(); + } + + fn unavailable(result: Result, expected: &str) { + let error = result.unwrap_err(); + assert_eq!(error.code(), ErrorCode::BackendUnavailable, "{error}"); + assert!(error.message().contains(expected), "{error}"); + } + + #[test] + fn image_ids_display_parse_and_serialize_as_prefixed_digests() { + let id = ImageId::from_sha256([0xab; 32]); + let text = format!("sha256:{}", "ab".repeat(32)); + assert_eq!(id.to_string(), text); + assert_eq!(ImageId::parse(&text).unwrap(), id); + assert_eq!(text.parse::().unwrap(), id); + assert_eq!(id.sha256(), &[0xab; 32]); + assert_eq!(serde_json::to_string(&id).unwrap(), format!("\"{text}\"")); + assert_eq!( + serde_json::from_str::(&format!("\"{text}\"")).unwrap(), + id + ); + for malformed in [ + String::new(), + "ab".repeat(32), + format!("sha512:{}", "ab".repeat(32)), + format!("sha256:{}", "AB".repeat(32)), + format!("sha256:{}", "ab".repeat(31)), + format!("sha256:{}0", "ab".repeat(32)), + format!("sha256:{}+0", "ab".repeat(31)), + ] { + let error = ImageId::parse(&malformed).unwrap_err(); + assert_eq!(error.code(), ErrorCode::MalformedRequest, "{malformed:?}"); + assert!(serde_json::from_str::(&format!("{malformed:?}")).is_err()); + } + } + + #[test] + fn registration_hashes_once_and_is_reused_while_the_file_is_unchanged() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + assert_eq!(record.id, ImageId(sha256(b"image content"))); + assert_eq!( + record.describe(true), + RegisteredImage { + id: record.id, + path: path.clone(), + length: 13, + trusted: false, + intact: true, + } + ); + + // Another process sees the same registration, and the expected digest matches it. + let reopened = registry(root.path()); + assert_eq!(reopened.list().unwrap(), std::slice::from_ref(&record)); + assert_eq!(reopened.get(&record.id).unwrap(), Some(record.clone())); + assert_eq!( + reopened + .register(&path, ImageDigest::Expect(record.id.0)) + .unwrap(), + record + ); + drop(reopened.open_checked(&record, true).unwrap()); + reopened.verify(&record).unwrap(); + } + + #[test] + fn trusted_digests_are_recorded_without_hashing_and_then_reused() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + // The content does not have this digest, so a registration that hashed would differ. + let claimed = [7; 32]; + let record = store + .register(&path, ImageDigest::Trusted(claimed)) + .unwrap(); + assert_eq!(record.id, ImageId(claimed)); + assert!(record.describe(true).trusted); + assert_eq!(store.register(&path, ImageDigest::Compute).unwrap(), record); + assert_eq!( + store + .register(&path, ImageDigest::Trusted(claimed)) + .unwrap(), + record + ); + // The diagnostic hash exposes a trusted digest that the content does not have. + unavailable(store.verify(&record), "no longer matches its SHA-256"); + + for conflicting in [ + ImageDigest::Trusted(sha256(b"image content")), + ImageDigest::Expect(sha256(b"image content")), + ] { + unavailable( + store.register(&path, conflicting), + &format!("is registered as {}", record.id), + ); + } + assert_eq!(store.list().unwrap(), [record]); + } + + #[test] + fn an_unexpected_digest_registers_nothing() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + unavailable( + store.register(&path, ImageDigest::Expect([1; 32])), + &format!("not the expected {}", ImageId([1; 32])), + ); + assert!(store.list().unwrap().is_empty()); + unavailable( + store.register(&root.path().join("missing.vhd"), ImageDigest::Compute), + "cannot open image", + ); + unavailable(store.register(root.path(), ImageDigest::Compute), "image"); + assert!(store.list().unwrap().is_empty()); + } + + #[test] + fn a_changed_image_fails_closed_until_it_is_registered_again() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + + // Only the timestamp moves: the content, and thus the ID, stays the same. + touch(&path); + unavailable( + store.open_checked(&record, false), + "changed since it was registered; register it again", + ); + unavailable(store.verify(&record), "changed since it was registered"); + let refreshed = store.register(&path, ImageDigest::Compute).unwrap(); + assert_eq!(refreshed.id, record.id); + assert_ne!(refreshed.seal, record.seal); + assert_eq!(store.list().unwrap(), std::slice::from_ref(&refreshed)); + drop(store.open_checked(&refreshed, true).unwrap()); + + // New content gets a new ID; the old registration stays, but no longer matches. + File::options() + .append(true) + .open(&path) + .unwrap() + .write_all(b" changed") + .unwrap(); + unavailable(store.open_checked(&refreshed, true), "changed since"); + let changed = store.register(&path, ImageDigest::Compute).unwrap(); + assert_eq!(changed.id, ImageId(sha256(b"image content changed"))); + assert_eq!(store.list().unwrap().len(), 2); + drop(store.open_checked(&changed, false).unwrap()); + } + + #[test] + fn a_replaced_file_fails_closed_even_with_identical_content_and_times() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + let modified = fs::metadata(&path).unwrap().modified().unwrap(); + let replacement = image(root.path(), "replacement.vhd", b"image content"); + File::options() + .write(true) + .open(&replacement) + .unwrap() + .set_modified(modified) + .unwrap(); + fs::remove_file(&path).unwrap(); + fs::rename(&replacement, &path).unwrap(); + unavailable(store.open_checked(&record, false), "changed since"); + } + + #[test] + fn the_same_content_moves_to_its_latest_registered_location() { + let root = tempfile::tempdir().unwrap(); + let first = image(root.path(), "first.vhd", b"image content"); + let second = image(root.path(), "second.vhd", b"image content"); + let store = registry(root.path()); + let original = store.register(&first, ImageDigest::Compute).unwrap(); + let moved = store.register(&second, ImageDigest::Compute).unwrap(); + assert_eq!(moved.id, original.id); + assert_eq!(store.get(&original.id).unwrap().unwrap().path, second); + assert_eq!(store.list().unwrap(), [moved]); + } + + #[test] + fn records_are_private_and_unrelated_files_are_ignored() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + let dir = root.path().join("images"); + for name in [".lock", ".record.json.123.tmp", "notes.json", "README"] { + fs::write(dir.join(name), b"not a record").unwrap(); + } + assert_eq!(store.list().unwrap(), std::slice::from_ref(&record)); + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + let mode = |path: PathBuf| fs::metadata(path).unwrap().permissions().mode() & 0o777; + assert_eq!(mode(dir.clone()), 0o700); + assert_eq!(mode(store.record_path(&record.id)), 0o600); + } + + fs::write(store.record_path(&record.id), b"{ truncated").unwrap(); + assert_eq!(store.list().unwrap_err().code(), ErrorCode::BackendError); + let other = ImageId([9; 32]); + fs::write( + store.record_path(&other), + serde_json::to_vec(&record).unwrap(), + ) + .unwrap(); + fs::remove_file(store.record_path(&record.id)).unwrap(); + let error = store.get(&other).unwrap_err(); + assert!(error.message().contains("does not describe"), "{error}"); + } + + #[test] + fn removal_reports_whether_a_registration_existed() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + let _registry = store.lock().unwrap(); + assert!(store.remove(&record.id).unwrap()); + assert!(!store.remove(&record.id).unwrap()); + assert_eq!(store.get(&record.id).unwrap(), None); + assert!(path.exists()); + } + + #[test] + fn runtime_files_are_hashed_once_and_then_checked_by_seal() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "vmlinux", b"kernel"); + let verified = VerifiedFile::verify(&path, "guest kernel", None).unwrap(); + assert_eq!(verified.sha256(), &sha256(b"kernel")); + assert_eq!( + VerifiedFile::verify(&path, "guest kernel", Some(sha256(b"kernel"))) + .unwrap() + .sha256(), + &sha256(b"kernel") + ); + unavailable( + VerifiedFile::verify(&path, "guest kernel", Some([0; 32])), + "the guest kernel", + ); + let mut handle = verified.open_checked().unwrap(); + verified.verify_content(&mut handle).unwrap(); + drop(handle); + + touch(&path); + unavailable( + verified.open_checked(), + "changed since the backend verified it; create the backend again", + ); + } + + #[cfg(windows)] + #[test] + fn sealed_handles_keep_writers_out_and_writers_block_registration() { + const ERROR_SHARING_VIOLATION: i32 = 32; + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + + let held = store.open_checked(&record, true).unwrap(); + let error = File::options().write(true).open(&path).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(ERROR_SHARING_VIOLATION)); + assert!(fs::remove_file(&path).is_err()); + // Readers, including a second start of the same image, still share it. + drop(File::open(&path).unwrap()); + drop(store.open_checked(&record, true).unwrap()); + drop(held); + + let writer = File::options().write(true).open(&path).unwrap(); + unavailable( + store.register(&path, ImageDigest::Compute), + "cannot open image", + ); + unavailable(store.open_checked(&record, true), "cannot open image"); + // Listing does not deny writers, so it still inspects an image that is open for writing. + drop(store.open_checked(&record, false).unwrap()); + drop(writer); + } + + /// Restoring the last-write time after a write preserves the seal on Windows, so only content + /// verification detects the change. + #[cfg(windows)] + #[test] + fn content_verification_detects_a_change_that_preserves_the_seal() { + use std::os::windows::io::AsRawHandle; + + use windows_sys::Win32::Storage::FileSystem::{ + FILE_BASIC_INFO, FileBasicInfo, GetFileInformationByHandleEx, + SetFileInformationByHandle, + }; + + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let store = registry(root.path()); + let record = store.register(&path, ImageDigest::Compute).unwrap(); + + let mut file = File::options().read(true).write(true).open(&path).unwrap(); + let handle = file.as_raw_handle(); + let mut times = FILE_BASIC_INFO::default(); + let size = std::mem::size_of::() as u32; + // SAFETY: the handle is open and the buffer is a FILE_BASIC_INFO of the size passed. + assert_ne!( + unsafe { + GetFileInformationByHandleEx(handle, FileBasicInfo, (&raw mut times).cast(), size) + }, + 0 + ); + file.write_all(b"IMAGE").unwrap(); + // SAFETY: as above; the handle has write access. + assert_ne!( + unsafe { + SetFileInformationByHandle(handle, FileBasicInfo, (&raw const times).cast(), size) + }, + 0 + ); + drop(file); + assert_eq!(fs::read(&path).unwrap(), b"IMAGE content"); + + drop(store.open_checked(&record, true).unwrap()); + unavailable(store.verify(&record), "no longer matches its SHA-256"); + } + + #[cfg(target_os = "linux")] + #[test] + fn special_files_are_refused_without_blocking() { + use std::ffi::CString; + use std::os::unix::ffi::OsStrExt; + + let root = tempfile::tempdir().unwrap(); + let fifo = root.path().join("image.fifo"); + let name = CString::new(fifo.as_os_str().as_bytes()).unwrap(); + // SAFETY: the path is a NUL-terminated string. + assert_eq!(unsafe { libc::mkfifo(name.as_ptr(), 0o600) }, 0); + let store = registry(root.path()); + unavailable( + store.register(&fifo, ImageDigest::Compute), + "does not name a regular file", + ); + assert!(store.list().unwrap().is_empty()); + } + + #[test] + fn seal_changes_are_visible_to_new_handles() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let sealed = + |path: &Path| platform::file_seal(&platform::open_sealable(path, false).unwrap()); + let first = sealed(&path).unwrap(); + assert_eq!(sealed(&path).unwrap(), first); + assert_eq!(first.length, 13); + let modified = SystemTime::now() + Duration::from_secs(60); + File::options() + .write(true) + .open(&path) + .unwrap() + .set_modified(modified) + .unwrap(); + let second = sealed(&path).unwrap(); + assert_ne!(second.modified, first.modified); + assert_eq!((&second.file, second.volume), (&first.file, first.volume)); + } + + /// Windows security components update a file's change time when they cache its hash in an + /// extended attribute, so Windows seals ignore metadata-only changes. Linux seals keep the + /// change time, which writers cannot restore. + #[test] + fn metadata_only_changes_alter_the_seal_only_where_the_change_time_is_sealed() { + let root = tempfile::tempdir().unwrap(); + let path = image(root.path(), "image.vhd", b"image content"); + let sealed = |path: &Path| { + platform::file_seal(&platform::open_sealable(path, false).unwrap()).unwrap() + }; + let before = sealed(&path); + // Let a coarse change-time clock advance. + std::thread::sleep(Duration::from_millis(50)); + let mut permissions = fs::metadata(&path).unwrap().permissions(); + permissions.set_readonly(true); + fs::set_permissions(&path, permissions.clone()).unwrap(); + let after = sealed(&path); + assert_eq!(after.modified, before.modified); + assert_eq!(after == before, cfg!(windows), "{before:?} -> {after:?}"); + // A read-only file would outlive the temporary directory on Windows. + #[cfg(windows)] + { + #[allow(clippy::permissions_set_readonly_false)] + permissions.set_readonly(false); + fs::set_permissions(&path, permissions).unwrap(); + } + } +} diff --git a/aci_edge_sandboxes/src/openvmm/launch.rs b/aci_edge_sandboxes/src/openvmm/launch.rs index db3a67ee5..2f0441564 100644 --- a/aci_edge_sandboxes/src/openvmm/launch.rs +++ b/aci_edge_sandboxes/src/openvmm/launch.rs @@ -107,6 +107,7 @@ mod tests { backend: BACKEND_KEY.to_owned(), network: None, filesystem: None, + native: None, memory_mib: 256, workload_uid: 65534, workload_gid: 65534, diff --git a/aci_edge_sandboxes/src/openvmm/mod.rs b/aci_edge_sandboxes/src/openvmm/mod.rs index e0d07f864..db2386914 100644 --- a/aci_edge_sandboxes/src/openvmm/mod.rs +++ b/aci_edge_sandboxes/src/openvmm/mod.rs @@ -91,7 +91,11 @@ mod artifacts; mod config; mod contract; mod filesystem; +#[cfg(feature = "nvxhost")] +mod images; mod launch; +#[cfg(feature = "nvxhost")] +mod native; mod network; mod platform; mod process; @@ -111,6 +115,10 @@ use serde_json::Value; pub use self::artifacts::Artifacts; pub use self::config::{Hypervisor, OpenVmmConfig}; pub use self::filesystem::{guest_path, resolve_guest_path}; +#[cfg(feature = "nvxhost")] +pub use self::images::{ImageDigest, ImageId, RegisteredImage}; +#[cfg(feature = "nvxhost")] +pub use self::native::{NvxHostBackend, NvxHostConfig, RuntimeDigests}; use self::protocol::{ CAPABILITY_LEN, ExitCategory, GuestFeatures, MAX_ARGUMENT_BYTES, MAX_CWD_BYTES, MAX_OUTPUT_BYTES, MAX_TIMEOUT_MS, Workload, WorkloadEnvironment, @@ -211,7 +219,8 @@ impl OpenVmmBackend { /// /// Lifecycle locks serialize starts, so a launch marker seen under the lock always belongs /// to an interrupted start. Recorded identity is authoritative even before an endpoint - /// exists; legacy markers can be recovered only when their endpoint identifies the child. + /// exists; other markers can be recovered only when their endpoint identifies the child, or + /// when a released claim on the OpenVMM log proves that nothing of the launch still runs. fn recover_launch(&self, sandbox_id: &SandboxId, launch: &LaunchRecord) -> Result<()> { let interrupted = |detail: &str| { Error::backend_error(format!( @@ -221,13 +230,8 @@ impl OpenVmmBackend { let identity = match &launch.process { Some(identity) => identity.clone(), None => { - let pid = platform::endpoint_server_pid(&launch.endpoint) - .map_err(|error| interrupted("cannot be recovered yet").with_source(error))?; - let Some(pid) = pid else { - return Err(interrupted( - "has no recorded process identity or observable endpoint; cannot verify \ - that OpenVMM exited, so the launch marker is retained", - )); + let Some(pid) = self.launched_server(sandbox_id, launch)? else { + return Ok(()); }; let Some(start_time) = platform::process_start_time(pid) .map_err(|error| interrupted("cannot be recovered yet").with_source(error))? @@ -251,6 +255,52 @@ impl OpenVmmBackend { } } + /// Finds the process that serves the endpoint of an interrupted start that recorded no + /// process identity, returning `None` once the launch provably has no running process. + /// + /// Without a claim on the OpenVMM log, an absent endpoint proves nothing. With one, OpenVMM + /// shares the claim, so a held claim means that OpenVMM has yet to open its endpoint or to + /// exit; either is awaited for up to the start timeout. + fn launched_server( + &self, + sandbox_id: &SandboxId, + launch: &LaunchRecord, + ) -> Result> { + let interrupted = |detail: &str| { + Error::backend_error(format!( + "an interrupted start of sandbox {sandbox_id} {detail}" + )) + }; + let deadline = Instant::now() + self.config.start_timeout; + loop { + let pid = platform::endpoint_server_pid(&launch.endpoint) + .map_err(|error| interrupted("cannot be recovered yet").with_source(error))?; + if pid.is_some() { + return Ok(pid); + } + if !launch.log_claimed { + return Err(interrupted( + "has no recorded process identity or observable endpoint; cannot verify that \ + OpenVMM exited, so the launch marker is retained", + )); + } + let log = self.store.log_path(sandbox_id); + if platform::launch_log_released(&log) + .map_err(|error| interrupted("cannot be recovered yet").with_source(error))? + { + return Ok(None); + } + if Instant::now() >= deadline { + return Err(interrupted(&format!( + "has no recorded process identity or observable endpoint, and its OpenVMM log \ + {} is missing or still held; the launch marker is retained", + log.display() + ))); + } + thread::sleep(Duration::from_millis(50)); + } + } + fn session_error(&self, sandbox_id: &SandboxId, error: SessionError) -> Error { match error { SessionError::ProcessExited => { @@ -529,6 +579,7 @@ impl Backend for OpenVmmBackend { backend: BACKEND_KEY.to_owned(), network: request.network.clone(), filesystem, + native: None, memory_mib, workload_uid, workload_gid, @@ -582,6 +633,7 @@ impl Backend for OpenVmmBackend { format: STATE_FORMAT, endpoint: endpoint.clone(), process: None, + log_claimed: false, }, )?; let started = Instant::now(); @@ -627,6 +679,7 @@ impl Backend for OpenVmmBackend { format: STATE_FORMAT, endpoint: endpoint.clone(), process: Some(ProcessIdentity { pid, start_time }), + log_claimed: false, }; if let Err(error) = self .store diff --git a/aci_edge_sandboxes/src/openvmm/native.rs b/aci_edge_sandboxes/src/openvmm/native.rs new file mode 100644 index 000000000..b01c0c40a --- /dev/null +++ b/aci_edge_sandboxes/src/openvmm/native.rs @@ -0,0 +1,1695 @@ +//! Image-backed guest lifecycle over the separately supplied nvxhost library. + +use std::collections::HashMap; +use std::fs::File; +use std::io::{self, Write}; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, Mutex}; +use std::thread; +use std::time::{Duration, Instant}; + +use prost::Message; + +use super::artifacts::absolute; +use super::config::{OpenVmmConfig, validate_unix_socket_path}; +use super::images::{ + ImageDigest, ImageId, ImageRecord, ImageStore, RegisteredImage, VerifiedFile, encode_hex, +}; +use super::platform::{self, Transport}; +use super::process; +use super::state::{ + BOOT_SOCKET_NAME, IMAGES_NAME, LaunchRecord, NATIVE_BACKEND_KEY, NativeArtifactRecord, + ProcessIdentity, RuntimeRecord, SOCKET_NAME, STATE_FORMAT, SandboxRecord, StateStore, + remove_if_present, +}; +use super::{OpenVmmBackend, RunState}; +use crate::backend::{Backend, ExecControl, ExecIo}; +use crate::capabilities::Capabilities; +use crate::error::{Error, ErrorCode, Result}; +use crate::exec::{Completion, ExecOutcome}; +use crate::id::SandboxId; +use crate::model::{ + Access, Command, DeprovisionResult, ExecRequest, Metadata, ProvisionRequest, ProvisionResult, + StartResult, StdinMode, StopResult, +}; +use crate::nvxhost::{HostLibrary, LaunchInputs, Session}; + +const BACKEND_KEY: &str = NATIVE_BACKEND_KEY; +const RUNTIME_ABI: &str = "microvm-abi-v2-edge-ramfs-v1"; +const MAX_EXEC_SECONDS: u64 = 3600; +const MAX_OUTPUT_BYTES: usize = 1 << 20; +const SHELL: &str = "/bin/sh"; +const CONSOLE_POLL: Duration = Duration::from_millis(250); + +/// Artifact paths and approved native-library digest for an image-backed guest. +/// +/// The image is a caller-prepared GPT disk, not a container image reference. This backend +/// creates no disks, filesystem mappings, or network devices. Construct it with +/// [`NvxHostConfig::new`] and its builder methods, which keep callers compatible as options +/// are added. +#[derive(Debug, Clone)] +#[non_exhaustive] +pub struct NvxHostConfig { + /// OpenVMM and guest boot artifacts, hypervisor, state root, and deadlines. + pub openvmm: OpenVmmConfig, + /// GPT image attached read-only as the distro block device. + /// + /// The backend registers the image when it is created, and the sandboxes that it provisions + /// refer to the image by its [`ImageId`]. + pub image: PathBuf, + /// How the backend establishes the content digest of `image` when it registers it. + pub image_digest: ImageDigest, + /// Absolute path to a separately installed nvxhost DLL or shared library. + pub library: PathBuf, + /// SHA-256 approved by the caller's independent artifact policy. + pub library_sha256: [u8; 32], + /// SHA-256 digests of the OpenVMM runtime files approved by the caller's policy, if any. + /// + /// The backend hashes the runtime files once, when it is created, and requires these digests + /// when they are set. + pub runtime_sha256: Option, + /// Hashes the image and the runtime files again before every start. + /// + /// Otherwise a start compares only the files' seals with those taken when they were hashed. + /// This diagnostic reads every file in full, which adds the hashing time to each start. + pub content_verification: bool, + /// Requests verbose guest kernel diagnostics on the boot console. + pub guest_debug: bool, +} + +impl NvxHostConfig { + /// Uses caller-provided artifacts and an independently supplied native-library digest. + pub fn new( + openvmm: OpenVmmConfig, + image: impl Into, + library: impl Into, + library_sha256: [u8; 32], + ) -> Self { + Self { + openvmm, + image: image.into(), + image_digest: ImageDigest::Compute, + library: library.into(), + library_sha256, + runtime_sha256: None, + content_verification: false, + guest_debug: false, + } + } + + /// Sets how the backend establishes the digest of the image that it registers. + #[must_use] + pub fn with_image_digest(mut self, digest: ImageDigest) -> Self { + self.image_digest = digest; + self + } + + /// Requires the OpenVMM runtime files to have these approved digests. + #[must_use] + pub fn with_runtime_digests(mut self, digests: RuntimeDigests) -> Self { + self.runtime_sha256 = Some(digests); + self + } + + /// Hashes the image and the runtime files again before every start, as a diagnostic. + #[must_use] + pub fn with_content_verification(mut self, enabled: bool) -> Self { + self.content_verification = enabled; + self + } + + /// Enables verbose guest boot diagnostics without changing the guest lifecycle. + #[must_use] + pub fn with_guest_debug(mut self, enabled: bool) -> Self { + self.guest_debug = enabled; + self + } +} + +/// SHA-256 digests of the OpenVMM executable, guest kernel, and guest initramfs. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RuntimeDigests { + /// Digest of the OpenVMM executable. + pub openvmm: [u8; 32], + /// Digest of the guest kernel. + pub kernel: [u8; 32], + /// Digest of the guest initramfs. + pub initrd: [u8; 32], +} + +/// Owns sandbox state and OpenVMM processes; the private native library owns only +/// the launch arguments and authenticated guest channel. +#[derive(Debug)] +pub struct NvxHostBackend { + config: NvxHostConfig, + base: OpenVmmBackend, + host: Arc, + images: ImageStore, + image: ImageId, + runtime: RuntimeFiles, + console_pumps: Mutex>, +} + +/// The OpenVMM runtime files, hashed once, when the backend is created. +#[derive(Debug)] +struct RuntimeFiles { + openvmm: VerifiedFile, + kernel: VerifiedFile, + initrd: VerifiedFile, +} + +impl RuntimeFiles { + fn verify(config: &OpenVmmConfig, approved: Option) -> Result { + Ok(Self { + openvmm: VerifiedFile::verify( + &config.openvmm, + "OpenVMM executable", + approved.map(|digests| digests.openvmm), + )?, + kernel: VerifiedFile::verify( + &config.kernel, + "guest kernel", + approved.map(|digests| digests.kernel), + )?, + initrd: VerifiedFile::verify( + &config.initrd, + "guest initramfs", + approved.map(|digests| digests.initrd), + )?, + }) + } + + fn digests(&self) -> RuntimeDigests { + RuntimeDigests { + openvmm: *self.openvmm.sha256(), + kernel: *self.kernel.sha256(), + initrd: *self.initrd.sha256(), + } + } + + /// Opens every file, keeping writers out on Windows while the handles live, and checks that + /// none changed since it was hashed. With `rehash`, it also hashes every file again. + fn open_checked(&self, rehash: bool) -> Result> { + [&self.openvmm, &self.kernel, &self.initrd] + .into_iter() + .map(|verified| { + let mut file = verified.open_checked()?; + if rehash { + verified.verify_content(&mut file)?; + } + Ok(file) + }) + .collect() + } +} + +/// Copies the boot console of one OpenVMM launch into the sandbox's console log. +#[derive(Debug)] +struct ConsolePump { + pid: u32, + start_time: u64, + thread: thread::JoinHandle>, +} + +#[derive(Clone, PartialEq, Message)] +struct GuestInfo { + #[prost(int32, tag = "10")] + readiness: i32, + #[prost(uint64, tag = "11")] + readiness_generation: u64, + #[prost(string, tag = "15")] + agent_build_id: String, + #[prost(string, tag = "16")] + runtime_abi: String, + #[prost(string, tag = "17")] + startup_error: String, +} + +#[derive(Clone, PartialEq, Message)] +struct WaitReadyRequest { + #[prost(uint64, tag = "1")] + minimum_generation: u64, +} + +#[derive(Clone, PartialEq, Message)] +struct WaitReadyResponse { + #[prost(int32, tag = "1")] + readiness: i32, + #[prost(uint64, tag = "2")] + generation: u64, + #[prost(string, tag = "3")] + startup_error: String, +} + +#[derive(Clone, PartialEq, Message)] +struct ExecuteCommandRequest { + #[prost(string, tag = "1")] + command: String, + #[prost(string, repeated, tag = "2")] + args: Vec, + #[prost(int32, tag = "5")] + timeout_seconds: i32, +} + +#[derive(Clone, PartialEq, Message)] +struct ExecuteCommandResponse { + #[prost(int32, tag = "1")] + exit_code: i32, + #[prost(string, tag = "2")] + stdout: String, + #[prost(string, tag = "3")] + stderr: String, + #[prost(bool, tag = "4")] + timed_out: bool, +} + +#[derive(Clone, PartialEq, Message)] +struct ShutdownRequest { + #[prost(int64, tag = "1")] + grace_period_milliseconds: i64, +} + +#[derive(Clone, PartialEq, Message)] +struct ShutdownResponse {} + +#[derive(Clone, PartialEq, Message)] +struct StreamLogsRequest { + #[prost(string, tag = "1")] + container_id: String, + #[prost(uint64, tag = "2")] + offset: u64, + #[prost(bool, tag = "3")] + follow: bool, + #[prost(uint32, tag = "4")] + max_chunk_bytes: u32, +} + +#[derive(Clone, PartialEq, Message)] +struct LogChunk { + #[prost(string, tag = "1")] + container_id: String, + #[prost(uint64, tag = "2")] + offset: u64, + #[prost(bytes, tag = "3")] + data: Vec, + #[prost(uint64, tag = "4")] + next_offset: u64, + #[prost(uint64, tag = "5")] + earliest_retained_offset: u64, + #[prost(bool, tag = "6")] + loss: bool, +} + +struct ExecJob { + host: Arc, + runtime: RuntimeRecord, + capability: [u8; 32], + config: OpenVmmConfig, + command: ExecuteCommandRequest, + timeout: Option, + io: ExecIo, +} + +impl NvxHostBackend { + /// Opens the state root, verifies the explicitly supplied native asset, hashes the OpenVMM + /// runtime files, and registers the configured image. + /// + /// Creating a backend reads every runtime file in full. It reads the image only if no + /// registration that the file still matches exists and `image_digest` is not + /// [`ImageDigest::Trusted`]. + pub fn new(mut config: NvxHostConfig) -> Result { + config.openvmm = config.openvmm.normalized()?; + config.image = absolute(&config.image)?; + config.library = absolute(&config.library)?; + let store_root = config.openvmm.state_root.join(BACKEND_KEY); + validate_unix_socket_path(&store_root, SOCKET_NAME, "control")?; + validate_unix_socket_path(&store_root, BOOT_SOCKET_NAME, "boot-console")?; + let host = HostLibrary::load(&config.library, &config.library_sha256)?; + let store = StateStore::open_for(&store_root, BACKEND_KEY).map_err(|error| { + Error::backend_unavailable(format!( + "cannot open native sandbox state root {}", + store_root.display() + )) + .with_source(error) + })?; + let images_dir = store_root.join(IMAGES_NAME); + let images = ImageStore::open(&images_dir).map_err(|error| { + Error::backend_unavailable(format!( + "cannot open the image registry {}", + images_dir.display() + )) + .with_source(error) + })?; + let runtime = RuntimeFiles::verify(&config.openvmm, config.runtime_sha256)?; + let image = images.register(&config.image, config.image_digest)?.id; + let base = OpenVmmBackend { + config: config.openvmm.clone(), + store, + }; + Ok(Self { + config, + base, + host, + images, + image, + runtime, + console_pumps: Mutex::new(HashMap::new()), + }) + } + + /// Returns the normalized configuration. + pub fn config(&self) -> &NvxHostConfig { + &self.config + } + + /// Returns the OpenVMM diagnostic log for this sandbox. + pub fn log_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.base.log_path(sandbox_id) + } + + /// Returns the guest boot-console log, including startup errors before ttrpc is ready. + pub fn console_log_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.base.store.console_path(sandbox_id) + } + + /// Returns OpenVMM's bounded outcome report path for the latest launch. + pub fn outcome_report_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.base.store.outcome_path(sandbox_id) + } + + /// Returns the ID of the configured image, which the sandboxes that this backend provisions + /// boot. + pub fn image_id(&self) -> ImageId { + self.image + } + + /// Returns the digests of the OpenVMM runtime files, computed when the backend was created. + pub fn runtime_digests(&self) -> RuntimeDigests { + self.runtime.digests() + } + + /// Registers the image at `path` in the state root's image registry. + /// + /// A registration that the file still matches is reused without reading the file. Otherwise + /// the image is hashed, or recorded with an [`ImageDigest::Trusted`] digest, and its + /// registration replaces any earlier one of the same content. Registering a changed image + /// again lets the sandboxes that boot it start again if its content is unchanged. + pub fn register_image( + &self, + path: impl AsRef, + digest: ImageDigest, + ) -> Result { + let path = absolute(path.as_ref())?; + Ok(self.images.register(&path, digest)?.describe(true)) + } + + /// Lists the registered images and whether each file still matches its registration. + pub fn images(&self) -> Result> { + Ok(self + .images + .list()? + .into_iter() + .map(|record| { + let intact = self.images.open_checked(&record, false).is_ok(); + record.describe(intact) + }) + .collect()) + } + + /// Hashes a registered image and checks that it still has the content that its ID names. + /// + /// This diagnostic reads the whole image, whereas provisioning and starting compare only the + /// image's seal with its registration. + pub fn verify_image(&self, id: &ImageId) -> Result<()> { + let record = self + .images + .get(id)? + .ok_or_else(|| Error::backend_unavailable(format!("image {id} is not registered")))?; + self.images.verify(&record) + } + + /// Removes the registration of an image, returning whether it existed. The file stays. + /// + /// Fails with [`ErrorCode::PolicyValidation`] while a provisioned sandbox refers to the image. + /// Removing the registration of the configured image makes provisioning fail until the image + /// is registered again. + pub fn unregister_image(&self, id: &ImageId) -> Result { + let _registry = self.images.lock()?; + let image = id.to_string(); + for sandbox_id in self.base.store.sandbox_ids()? { + let record = match self.base.store.load(&sandbox_id) { + Ok(record) => record, + // The sandbox was deprovisioned after the listing. + Err(error) if error.code() == ErrorCode::StaleId => continue, + Err(error) => return Err(error), + }; + if record.native.is_some_and(|native| native.image == image) { + return Err(Error::policy_validation(format!( + "image {id} is used by sandbox {sandbox_id}; deprovision the sandbox first" + ))); + } + } + self.images.remove(id) + } + + /// Collects the guest's bounded, non-follow log snapshot through ttrpc. + pub fn guest_logs(&self, sandbox_id: &SandboxId) -> Result> { + let (runtime, capability) = self.running(sandbox_id)?; + let deadline = Instant::now() + self.config.openvmm.control_timeout; + let mut session = self.connect( + &runtime, + &capability, + deadline.saturating_duration_since(Instant::now()), + )?; + let mut output = Vec::new(); + let mut offset = 0u64; + let request = StreamLogsRequest { + container_id: "guest".to_owned(), + offset, + follow: false, + max_chunk_bytes: 64 * 1024, + }; + let streamed = session.server_stream( + "StreamLogs", + &request.encode_to_vec(), + deadline.saturating_duration_since(Instant::now()), + |bytes| { + let chunk = LogChunk::decode(bytes).map_err(|error| { + Error::backend_error("the guest returned an invalid log chunk") + .with_source(error) + })?; + let expected_next = offset + .checked_add(chunk.data.len() as u64) + .ok_or_else(|| Error::backend_error("guest log cursor overflowed"))?; + if chunk.container_id != "guest" + || chunk.loss + || chunk.offset != offset + || chunk.next_offset != expected_next + || chunk.earliest_retained_offset > offset + { + return Err(Error::backend_error( + "the guest returned a missing or out-of-order log chunk", + )); + } + if output + .len() + .checked_add(chunk.data.len()) + .is_none_or(|size| size > MAX_OUTPUT_BYTES) + { + return Err(Error::backend_error( + "guest logs exceed the one-megabyte collection limit", + )); + } + offset = chunk.next_offset; + output.extend_from_slice(&chunk.data); + Ok(()) + }, + ); + finish_session(session, streamed, Some(deadline))?; + Ok(output) + } + + fn boot_endpoint(&self, sandbox_id: &SandboxId, control: &str) -> Result { + if cfg!(windows) { + let suffix = control + .strip_prefix("//./pipe/openvmm-microvm-") + .ok_or_else(|| { + Error::backend_error("the recorded control pipe is not canonical") + })?; + if suffix.is_empty() + || !suffix + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'.')) + { + return Err(Error::backend_error( + "the recorded control pipe has an invalid name", + )); + } + return Ok(format!("{control}-boot")); + } + self.base + .store + .boot_socket_path(sandbox_id) + .to_str() + .map(str::to_owned) + .ok_or_else(|| Error::backend_unavailable("the boot-console socket path is not UTF-8")) + } + + fn start_console_pump(&self, sandbox_id: &SandboxId, runtime: &RuntimeRecord) -> Result<()> { + let mut pumps = self + .console_pumps + .lock() + .map_err(|_| Error::backend_error("the boot-console pump table is unavailable"))?; + if let Some(pump) = pumps.get(sandbox_id) + && pump.pid == runtime.pid + && pump.start_time == runtime.start_time + { + return Ok(()); + } + // A listener of an earlier launch ends promptly because its OpenVMM has exited; its + // result only describes that launch. + if let Some(stale) = pumps.remove(sandbox_id) { + let _ = stale.thread.join(); + } + let endpoint = self.boot_endpoint(sandbox_id, &runtime.endpoint)?; + let file = self.base.store.open_console_log(sandbox_id)?; + let runtime = runtime.clone(); + let (pid, start_time) = (runtime.pid, runtime.start_time); + let timeout = self.config.openvmm.start_timeout; + let thread = thread::Builder::new() + .name("nvxhost-boot-console".to_owned()) + .spawn(move || pump_console(runtime, endpoint, file, timeout)) + .map_err(|error| { + Error::backend_error("cannot start the guest boot-console listener") + .with_source(error) + })?; + pumps.insert( + sandbox_id.clone(), + ConsolePump { + pid, + start_time, + thread, + }, + ); + Ok(()) + } + + fn finish_console_pump(&self, sandbox_id: &SandboxId) -> Result<()> { + let pump = self + .console_pumps + .lock() + .map_err(|_| Error::backend_error("the boot-console pump table is unavailable"))? + .remove(sandbox_id); + if let Some(pump) = pump { + pump.thread + .join() + .map_err(|_| Error::backend_error("the boot-console listener panicked"))??; + } + Ok(()) + } + + fn artifact_record(&self) -> NativeArtifactRecord { + let digests = self.runtime.digests(); + NativeArtifactRecord { + image: self.image.to_string(), + openvmm_sha256: encode_hex(&digests.openvmm), + kernel_sha256: encode_hex(&digests.kernel), + initrd_sha256: encode_hex(&digests.initrd), + library_sha256: encode_hex(&self.config.library_sha256), + } + } + + /// Returns a sandbox's artifact identity after checking that it uses this backend's host + /// library. + fn native_record<'a>(&self, record: &'a SandboxRecord) -> Result<&'a NativeArtifactRecord> { + let recorded = record.native.as_ref().ok_or_else(|| { + Error::backend_error("the native sandbox has no pinned artifact identity") + })?; + if recorded.library_sha256 != encode_hex(&self.config.library_sha256) { + return Err(Error::backend_unavailable( + "the native sandbox was provisioned with a different host library", + )); + } + Ok(recorded) + } + + /// Checks the configured image and the runtime files against their seals. + fn check_configured_artifacts(&self) -> Result<()> { + let image = self.images.get(&self.image)?.ok_or_else(|| { + Error::backend_unavailable(format!( + "image {} is no longer registered; register it again", + self.image + )) + })?; + self.images.open_checked(&image, false)?; + self.runtime.open_checked(false)?; + Ok(()) + } + + /// Checks a sandbox's artifacts before it launches, without hashing them unless content + /// verification is enabled, and returns its image registration. + /// + /// The returned handles keep writers out of the files on Windows until they are dropped, after + /// OpenVMM has opened the files. + fn launch_artifacts(&self, record: &SandboxRecord) -> Result<(ImageRecord, Vec)> { + let recorded = self.native_record(record)?; + let digests = self.runtime.digests(); + if recorded.openvmm_sha256 != encode_hex(&digests.openvmm) + || recorded.kernel_sha256 != encode_hex(&digests.kernel) + || recorded.initrd_sha256 != encode_hex(&digests.initrd) + { + return Err(Error::backend_unavailable( + "the native sandbox was provisioned with different OpenVMM runtime files", + )); + } + let id = ImageId::parse(&recorded.image).map_err(|_| { + Error::backend_error(format!( + "the native sandbox records a malformed image ID {:?}", + recorded.image + )) + })?; + let image = self.images.get(&id)?.ok_or_else(|| { + Error::backend_unavailable(format!( + "image {id} of the native sandbox is not registered; register it again" + )) + })?; + let rehash = self.config.content_verification; + let mut held = self.runtime.open_checked(rehash)?; + let mut file = self.images.open_checked(&image, true)?; + if rehash { + image.verify_content(&mut file)?; + } + held.push(file); + Ok((image, held)) + } + + fn running(&self, sandbox_id: &SandboxId) -> Result<(RuntimeRecord, [u8; 32])> { + let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; + self.native_record(&record)?; + match self.base.reconcile(sandbox_id)? { + RunState::Provisioned => Err(Error::not_started(format!( + "sandbox {sandbox_id} is not running" + ))), + RunState::Running(runtime) => { + let capability = self.base.store.read_capability(sandbox_id)?; + self.start_console_pump(sandbox_id, &runtime)?; + Ok((runtime, capability)) + } + } + } + + fn connect( + &self, + runtime: &RuntimeRecord, + capability: &[u8; 32], + timeout: Duration, + ) -> Result { + let deadline = Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("control timeout is too long"))?; + Session::connect_verified( + Arc::clone(&self.host), + &runtime.endpoint, + capability, + runtime.pid, + deadline, + || { + platform::process_start_time(runtime.pid) + .map(|current| current == Some(runtime.start_time)) + .map_err(|error| { + Error::backend_error("cannot verify the running OpenVMM process") + .with_source(error) + }) + }, + ) + } + + fn wait_ready(&self, session: &mut Session, deadline: Instant) -> Result { + let remaining = || { + let timeout = deadline.saturating_duration_since(Instant::now()); + if timeout.is_zero() { + Err(Error::backend_error( + "the guest did not become ready before the start deadline", + )) + } else { + Ok(timeout) + } + }; + let info = GuestInfo::decode( + session + .unary("GetGuestInfo", &[], Some(remaining()?))? + .as_slice(), + ) + .map_err(|error| { + Error::backend_error("the guest returned malformed identity data").with_source(error) + })?; + if info.runtime_abi != RUNTIME_ABI || info.agent_build_id.is_empty() { + return Err(Error::backend_unavailable(format!( + "the guest reported runtime ABI {:?}, expected {RUNTIME_ABI}", + info.runtime_abi + ))); + } + if info.readiness == 3 || info.readiness == 2 { + return Err(Error::backend_error(format!( + "the guest cannot become ready: {}", + info.startup_error + ))); + } + let ready = WaitReadyResponse::decode( + session + .unary( + "WaitReady", + &WaitReadyRequest { + minimum_generation: info.readiness_generation, + } + .encode_to_vec(), + Some(remaining()?), + )? + .as_slice(), + ) + .map_err(|error| { + Error::backend_error("the guest returned malformed readiness data").with_source(error) + })?; + if ready.readiness != 1 || ready.generation < info.readiness_generation { + return Err(Error::backend_error(format!( + "the guest did not reach the required readiness generation: {}", + ready.startup_error + ))); + } + Ok(info.agent_build_id) + } +} + +impl Backend for NvxHostBackend { + fn name(&self) -> &str { + BACKEND_KEY + } + + fn capabilities(&self) -> Capabilities { + capabilities() + } + + fn probe(&self) -> Result<()> { + self.base.probe()?; + self.check_configured_artifacts() + } + + fn validate_provision(&self, request: &ProvisionRequest) -> Result<()> { + let memory = request + .microvm + .provision + .memory_mib + .unwrap_or(self.config.openvmm.memory_mib); + if memory == 0 || memory > i32::MAX as u32 { + return Err(Error::policy_validation( + "microvm.provision.memoryMib must fit a positive signed 32-bit integer", + )); + } + if request + .filesystem + .as_ref() + .is_some_and(|policy| !policy.is_empty()) + || request.network.as_ref().is_some_and(|policy| { + policy.egress.default != Access::Deny + || policy.ingress.default != Access::Deny + || policy.ingress.host_loopback == Some(Access::Allow) + || !policy.egress.allow.is_empty() + || !policy.egress.deny.is_empty() + }) + { + return Err(Error::policy_validation( + "the nvxhost backend currently supports no host filesystem or guest network", + )); + } + Ok(()) + } + + fn validate_exec(&self, request: &ExecRequest) -> Result<()> { + prepare_exec(request).map(|_| ()) + } + + fn provision(&self, request: &ProvisionRequest) -> Result { + self.validate_provision(request)?; + self.base.probe()?; + // Unregistering an image waits for this lock, so the image stays registered until the + // sandbox that refers to it is recorded. + let _registry = self.images.lock()?; + self.check_configured_artifacts()?; + let record = SandboxRecord { + format: STATE_FORMAT, + backend: BACKEND_KEY.to_owned(), + network: None, + filesystem: None, + native: Some(self.artifact_record()), + memory_mib: request + .microvm + .provision + .memory_mib + .unwrap_or(self.config.openvmm.memory_mib), + workload_uid: 65534, + workload_gid: 65534, + create_workload_account: false, + hostname: self.config.openvmm.hostname.clone(), + }; + let sandbox_id = SandboxId::generate()?; + self.base.store.create(&sandbox_id, &record)?; + Ok(ProvisionResult { + sandbox_id, + metadata: None, + }) + } + + fn start(&self, sandbox_id: &SandboxId) -> Result { + let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; + if let RunState::Running(_) = self.base.reconcile(sandbox_id)? { + return Err(Error::already_started(format!( + "sandbox {sandbox_id} is already running" + ))); + } + // A listener of a launch that exited on its own ends promptly. Retire it now so that the + // new launch's listener connects before the guest writes its first boot output. + let _ = self.finish_console_pump(sandbox_id); + // `_held` keeps writers out of the checked files on Windows until OpenVMM has opened them. + let (image, _held) = self.launch_artifacts(&record)?; + self.base.probe()?; + + let mut capability = [0u8; 32]; + while capability == [0; 32] { + getrandom::fill(&mut capability).map_err(|error| { + Error::backend_error("cannot generate a broker capability").with_source(error) + })?; + } + let socket = self.base.store.socket_path(sandbox_id); + let boot_socket = self.base.store.boot_socket_path(sandbox_id); + remove_if_present(&socket)?; + remove_if_present(&boot_socket)?; + let endpoint = platform::control_endpoint(&socket).map_err(|error| { + Error::backend_error("cannot choose a control endpoint").with_source(error) + })?; + let boot = self.boot_endpoint(sandbox_id, &endpoint)?; + let mut arguments = self.host.launch_arguments(&LaunchInputs { + kernel: &self.config.openvmm.kernel, + initrd: &self.config.openvmm.initrd, + image: &image.path, + control: &endpoint, + boot: &boot, + hypervisor: self.config.openvmm.hypervisor.as_str(), + memory_mb: record.memory_mib, + guest_debug: self.config.guest_debug, + })?; + let report = self.base.store.outcome_path(sandbox_id); + remove_if_present(&report)?; + arguments.push("--microvm-report".into()); + arguments.push(report.into_os_string()); + + self.base.store.write_capability(sandbox_id, &capability)?; + let log = self.base.store.create_log(sandbox_id)?; + // OpenVMM inherits the log and this claim, so if this process dies before it records + // OpenVMM's identity, recovery can still tell whether anything of the launch runs. + platform::claim_launch_log(&log).map_err(|error| { + Error::backend_error("cannot claim the OpenVMM log").with_source(error) + })?; + self.base.store.write_launch( + sandbox_id, + &LaunchRecord { + format: STATE_FORMAT, + endpoint: endpoint.clone(), + process: None, + log_claimed: true, + }, + )?; + let started = Instant::now(); + let child = match process::spawn( + &self.config.openvmm, + &arguments, + &capability, + log, + &self.base.store.dir(sandbox_id), + ) { + Ok(child) => child, + Err(error) => { + self.base.store.clear_runtime(sandbox_id)?; + return Err(Error::backend_error("cannot launch OpenVMM").with_source(error)); + } + }; + let pid = child.id(); + let start_time = match platform::process_start_time(pid) { + Ok(Some(value)) => value, + other => { + if process::kill_child(child) { + self.base.store.clear_runtime(sandbox_id)?; + } + return Err(self.base.start_failure( + sandbox_id, + &format!("cannot identify OpenVMM process {pid}: {other:?}"), + )); + } + }; + let runtime = RuntimeRecord { + format: STATE_FORMAT, + pid, + start_time, + endpoint: endpoint.clone(), + }; + if let Err(error) = self + .base + .store + .write_launch( + sandbox_id, + &LaunchRecord { + format: STATE_FORMAT, + endpoint, + process: Some(ProcessIdentity { pid, start_time }), + log_claimed: true, + }, + ) + .and_then(|()| self.base.store.write_runtime(sandbox_id, &runtime)) + { + if process::kill_child(child) { + self.base.store.clear_runtime(sandbox_id)?; + } + return Err(error); + } + process::detach_child(child); + self.base.store.remove_launch(sandbox_id); + if let Err(error) = self.start_console_pump(sandbox_id, &runtime) { + return Err(self + .base + .abort_start(sandbox_id, &runtime, &error.to_string())); + } + let ready = (|| { + let deadline = started + .checked_add(self.config.openvmm.start_timeout) + .ok_or_else(|| Error::backend_error("start timeout is too long"))?; + let mut session = self.connect( + &runtime, + &capability, + deadline.saturating_duration_since(Instant::now()), + )?; + let result = self.wait_ready(&mut session, deadline); + finish_session(session, result, Some(deadline)) + })(); + let agent = match ready { + Ok(agent) => agent, + Err(error) => { + let failure = self + .base + .abort_start(sandbox_id, &runtime, &error.to_string()); + match platform::process_start_time(runtime.pid) { + Ok(Some(current)) if current == runtime.start_time => return Err(failure), + Ok(_) => {} + Err(check) => { + return Err(Error::backend_error(format!( + "{failure}; cannot verify boot-console cleanup: {check}" + ))); + } + } + return match self.finish_console_pump(sandbox_id) { + Ok(()) => Err(failure), + Err(console) => Err(Error::backend_error(format!( + "{failure}; boot-console listener also failed: {console}" + ))), + }; + } + }; + let mut metadata = Metadata::new(); + let boot_ms = u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX); + metadata.insert("bootMilliseconds".to_owned(), boot_ms.into()); + metadata.insert("guestBuildId".to_owned(), agent.into()); + Ok(StartResult { + metadata: Some(metadata), + }) + } + + fn exec( + &self, + sandbox_id: &SandboxId, + request: &ExecRequest, + io: ExecIo, + ) -> Result> { + if request.stdin == StdinMode::Piped || io.stdin.is_some() { + return Err(Error::policy_validation( + "the nvxhost backend does not support piped stdin", + )); + } + let (message, timeout) = prepare_exec(request)?; + let (runtime, capability) = self.running(sandbox_id)?; + let shared = Arc::new(Completion::default()); + let result = Arc::clone(&shared); + let job = ExecJob { + host: Arc::clone(&self.host), + runtime, + capability, + config: self.config.openvmm.clone(), + command: message, + timeout, + io, + }; + thread::Builder::new() + .name("nvxhost-exec".to_owned()) + .spawn(move || { + let outcome = execute(job); + result.finish(outcome); + }) + .map_err(|error| { + Error::backend_error("cannot start an execution thread").with_source(error) + })?; + Ok(Box::new(NvxHostExecution { shared })) + } + + fn stop(&self, sandbox_id: &SandboxId) -> Result { + let (_guard, record) = self.base.store.lock_and_load(sandbox_id)?; + self.native_record(&record)?; + let RunState::Running(runtime) = self.base.reconcile(sandbox_id)? else { + return Err(Error::already_stopped(format!( + "sandbox {sandbox_id} is not running" + ))); + }; + let deadline = Instant::now() + self.config.openvmm.stop_timeout; + let graceful = (|| { + let capability = self.base.store.read_capability(sandbox_id)?; + let remaining = deadline.saturating_duration_since(Instant::now()); + let mut session = self.connect(&runtime, &capability, remaining)?; + let grace_period_milliseconds = i64::try_from(remaining.as_millis()) + .map_err(|_| Error::backend_error("guest shutdown grace period is too long"))?; + let response = session.unary( + "Shutdown", + &ShutdownRequest { + grace_period_milliseconds, + } + .encode_to_vec(), + Some(remaining), + ); + let response = finish_session(session, response, Some(deadline))?; + ShutdownResponse::decode(response.as_slice()).map_err(|error| { + Error::backend_error("the guest returned an invalid shutdown response") + .with_source(error) + })?; + if process::wait_for_exit(runtime.pid, runtime.start_time, deadline) { + Ok(()) + } else { + Err(Error::backend_error("guest shutdown timed out")) + } + })(); + let needs_force = graceful.is_err(); + let forced = needs_force + && platform::process_start_time(runtime.pid).map_err(|error| { + Error::backend_error("cannot check OpenVMM before forced stop").with_source(error) + })? == Some(runtime.start_time); + if forced + && !process::kill(runtime.pid, runtime.start_time).map_err(|error| { + Error::backend_error(format!("cannot terminate OpenVMM process {}", runtime.pid)) + .with_source(error) + })? + { + return Err(Error::backend_error(format!( + "OpenVMM process {} did not terminate", + runtime.pid + ))); + } + let mut metadata = Metadata::new(); + metadata.insert("forced".to_owned(), forced.into()); + if let Err(error) = graceful { + metadata.insert("gracefulError".to_owned(), error.to_string().into()); + } + // The guest has stopped; a boot-console capture failure is only a diagnostic. + if let Err(error) = self.finish_console_pump(sandbox_id) { + metadata.insert("consoleError".to_owned(), error.to_string().into()); + } + self.base.store.clear_runtime(sandbox_id)?; + Ok(StopResult { + metadata: Some(metadata), + }) + } + + fn deprovision(&self, sandbox_id: &SandboxId) -> Result { + let (guard, record) = self.base.store.lock_and_load(sandbox_id)?; + self.native_record(&record)?; + if let RunState::Running(_) = self.base.reconcile(sandbox_id)? { + return Err(Error::already_started(format!( + "sandbox {sandbox_id} is running; stop it before deprovisioning" + ))); + } + // A listener remains only if the guest exited without a stop through this backend. + let console = self.finish_console_pump(sandbox_id); + self.base.store.remove(sandbox_id)?; + drop(guard); + self.base.store.remove_lock(sandbox_id); + Ok(DeprovisionResult { + metadata: console.err().map(|error| { + Metadata::from_iter([("consoleError".to_owned(), error.to_string().into())]) + }), + }) + } +} + +fn capabilities() -> Capabilities { + let mut capabilities = Capabilities::new(BACKEND_KEY); + capabilities.exec.command_line = true; + capabilities.exec.argv = true; + capabilities.exec.max_timeout_ms = Some(MAX_EXEC_SECONDS * 1_000); + capabilities.exec.max_output_bytes = Some(MAX_OUTPUT_BYTES as u64); + capabilities.network.egress_deny = true; + capabilities.network.ingress_deny = true; + capabilities.network.host_loopback_deny = true; + capabilities +} + +fn finish_session(session: Session, result: Result, deadline: Option) -> Result { + match (result, session.close(deadline)) { + (Ok(value), Ok(())) => Ok(value), + (Err(error), Ok(())) | (Ok(_), Err(error)) => Err(error), + (Err(error), Err(close)) => { + eprintln!("nvxhost could not close a failed guest session: {close}"); + Err(error) + } + } +} + +fn pump_console( + runtime: RuntimeRecord, + endpoint: String, + output: File, + timeout: Duration, +) -> Result<()> { + copy_console( + || platform::connect_endpoint(&endpoint, runtime.pid, CONSOLE_POLL), + |activity| { + let current = platform::process_start_time(runtime.pid).map_err(|error| { + Error::backend_error(format!( + "cannot verify OpenVMM while {activity} the console" + )) + .with_source(error) + })?; + Ok(current != Some(runtime.start_time)) + }, + output, + timeout, + ) +} + +/// Copies the guest boot console into `output` until OpenVMM exits. +/// +/// OpenVMM serves one console client at a time, so the listener of another backend instance or +/// process may hold the console. This listener then waits, past `timeout`, to take over, and +/// ends without error if OpenVMM exits first. +fn copy_console( + mut connect: impl FnMut() -> io::Result>, + exited: impl Fn(&str) -> Result, + mut output: File, + timeout: Duration, +) -> Result<()> { + let deadline = Instant::now() + .checked_add(timeout) + .ok_or_else(|| Error::backend_error("boot-console timeout is too long"))?; + let mut held_elsewhere = false; + let mut console = loop { + match connect() { + Ok(console) => break console, + Err(error) + if matches!( + error.kind(), + io::ErrorKind::NotFound + | io::ErrorKind::ConnectionRefused + | io::ErrorKind::WouldBlock + | io::ErrorKind::TimedOut + ) => + { + held_elsewhere |= error.kind() == io::ErrorKind::WouldBlock; + if exited("connecting")? { + if held_elsewhere { + return Ok(()); + } + return Err(Error::backend_error( + "OpenVMM exited before the guest boot console could connect", + )); + } + if !held_elsewhere && Instant::now() >= deadline { + return Err(Error::backend_error( + "OpenVMM did not open its guest boot-console endpoint", + )); + } + thread::sleep(Duration::from_millis(25)); + } + Err(error) => { + return Err( + Error::backend_error("cannot connect the guest boot console") + .with_source(error), + ); + } + } + }; + let mut buffer = [0u8; 4096]; + loop { + match console.read(&mut buffer, Some(CONSOLE_POLL)) { + Ok(0) => break, + Ok(size) => output.write_all(&buffer[..size]).map_err(|error| { + Error::backend_error("cannot write the guest boot-console log").with_source(error) + })?, + // A client still queued in a socket's listen backlog is reset, rather than reaching + // end of stream, when OpenVMM exits. + Err(error) => { + if exited("reading")? { + break; + } + if error.kind() != io::ErrorKind::TimedOut { + return Err(Error::backend_error("cannot read the guest boot console") + .with_source(error)); + } + } + } + } + output.sync_all().map_err(|error| { + Error::backend_error("cannot flush the guest boot-console log").with_source(error) + }) +} + +fn prepare_exec(request: &ExecRequest) -> Result<(ExecuteCommandRequest, Option)> { + if request.stdin == StdinMode::Piped + || request.process.cwd.is_some() + || request.process.env.is_some() + { + return Err(Error::policy_validation( + "the nvxhost backend supports no piped stdin, custom working directory, or environment", + )); + } + let argv = match &request.process.command { + Command::CommandLine(line) => vec![SHELL.to_owned(), "-c".to_owned(), line.clone()], + Command::Argv(argv) => argv.clone(), + }; + if argv.is_empty() + || argv.iter().any(|argument| argument.contains('\0')) + || argv.iter().map(String::len).sum::() > 4096 + { + return Err(Error::policy_validation( + "the guest command must contain non-NUL arguments totaling at most 4096 bytes", + )); + } + let timeout = request.process.timeout.filter(|timeout| !timeout.is_zero()); + if let Some(timeout) = timeout { + if timeout > Duration::from_secs(MAX_EXEC_SECONDS) { + return Err(Error::policy_validation( + "the guest command timeout exceeds one hour", + )); + } + if timeout.subsec_nanos() != 0 { + return Err(Error::policy_validation( + "the nvxhost backend requires whole-second precision for positive command timeouts", + )); + } + } + let seconds = timeout.map_or(0, |duration| duration.as_secs() as i32); + Ok(( + ExecuteCommandRequest { + command: argv[0].clone(), + args: argv[1..].to_vec(), + timeout_seconds: seconds, + }, + timeout, + )) +} + +fn execute(job: ExecJob) -> Result { + let ExecJob { + host, + runtime, + capability, + config, + command, + timeout, + mut io, + } = job; + let mut session = Session::connect_verified( + host, + &runtime.endpoint, + &capability, + runtime.pid, + Instant::now() + config.control_timeout, + || { + platform::process_start_time(runtime.pid) + .map(|current| current == Some(runtime.start_time)) + .map_err(|error| { + Error::backend_error("cannot verify the running OpenVMM process") + .with_source(error) + }) + }, + )?; + let deadline = timeout.map(|timeout| Instant::now() + timeout + config.exec_response_grace); + let response = session.unary( + "ExecuteCommand", + &command.encode_to_vec(), + deadline.map(|deadline| deadline.saturating_duration_since(Instant::now())), + ); + let response = finish_session(session, response, deadline)?; + let reply = ExecuteCommandResponse::decode(response.as_slice()).map_err(|error| { + Error::backend_error("the guest returned an invalid execution result").with_source(error) + })?; + let _ = io.stdout.write(reply.stdout.as_bytes()); + let _ = io.stderr.write(reply.stderr.as_bytes()); + drop(io); + if reply.timed_out { + Ok(ExecOutcome::TimedOut) + } else { + Ok(ExecOutcome::Exited(reply.exit_code)) + } +} + +struct NvxHostExecution { + shared: Arc, +} + +impl ExecControl for NvxHostExecution { + fn wait(&self) -> Result { + self.shared.wait() + } + + fn cancel(&self) -> Result<()> { + Err(Error::unsupported( + "the nvxhost backend does not support canceling an active guest command", + )) + } +} + +#[cfg(test)] +mod tests { + use std::cell::Cell; + use std::collections::VecDeque; + use std::io::{Read, Seek, SeekFrom}; + + use super::*; + use crate::ErrorCode; + + /// Replays scripted boot-console reads. + struct Replay(VecDeque>); + + impl Transport for Replay { + fn read(&mut self, buffer: &mut [u8], _timeout: Option) -> io::Result { + let chunk = self + .0 + .pop_front() + .expect("the console read past its script")?; + buffer[..chunk.len()].copy_from_slice(chunk); + Ok(chunk.len()) + } + + fn write_all(&mut self, _data: &[u8], _timeout: Option) -> io::Result<()> { + unreachable!("the console listener never writes") + } + } + + fn replay( + reads: [io::Result<&'static [u8]>; N], + ) -> io::Result> { + Ok(Box::new(Replay(reads.into()))) + } + + fn busy() -> io::Result> { + Err(io::ErrorKind::WouldBlock.into()) + } + + fn read_log(mut log: File) -> String { + let mut text = String::new(); + log.seek(SeekFrom::Start(0)).unwrap(); + log.read_to_string(&mut text).unwrap(); + text + } + + #[test] + fn a_console_held_elsewhere_is_awaited_past_the_deadline_until_openvmm_exits() { + let log = tempfile::tempfile().unwrap(); + let (attempts, checks) = (Cell::new(0), Cell::new(0)); + copy_console( + || { + attempts.set(attempts.get() + 1); + busy() + }, + |_| { + checks.set(checks.get() + 1); + Ok(checks.get() > 4) + }, + log.try_clone().unwrap(), + Duration::ZERO, + ) + .unwrap(); + assert_eq!(attempts.get(), 5); + assert_eq!(read_log(log), ""); + } + + #[test] + fn a_released_console_is_taken_over_and_copied_to_its_end() { + let log = tempfile::tempfile().unwrap(); + let mut attempts = VecDeque::from([ + busy(), + replay([ + Ok(b"late ".as_slice()), + Err(io::ErrorKind::TimedOut.into()), + Ok(b"output".as_slice()), + Ok(b"".as_slice()), + ]), + ]); + copy_console( + || attempts.pop_front().unwrap(), + |_| Ok(false), + log.try_clone().unwrap(), + Duration::ZERO, + ) + .unwrap(); + assert_eq!(read_log(log), "late output"); + } + + #[test] + fn a_console_that_never_opens_fails_at_its_deadline_or_when_openvmm_exits() { + for (exited, expected) in [ + (false, "did not open its guest boot-console endpoint"), + (true, "exited before the guest boot console could connect"), + ] { + let error = copy_console( + || Err(io::ErrorKind::NotFound.into()), + |_| Ok(exited), + tempfile::tempfile().unwrap(), + Duration::ZERO, + ) + .unwrap_err(); + assert!(error.message().contains(expected), "{error}"); + } + } + + #[test] + fn a_console_reset_ends_the_log_only_after_openvmm_exits() { + let log = tempfile::tempfile().unwrap(); + let reset = || { + replay([ + Ok(b"boot".as_slice()), + Err(io::ErrorKind::ConnectionReset.into()), + ]) + }; + copy_console( + reset, + |_| Ok(true), + log.try_clone().unwrap(), + Duration::ZERO, + ) + .unwrap(); + assert_eq!(read_log(log), "boot"); + + let error = copy_console( + reset, + |_| Ok(false), + tempfile::tempfile().unwrap(), + Duration::ZERO, + ) + .unwrap_err(); + assert!( + error + .message() + .contains("cannot read the guest boot console"), + "{error}" + ); + } + + #[test] + fn simple_exec_has_no_container_metadata_and_respects_timeout() { + let (request, timeout) = prepare_exec( + &ExecRequest::command_line("printf READY").with_timeout(Duration::from_secs(2)), + ) + .unwrap(); + assert_eq!(request.command, SHELL); + assert_eq!(request.args, ["-c", "printf READY"]); + assert_eq!(request.timeout_seconds, 2); + assert_eq!(timeout, Some(Duration::from_secs(2))); + assert!(request.encode_to_vec().len() < 100); + assert!(capabilities().exec.command_line); + assert!(!capabilities().exec.cancel); + assert!(!capabilities().filesystem.readonly_paths); + } + + #[test] + fn fractional_second_exec_timeouts_are_rejected_but_zero_is_unbounded() { + let (unbounded, timeout) = + prepare_exec(&ExecRequest::command_line("true").with_timeout(Duration::ZERO)).unwrap(); + assert_eq!(unbounded.timeout_seconds, 0); + assert_eq!(timeout, None); + for timeout in [ + Duration::from_nanos(1), + Duration::from_millis(1), + Duration::from_millis(999), + Duration::from_millis(1001), + ] { + let error = + prepare_exec(&ExecRequest::command_line("true").with_timeout(timeout)).unwrap_err(); + assert_eq!(error.code(), ErrorCode::PolicyValidation); + assert!(error.message().contains("whole-second precision")); + } + assert_eq!( + prepare_exec( + &ExecRequest::command_line("true") + .with_timeout(Duration::from_secs(MAX_EXEC_SECONDS + 1)), + ) + .unwrap_err() + .code(), + ErrorCode::PolicyValidation, + ); + } + + #[cfg(unix)] + #[test] + fn native_socket_paths_are_checked_before_loading_the_library() { + let suffix = PathBuf::from("0".repeat(32)).join(SOCKET_NAME); + let target_root_len = 107 - 1 - suffix.as_os_str().len(); + let state_root = PathBuf::from("/").join("x".repeat(target_root_len - 1)); + validate_unix_socket_path(&state_root, SOCKET_NAME, "control").unwrap(); + let native_root = state_root.join(BACKEND_KEY); + assert!(validate_unix_socket_path(&native_root, BOOT_SOCKET_NAME, "boot-console").is_err()); + let config = OpenVmmConfig::new( + "/tmp/openvmm", + "/tmp/vmlinux", + "/tmp/initramfs", + super::super::Hypervisor::Kvm, + &state_root, + ); + let error = NvxHostBackend::new(NvxHostConfig::new( + config, + "/tmp/image.vhd", + "/tmp/libnvxhost.so", + [0; 32], + )) + .unwrap_err(); + assert_eq!(error.code(), ErrorCode::BackendUnavailable); + assert!(error.message().contains("Unix control socket path")); + assert!(!native_root.exists()); + } + + #[test] + fn unsupported_exec_options_fail_before_guest_work() { + let request = ExecRequest::command_line("echo READY").with_cwd("/tmp"); + assert_eq!( + prepare_exec(&request).unwrap_err().code(), + ErrorCode::PolicyValidation + ); + let request = ExecRequest::command_line("echo READY").with_env("K=V"); + assert_eq!( + prepare_exec(&request).unwrap_err().code(), + ErrorCode::PolicyValidation + ); + } + + fn host_hypervisor() -> super::super::Hypervisor { + if cfg!(windows) { + super::super::Hypervisor::Whp + } else { + super::super::Hypervisor::Kvm + } + } + + #[test] + fn a_launch_log_claim_lasts_until_the_launcher_and_openvmm_have_exited() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("openvmm.log"); + assert!( + !platform::launch_log_released(&path).unwrap(), + "a missing log proves nothing" + ); + let log = File::create(&path).unwrap(); + platform::claim_launch_log(&log).unwrap(); + assert!(!platform::launch_log_released(&path).unwrap()); + + // A long-running child stands in for OpenVMM, which inherits the log as its output. + #[cfg(windows)] + let (program, arguments) = ( + PathBuf::from(std::env::var_os("SystemRoot").unwrap()).join(r"System32\PING.EXE"), + ["-n", "60", "127.0.0.1"].map(std::ffi::OsString::from), + ); + #[cfg(not(windows))] + let (program, arguments) = (PathBuf::from("sleep"), [std::ffi::OsString::from("60")]); + let config = OpenVmmConfig::new( + program, + "vmlinux", + "initramfs.cpio.gz", + host_hypervisor(), + directory.path(), + ); + let launched = + process::spawn(&config, &arguments, &[1; 32], log, directory.path()).unwrap(); + assert!( + !platform::launch_log_released(&path).unwrap(), + "the child's inherited copy no longer holds the claim" + ); + assert!(process::kill_child(launched)); + let deadline = Instant::now() + Duration::from_secs(10); + while !platform::launch_log_released(&path).unwrap() { + assert!(Instant::now() < deadline, "the claim outlived its holders"); + thread::sleep(Duration::from_millis(20)); + } + } + + #[test] + fn claimed_launch_markers_are_cleared_only_once_nothing_holds_the_log() { + let directory = tempfile::tempdir().unwrap(); + let mut config = OpenVmmConfig::new( + "openvmm", + "vmlinux", + "initramfs.cpio.gz", + host_hypervisor(), + directory.path(), + ); + config.start_timeout = Duration::from_millis(200); + let store = StateStore::open_for(&directory.path().join(BACKEND_KEY), BACKEND_KEY).unwrap(); + let base = OpenVmmBackend { config, store }; + let sandbox_id = SandboxId::generate().unwrap(); + let record = SandboxRecord { + format: STATE_FORMAT, + backend: BACKEND_KEY.to_owned(), + network: None, + filesystem: None, + native: None, + memory_mib: 256, + workload_uid: 65534, + workload_gid: 65534, + create_workload_account: false, + hostname: "nvx-sandbox".to_owned(), + }; + base.store.create(&sandbox_id, &record).unwrap(); + let endpoint = platform::control_endpoint(&base.store.socket_path(&sandbox_id)).unwrap(); + let marker = |log_claimed| LaunchRecord { + format: STATE_FORMAT, + endpoint: endpoint.clone(), + process: None, + log_claimed, + }; + let retained = |expected: &str| { + let Err(error) = base.reconcile(&sandbox_id) else { + panic!("an unproven launch marker was cleared"); + }; + assert_eq!(error.code(), ErrorCode::BackendError, "{error}"); + assert!(error.message().contains(expected), "{error}"); + assert!(base.store.launch(&sandbox_id).unwrap().is_some()); + }; + + // A claim whose log is missing proves nothing. + base.store.write_launch(&sandbox_id, &marker(true)).unwrap(); + retained("is missing or still held"); + + // Without a claim, an absent endpoint never proves that the launch exited. + drop(base.store.create_log(&sandbox_id).unwrap()); + base.store + .write_launch(&sandbox_id, &marker(false)) + .unwrap(); + retained("cannot verify that OpenVMM exited"); + + // A held claim is awaited for the start timeout, and the marker then retained. + let log = base.store.create_log(&sandbox_id).unwrap(); + platform::claim_launch_log(&log).unwrap(); + base.store.write_launch(&sandbox_id, &marker(true)).unwrap(); + let waited = Instant::now(); + retained("is missing or still held"); + assert!(waited.elapsed() >= Duration::from_millis(200)); + + // A released claim proves that nothing of the launch runs. + drop(log); + assert!(matches!( + base.reconcile(&sandbox_id).unwrap(), + RunState::Provisioned + )); + assert!(base.store.launch(&sandbox_id).unwrap().is_none()); + } +} diff --git a/aci_edge_sandboxes/src/openvmm/platform/linux.rs b/aci_edge_sandboxes/src/openvmm/platform/linux.rs index 76d1449d6..522892fbc 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/linux.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/linux.rs @@ -245,6 +245,59 @@ pub(crate) fn file_identity(path: &Path) -> io::Result<(u64, u64)> { Ok((metadata.dev(), metadata.ino())) } +#[cfg(feature = "nvxhost")] +pub(crate) use seal::{file_seal, open_sealable}; + +#[cfg(feature = "nvxhost")] +mod seal { + use std::fs::{File, OpenOptions}; + use std::io; + use std::os::unix::fs::{MetadataExt, OpenOptionsExt}; + use std::path::Path; + + use crate::openvmm::platform::FileSeal; + + /// Device and inode numbers, size, and modification and change times in nanoseconds. + const SCHEME: &str = "unix-file-v1"; + + /// Opens a file to seal it. + /// + /// Linux has no mandatory share modes, so `deny_writers` has no effect: comparing seals + /// taken from the handle detects a writer instead. + pub(crate) fn open_sealable(path: &Path, _deny_writers: bool) -> io::Result { + // A FIFO would otherwise block the open until a writer appears. + OpenOptions::new() + .read(true) + .custom_flags(libc::O_NONBLOCK) + .open(path) + } + + /// Returns the seal of an open regular file. + pub(crate) fn file_seal(file: &File) -> io::Result { + let metadata = file.metadata()?; + if !metadata.is_file() { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "the path does not name a regular file", + )); + } + Ok(FileSeal { + scheme: SCHEME.to_owned(), + volume: metadata.dev(), + file: format!("{:x}", metadata.ino()), + length: metadata.size(), + modified: nanoseconds(metadata.mtime(), metadata.mtime_nsec()), + changed: Some(nanoseconds(metadata.ctime(), metadata.ctime_nsec())), + }) + } + + fn nanoseconds(seconds: i64, nanoseconds: i64) -> i64 { + seconds + .saturating_mul(1_000_000_000) + .saturating_add(nanoseconds) + } +} + /// Returns the control endpoint OpenVMM should listen on for a sandbox. pub(crate) fn control_endpoint(socket_path: &Path) -> io::Result { socket_path @@ -300,6 +353,37 @@ pub(crate) fn endpoint_server_pid(endpoint: &str) -> io::Result> { } } +/// Claims a fresh OpenVMM log for one launch. +/// +/// The claim is an exclusive lock of the log's open file description, which OpenVMM shares +/// through its inherited standard output and error, so the claim lasts until both the launching +/// process and OpenVMM have exited. +#[cfg(feature = "nvxhost")] +pub(crate) fn claim_launch_log(log: &fs::File) -> io::Result<()> { + log.try_lock().map_err(|error| match error { + fs::TryLockError::WouldBlock => io::Error::new( + io::ErrorKind::WouldBlock, + "another process holds the OpenVMM log", + ), + fs::TryLockError::Error(error) => error, + }) +} + +/// Returns whether no process holds the claim on an OpenVMM log, which proves that the launch +/// that claimed it has no running process. A missing log proves nothing. +pub(crate) fn launch_log_released(path: &Path) -> io::Result { + let log = match fs::File::open(path) { + Ok(log) => log, + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false), + Err(error) => return Err(error), + }; + match log.try_lock() { + Ok(()) => Ok(true), + Err(fs::TryLockError::WouldBlock) => Ok(false), + Err(fs::TryLockError::Error(error)) => Err(error), + } +} + fn connect_nonblocking(path: &str) -> io::Result { // SAFETY: socket has no memory-safety preconditions. let descriptor = unsafe { diff --git a/aci_edge_sandboxes/src/openvmm/platform/mod.rs b/aci_edge_sandboxes/src/openvmm/platform/mod.rs index e1694540b..86fdac258 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/mod.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/mod.rs @@ -6,6 +6,9 @@ use std::io; use std::time::Duration; +#[cfg(feature = "nvxhost")] +use serde::{Deserialize, Serialize}; + #[cfg(target_os = "linux")] mod linux; #[cfg(not(any(target_os = "linux", windows)))] @@ -30,3 +33,31 @@ pub(crate) trait Transport: Send { /// Writes all of `data`, waiting at most `timeout` for each write to complete. fn write_all(&mut self, data: &[u8], timeout: Option) -> io::Result<()>; } + +/// Identity of a regular file and of its last change, compared instead of hashing the file again. +/// +/// Replacing the file or writing its data changes the seal. A writer that restores the last-write +/// time afterwards can preserve it on Windows, and a write within the file system's timestamp +/// granularity can preserve it on Linux, so a seal detects accidental change rather than +/// deliberate tampering. On Linux, any metadata change also changes the seal, because the seal +/// includes the change time, which writers cannot restore. Windows seals omit the change time: +/// writers can set it, and the system updates it when security components cache a file's hash in +/// its extended attributes, which would fail unchanged files closed. +#[cfg(feature = "nvxhost")] +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub(crate) struct FileSeal { + /// Platform scheme of the remaining fields. + pub(crate) scheme: String, + /// Volume serial number or device number. + pub(crate) volume: u64, + /// File ID or inode number, in hexadecimal. + pub(crate) file: String, + /// Length in bytes. + pub(crate) length: u64, + /// Last data modification, in the scheme's time unit. + pub(crate) modified: i64, + /// Last data or metadata change, in the scheme's time unit, where the scheme seals it. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(crate) changed: Option, +} diff --git a/aci_edge_sandboxes/src/openvmm/platform/other.rs b/aci_edge_sandboxes/src/openvmm/platform/other.rs index 4af82037a..0e555d17f 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/other.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/other.rs @@ -29,6 +29,16 @@ pub(crate) fn file_identity(_path: &Path) -> io::Result<(u64, u64)> { Err(unsupported()) } +#[cfg(feature = "nvxhost")] +pub(crate) fn open_sealable(_path: &Path, _deny_writers: bool) -> io::Result { + Err(unsupported()) +} + +#[cfg(feature = "nvxhost")] +pub(crate) fn file_seal(_file: &fs::File) -> io::Result { + Err(unsupported()) +} + pub(crate) fn detach(_command: &mut Command, _breakaway_from_job: bool) {} pub(crate) fn probe_hypervisor(_hypervisor: Hypervisor) -> Result<(), String> { @@ -57,3 +67,12 @@ pub(crate) fn connect_endpoint( pub(crate) fn endpoint_server_pid(_endpoint: &str) -> io::Result> { Ok(None) } + +#[cfg(feature = "nvxhost")] +pub(crate) fn claim_launch_log(_log: &fs::File) -> io::Result<()> { + Err(unsupported()) +} + +pub(crate) fn launch_log_released(_path: &Path) -> io::Result { + Err(unsupported()) +} diff --git a/aci_edge_sandboxes/src/openvmm/platform/windows.rs b/aci_edge_sandboxes/src/openvmm/platform/windows.rs index e8b8e3a39..ad3b277f1 100644 --- a/aci_edge_sandboxes/src/openvmm/platform/windows.rs +++ b/aci_edge_sandboxes/src/openvmm/platform/windows.rs @@ -11,13 +11,13 @@ use std::{mem, ptr}; use windows_sys::Win32::Foundation::{ ERROR_ACCESS_DENIED, ERROR_BROKEN_PIPE, ERROR_FILE_NOT_FOUND, ERROR_INVALID_PARAMETER, ERROR_IO_PENDING, ERROR_MORE_DATA, ERROR_NO_DATA, ERROR_OPERATION_ABORTED, ERROR_PIPE_BUSY, - ERROR_PIPE_NOT_CONNECTED, FILETIME, FreeLibrary, GENERIC_READ, GENERIC_WRITE, GetLastError, - HANDLE, HANDLE_FLAG_INHERIT, INVALID_HANDLE_VALUE, SetHandleInformation, WAIT_OBJECT_0, - WAIT_TIMEOUT, + ERROR_PIPE_NOT_CONNECTED, ERROR_SHARING_VIOLATION, FILETIME, FreeLibrary, GENERIC_READ, + GENERIC_WRITE, GetLastError, HANDLE, HANDLE_FLAG_INHERIT, INVALID_HANDLE_VALUE, + SetHandleInformation, WAIT_OBJECT_0, WAIT_TIMEOUT, }; use windows_sys::Win32::Storage::FileSystem::{ BY_HANDLE_FILE_INFORMATION, CreateFileW, FILE_FLAG_BACKUP_SEMANTICS, FILE_FLAG_OVERLAPPED, - GetFileInformationByHandle, OPEN_EXISTING, ReadFile, SECURITY_IDENTIFICATION, + FILE_SHARE_READ, GetFileInformationByHandle, OPEN_EXISTING, ReadFile, SECURITY_IDENTIFICATION, SECURITY_SQOS_PRESENT, WriteFile, }; use windows_sys::Win32::System::IO::{CancelIoEx, GetOverlappedResult, OVERLAPPED}; @@ -448,6 +448,97 @@ pub(crate) fn file_identity(path: &Path) -> io::Result<(u64, u64)> { )) } +#[cfg(feature = "nvxhost")] +pub(crate) use seal::{file_seal, open_sealable}; + +#[cfg(feature = "nvxhost")] +mod seal { + use std::fs::{File, OpenOptions}; + use std::io; + use std::mem; + use std::os::windows::fs::OpenOptionsExt; + use std::os::windows::io::AsRawHandle; + use std::path::Path; + + use windows_sys::Win32::Foundation::HANDLE; + use windows_sys::Win32::Storage::FileSystem::{ + FILE_BASIC_INFO, FILE_ID_INFO, FILE_INFO_BY_HANDLE_CLASS, FILE_READ_ATTRIBUTES, + FILE_SHARE_DELETE, FILE_SHARE_READ, FILE_SHARE_WRITE, FILE_STANDARD_INFO, FileBasicInfo, + FileIdInfo, FileStandardInfo, GetFileInformationByHandleEx, + }; + + use crate::openvmm::platform::FileSeal; + + /// Volume serial number, 128-bit file ID, end of file, and last-write time in 100-nanosecond + /// units. + const SCHEME: &str = "windows-file-v1"; + + /// Opens a file to seal it. + /// + /// With `deny_writers`, the handle reads the file and keeps writers, deletion, and renaming + /// out until it is closed, and opening fails while another handle can write the file. + /// Otherwise the handle reads only attributes and shares everything. + pub(crate) fn open_sealable(path: &Path, deny_writers: bool) -> io::Result { + let mut options = OpenOptions::new(); + if deny_writers { + options.read(true).share_mode(FILE_SHARE_READ); + } else { + options + .access_mode(FILE_READ_ATTRIBUTES) + .share_mode(FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE); + } + options.open(path) + } + + /// Returns the seal of an open regular file. + pub(crate) fn file_seal(file: &File) -> io::Result { + let handle = file.as_raw_handle() as HANDLE; + let standard: FILE_STANDARD_INFO = information(handle, FileStandardInfo)?; + if standard.Directory || standard.DeletePending { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "the path does not name a regular file", + )); + } + let basic: FILE_BASIC_INFO = information(handle, FileBasicInfo)?; + let identity: FILE_ID_INFO = information(handle, FileIdInfo)?; + Ok(FileSeal { + scheme: SCHEME.to_owned(), + volume: identity.VolumeSerialNumber, + file: identity + .FileId + .Identifier + .iter() + .rev() + .map(|byte| format!("{byte:02x}")) + .collect(), + length: u64::try_from(standard.EndOfFile) + .map_err(|_| io::Error::other("the file reports a negative length"))?, + modified: basic.LastWriteTime, + changed: None, + }) + } + + fn information(handle: HANDLE, class: FILE_INFO_BY_HANDLE_CLASS) -> io::Result { + let mut value = T::default(); + // SAFETY: the handle is open, and the buffer is a writable `T` of the size passed, which + // each caller pairs with the structure that `class` returns. + let succeeded = unsafe { + GetFileInformationByHandleEx( + handle, + class, + (&raw mut value).cast(), + mem::size_of::() as u32, + ) + } != 0; + if succeeded { + Ok(value) + } else { + Err(io::Error::last_os_error()) + } + } +} + /// Returns a fresh named-pipe endpoint for OpenVMM to listen on. pub(crate) fn control_endpoint(_socket_path: &Path) -> io::Result { let mut bytes = [0u8; 16]; @@ -550,6 +641,31 @@ pub(crate) fn endpoint_server_pid(endpoint: &str) -> io::Result> { } } +/// Claims a fresh OpenVMM log for one launch. +/// +/// The log's writable handles are the claim: OpenVMM inherits them as its standard output and +/// error, so the claim lasts until both the launching process and OpenVMM have exited. +#[cfg(feature = "nvxhost")] +pub(crate) fn claim_launch_log(_log: &fs::File) -> io::Result<()> { + Ok(()) +} + +/// Returns whether no process holds the claim on an OpenVMM log, which proves that the launch +/// that claimed it has no running process. A missing log proves nothing. +pub(crate) fn launch_log_released(path: &Path) -> io::Result { + // Opening without write sharing fails while any handle can write the log. + match OpenOptions::new() + .read(true) + .share_mode(FILE_SHARE_READ) + .open(path) + { + Ok(_) => Ok(true), + Err(error) if error.raw_os_error() == Some(ERROR_SHARING_VIOLATION as i32) => Ok(false), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(false), + Err(error) => Err(error), + } +} + struct PipeTransport { pipe: OwnedHandle, event: OwnedHandle, diff --git a/aci_edge_sandboxes/src/openvmm/state.rs b/aci_edge_sandboxes/src/openvmm/state.rs index 064ef51bb..8a60543ed 100644 --- a/aci_edge_sandboxes/src/openvmm/state.rs +++ b/aci_edge_sandboxes/src/openvmm/state.rs @@ -3,6 +3,7 @@ //! ```text //! / //! .locks/.lock serializes lifecycle transitions of one sandbox +//! images/ registered guest images (nvxhost backend only) //! / //! sandbox.json provisioned configuration //! launch.json endpoint and available process identity of a start in progress @@ -23,6 +24,8 @@ use serde::{Deserialize, Serialize}; use super::filesystem::HostMapping; use super::platform; use super::protocol::CAPABILITY_LEN; +#[cfg(any(test, feature = "nvxhost"))] +use crate::error::ErrorCode; use crate::error::{Error, Result}; use crate::id::SandboxId; use crate::model::NetworkPolicy; @@ -33,14 +36,22 @@ use crate::model::NetworkPolicy; pub(crate) const STATE_FORMAT: u32 = 2; /// Backend key recorded in every sandbox. pub(crate) const BACKEND_KEY: &str = "openvmm"; +/// Backend key for the separately supplied native host library. +pub(crate) const NATIVE_BACKEND_KEY: &str = "nvxhost"; /// Linux control socket name. pub(crate) const SOCKET_NAME: &str = "control.sock"; +/// Linux boot-console socket name. +pub(crate) const BOOT_SOCKET_NAME: &str = "boot.sock"; +/// Directory of the nvxhost backend's image registry. +#[cfg(feature = "nvxhost")] +pub(crate) const IMAGES_NAME: &str = "images"; const RECORD_NAME: &str = "sandbox.json"; const LAUNCH_NAME: &str = "launch.json"; const RUNTIME_NAME: &str = "runtime.json"; const CAPABILITY_NAME: &str = "control.capability"; const LOG_NAME: &str = "openvmm.log"; +const CONSOLE_LOG_NAME: &str = "console.log"; const OUTCOME_NAME: &str = "outcome.json"; const LOCKS_NAME: &str = ".locks"; @@ -54,6 +65,8 @@ pub(crate) struct SandboxRecord { pub(crate) network: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub(crate) filesystem: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(crate) native: Option, pub(crate) memory_mib: u32, pub(crate) workload_uid: u32, pub(crate) workload_gid: u32, @@ -64,6 +77,20 @@ pub(crate) struct SandboxRecord { pub(crate) hostname: String, } +/// Artifacts that a native sandbox was provisioned with. +/// +/// The image is referenced by its registered content ID (`sha256:`), whose registration +/// supplies the path; the digests of the runtime files and host library are lowercase hexadecimal. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +pub(crate) struct NativeArtifactRecord { + pub(crate) image: String, + pub(crate) openvmm_sha256: String, + pub(crate) kernel_sha256: String, + pub(crate) initrd_sha256: String, + pub(crate) library_sha256: String, +} + /// Identity of the OpenVMM process of a running sandbox. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "camelCase", deny_unknown_fields)] @@ -77,7 +104,9 @@ pub(crate) struct RuntimeRecord { /// Marker written before OpenVMM is launched and replaced by the runtime record afterwards. /// /// Process identity is added as soon as the child is identified, before writing runtime state. -/// A marker without identity is never expired: its child may be alive without an endpoint. +/// A marker without identity is never expired: its child may be alive without an endpoint. A +/// marker that records a claim on the OpenVMM log is the exception, because OpenVMM inherits the +/// claim, so a released claim proves that nothing of the launch still runs. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(rename_all = "camelCase", deny_unknown_fields)] pub(crate) struct LaunchRecord { @@ -85,6 +114,9 @@ pub(crate) struct LaunchRecord { pub(crate) endpoint: String, #[serde(default, skip_serializing_if = "Option::is_none")] pub(crate) process: Option, + /// Whether the launching process claimed the OpenVMM log before writing this marker. + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + pub(crate) log_claimed: bool, } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] @@ -104,6 +136,7 @@ pub(crate) struct LockGuard { #[derive(Debug)] pub(crate) struct StateStore { root: PathBuf, + backend: &'static str, } fn io_error(context: String) -> impl FnOnce(io::Error) -> Error { @@ -113,6 +146,10 @@ fn io_error(context: String) -> impl FnOnce(io::Error) -> Error { impl StateStore { /// Opens `root`, creating it and restricting it to the current user if needed. pub(crate) fn open(root: &Path) -> io::Result { + Self::open_for(root, BACKEND_KEY) + } + + pub(crate) fn open_for(root: &Path, backend: &'static str) -> io::Result { if let Some(parent) = root.parent() { fs::create_dir_all(parent)?; } @@ -120,6 +157,7 @@ impl StateStore { platform::create_private_dir(&root.join(LOCKS_NAME))?; Ok(Self { root: root.to_path_buf(), + backend, }) } @@ -131,6 +169,11 @@ impl StateStore { self.dir(sandbox_id).join(LOG_NAME) } + #[cfg(feature = "nvxhost")] + pub(crate) fn console_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.dir(sandbox_id).join(CONSOLE_LOG_NAME) + } + pub(crate) fn outcome_path(&self, sandbox_id: &SandboxId) -> PathBuf { self.dir(sandbox_id).join(OUTCOME_NAME) } @@ -139,6 +182,11 @@ impl StateStore { self.dir(sandbox_id).join(SOCKET_NAME) } + #[cfg(feature = "nvxhost")] + pub(crate) fn boot_socket_path(&self, sandbox_id: &SandboxId) -> PathBuf { + self.dir(sandbox_id).join(BOOT_SOCKET_NAME) + } + fn lock_path(&self, sandbox_id: &SandboxId) -> PathBuf { self.root .join(LOCKS_NAME) @@ -150,13 +198,63 @@ impl StateStore { lock(&self.lock_path(sandbox_id)) } + #[cfg(any(test, feature = "nvxhost"))] + pub(crate) fn lock_and_load( + &self, + sandbox_id: &SandboxId, + ) -> Result<(LockGuard, SandboxRecord)> { + match self.dir(sandbox_id).try_exists() { + Ok(false) => { + return Err(Error::stale_id(format!( + "sandbox {sandbox_id} is not provisioned" + ))); + } + Err(error) => { + return Err(Error::backend_error("cannot inspect sandbox state").with_source(error)); + } + Ok(true) => {} + } + let guard = self.lock(sandbox_id)?; + match self.load(sandbox_id) { + Ok(record) => Ok((guard, record)), + Err(error) if error.code() == ErrorCode::StaleId => { + drop(guard); + self.remove_lock(sandbox_id); + Err(error) + } + Err(error) => Err(error), + } + } + /// Removes the lock file of a deprovisioned sandbox. pub(crate) fn remove_lock(&self, sandbox_id: &SandboxId) { let _ = fs::remove_file(self.lock_path(sandbox_id)); } + /// Lists the IDs of the sandboxes that have state directories. + #[cfg(feature = "nvxhost")] + pub(crate) fn sandbox_ids(&self) -> Result> { + let list_error = || format!("cannot list sandbox state {}", self.root.display()); + let mut ids = Vec::new(); + for entry in fs::read_dir(&self.root).map_err(io_error(list_error()))? { + let name = entry.map_err(io_error(list_error()))?.file_name(); + if let Some(id) = name + .to_str() + .and_then(|token| SandboxId::parse(&format!("{}:{token}", SandboxId::PREFIX)).ok()) + { + ids.push(id); + } + } + Ok(ids) + } + /// Creates the state directory of a new sandbox. pub(crate) fn create(&self, sandbox_id: &SandboxId, record: &SandboxRecord) -> Result<()> { + if record.backend != self.backend { + return Err(Error::backend_error( + "the sandbox record does not belong to this backend", + )); + } let dir = self.dir(sandbox_id); fs::create_dir(&dir).map_err(io_error(format!( "cannot create sandbox state {}", @@ -185,7 +283,7 @@ impl StateStore { ))); } }; - if record.format != STATE_FORMAT || record.backend != BACKEND_KEY { + if record.format != STATE_FORMAT || record.backend != self.backend { return Err(Error::backend_error(format!( "sandbox state {} has an unsupported format", path.display() @@ -266,6 +364,9 @@ impl StateStore { for name in [RUNTIME_NAME, LAUNCH_NAME, CAPABILITY_NAME, SOCKET_NAME] { remove_if_present(&dir.join(name))?; } + if self.backend == NATIVE_BACKEND_KEY { + remove_if_present(&dir.join(BOOT_SOCKET_NAME))?; + } Ok(()) } @@ -278,6 +379,15 @@ impl StateStore { .map_err(io_error(format!("cannot create {}", path.display()))) } + #[cfg(feature = "nvxhost")] + pub(crate) fn open_console_log(&self, sandbox_id: &SandboxId) -> Result { + let path = self.console_path(sandbox_id); + private_options() + .append(true) + .open(&path) + .map_err(io_error(format!("cannot open {}", path.display()))) + } + /// Deletes the state of a stopped sandbox. /// /// Unknown files abort the removal before anything is deleted. The configuration is removed @@ -296,6 +406,7 @@ impl StateStore { OUTCOME_NAME, LOG_NAME, ]; + let native_only = [BOOT_SOCKET_NAME, CONSOLE_LOG_NAME]; let mut owned = Vec::new(); for entry in entries { let entry = entry.map_err(io_error(format!( @@ -305,7 +416,10 @@ impl StateStore { let name = entry.file_name(); let name = name.to_string_lossy(); let temporary = name.starts_with('.') && name.ends_with(".tmp"); - if known.contains(&name.as_ref()) || temporary { + if known.contains(&name.as_ref()) + || (self.backend == NATIVE_BACKEND_KEY && native_only.contains(&name.as_ref())) + || temporary + { owned.push(entry.path()); } else if name != RECORD_NAME { return Err(Error::backend_error(format!( @@ -334,7 +448,8 @@ pub(crate) fn remove_if_present(path: &Path) -> Result<()> { } } -fn lock(path: &Path) -> Result { +/// Takes an exclusive advisory lock on `path`, creating the lock file if needed. +pub(crate) fn lock(path: &Path) -> Result { let file = OpenOptions::new() .read(true) .write(true) @@ -378,21 +493,24 @@ fn write_private(path: &Path, contents: &[u8]) -> Result<()> { written.map_err(io_error(format!("cannot write {}", path.display()))) } -fn write_json(path: &Path, value: &T) -> Result<()> { - let mut text = serde_json::to_vec_pretty(value) - .map_err(|error| Error::backend_error("cannot encode sandbox state").with_source(error))?; +/// Atomically replaces `path` with `value` as JSON, readable only by the current user. +pub(crate) fn write_json(path: &Path, value: &T) -> Result<()> { + let mut text = serde_json::to_vec_pretty(value).map_err(|error| { + Error::backend_error(format!("cannot encode {}", path.display())).with_source(error) + })?; text.push(b'\n'); write_private(path, &text) } -fn read_json(path: &Path) -> Result> { +/// Reads a JSON state file, returning `None` if it does not exist. +pub(crate) fn read_json(path: &Path) -> Result> { let text = match fs::read(path) { Ok(text) => text, Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None), Err(error) => return Err(io_error(format!("cannot read {}", path.display()))(error)), }; serde_json::from_slice(&text).map(Some).map_err(|error| { - Error::backend_error(format!("sandbox state {} is malformed", path.display())) + Error::backend_error(format!("state file {} is malformed", path.display())) .with_source(error) }) } @@ -408,6 +526,7 @@ mod tests { backend: BACKEND_KEY.to_owned(), network: None, filesystem: None, + native: None, memory_mib: 256, workload_uid: 65534, workload_gid: 65534, @@ -422,11 +541,73 @@ mod tests { let store = StateStore::open(&root.path().join("sandboxes")).unwrap(); let id = SandboxId::generate().unwrap(); assert_eq!(store.load(&id).unwrap_err().code(), ErrorCode::StaleId); + assert_eq!( + store.lock_and_load(&id).unwrap_err().code(), + ErrorCode::StaleId + ); + assert!(!store.dir(&id).exists()); + assert!(!store.lock_path(&id).exists()); store.create(&id, &record()).unwrap(); assert_eq!(store.load(&id).unwrap(), record()); assert!(store.runtime(&id).unwrap().is_none()); } + #[test] + fn native_records_round_trip_only_through_their_backend() { + let root = tempfile::tempdir().unwrap(); + let store = StateStore::open_for(root.path(), NATIVE_BACKEND_KEY).unwrap(); + let id = SandboxId::generate().unwrap(); + assert_eq!( + store.create(&id, &record()).unwrap_err().code(), + ErrorCode::BackendError + ); + assert!(!store.dir(&id).exists()); + + let mut native = record(); + native.backend = NATIVE_BACKEND_KEY.to_owned(); + native.native = Some(NativeArtifactRecord { + image: format!("sha256:{}", "1".repeat(64)), + openvmm_sha256: "5".repeat(64), + kernel_sha256: "2".repeat(64), + initrd_sha256: "3".repeat(64), + library_sha256: "4".repeat(64), + }); + store.create(&id, &native).unwrap(); + let (_guard, loaded) = store.lock_and_load(&id).unwrap(); + assert_eq!(loaded, native); + assert_eq!( + StateStore::open(root.path()) + .unwrap() + .load(&id) + .unwrap_err() + .code(), + ErrorCode::BackendError + ); + drop(_guard); + fs::write(store.dir(&id).join(BOOT_SOCKET_NAME), b"socket").unwrap(); + fs::write(store.dir(&id).join(CONSOLE_LOG_NAME), b"console").unwrap(); + store.clear_runtime(&id).unwrap(); + assert!(!store.dir(&id).join(BOOT_SOCKET_NAME).exists()); + store.remove(&id).unwrap(); + assert!(!store.dir(&id).exists()); + } + + #[test] + fn direct_records_preserve_native_only_filenames() { + let root = tempfile::tempdir().unwrap(); + let store = StateStore::open(root.path()).unwrap(); + let id = SandboxId::generate().unwrap(); + store.create(&id, &record()).unwrap(); + for name in [BOOT_SOCKET_NAME, CONSOLE_LOG_NAME] { + fs::write(store.dir(&id).join(name), b"not owned").unwrap(); + } + store.clear_runtime(&id).unwrap(); + assert!(store.dir(&id).join(BOOT_SOCKET_NAME).exists()); + assert!(store.remove(&id).is_err()); + assert!(store.dir(&id).join(CONSOLE_LOG_NAME).exists()); + assert_eq!(store.load(&id).unwrap(), record()); + } + #[test] fn runtime_files_are_cleared_and_state_removed() { let root = tempfile::tempdir().unwrap(); @@ -438,9 +619,19 @@ mod tests { format: STATE_FORMAT, endpoint: "endpoint".to_owned(), process: None, + log_claimed: false, }; store.write_launch(&id, &launch).unwrap(); assert_eq!(store.launch(&id).unwrap(), Some(launch.clone())); + // Markers without a claim keep their earlier encoding. + let text = fs::read_to_string(store.dir(&id).join(LAUNCH_NAME)).unwrap(); + assert!(!text.contains("logClaimed"), "{text}"); + let claimed = LaunchRecord { + log_claimed: true, + ..launch.clone() + }; + store.write_launch(&id, &claimed).unwrap(); + assert_eq!(store.launch(&id).unwrap(), Some(claimed)); let identified = LaunchRecord { process: Some(ProcessIdentity { pid: 1, diff --git a/aci_edge_sandboxes/tests/nvxhost_guest.rs b/aci_edge_sandboxes/tests/nvxhost_guest.rs new file mode 100644 index 000000000..2f99801d0 --- /dev/null +++ b/aci_edge_sandboxes/tests/nvxhost_guest.rs @@ -0,0 +1,907 @@ +//! Opt-in WHP lifecycle and robustness proofs with a separately supplied native library and guest. +//! +//! The tests need `NVXHOST_TEST_OPENVMM`, `NVXHOST_TEST_KERNEL`, `NVXHOST_TEST_INITRD`, +//! `NVXHOST_TEST_IMAGE`, `NVXHOST_TEST_LIBRARY`, and the approved `NVXHOST_TEST_SHA256`. Run them +//! one at a time so that the check for leftover OpenVMM processes is exact, and optimized, since +//! creating a backend hashes the runtime files and registering an image hashes the image: +//! `cargo test --release --features nvxhost --test nvxhost_guest -- --ignored --test-threads=1`. +#![cfg(feature = "nvxhost")] + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::{Child, Command, Stdio}; +use std::sync::Arc; +use std::thread; +use std::time::{Duration, Instant}; + +use aci_edge_sandboxes::openvmm::{ + Hypervisor, ImageDigest, ImageId, NvxHostBackend, NvxHostConfig, OpenVmmConfig, +}; +use aci_edge_sandboxes::{ + AciEdgeSandbox, Error, ErrorCode, ExecOutcome, ExecOutput, ExecRequest, ProvisionRequest, + Result, SandboxId, StdinMode, StopResult, +}; + +const HELPER_STATE: &str = "NVXHOST_TEST_HELPER_STATE"; +const HELPER_SANDBOX: &str = "NVXHOST_TEST_HELPER_SANDBOX"; +/// Hashing the test image took most of a second of every start before images were registered. +const MAX_START_OVERHEAD: Duration = Duration::from_millis(500); + +fn required(name: &str) -> PathBuf { + PathBuf::from(std::env::var_os(name).unwrap_or_else(|| panic!("{name} must be set"))) +} + +fn approved_digest() -> [u8; 32] { + let hex = std::env::var("NVXHOST_TEST_SHA256").expect("NVXHOST_TEST_SHA256 must be set"); + assert_eq!(hex.len(), 64, "the approved digest must contain 64 digits"); + let mut digest = [0u8; 32]; + for (index, value) in digest.iter_mut().enumerate() { + *value = u8::from_str_radix(&hex[index * 2..index * 2 + 2], 16) + .expect("the approved digest must be hexadecimal"); + } + digest +} + +/// Returns the test configuration for `image`, with sandbox state under `state`. +fn native_config(state: &Path, image: &Path) -> NvxHostConfig { + NvxHostConfig::new( + OpenVmmConfig::new( + required("NVXHOST_TEST_OPENVMM"), + required("NVXHOST_TEST_KERNEL"), + required("NVXHOST_TEST_INITRD"), + Hypervisor::Whp, + state, + ), + image, + required("NVXHOST_TEST_LIBRARY"), + approved_digest(), + ) +} + +fn backend_with( + state: &Path, + guest_debug: bool, + adjust: impl FnOnce(&mut OpenVmmConfig), +) -> Arc { + let mut config = + native_config(state, &required("NVXHOST_TEST_IMAGE")).with_guest_debug(guest_debug); + adjust(&mut config.openvmm); + Arc::new(NvxHostBackend::new(config).unwrap()) +} + +fn backend(state: &Path) -> Arc { + backend_with(state, false, |_| {}) +} + +/// Copies the test image into `state`, so that a test can change the copy. +fn copied_image(state: &Path) -> PathBuf { + let image = state.join("image.vhd"); + std::fs::copy(required("NVXHOST_TEST_IMAGE"), &image).unwrap(); + image +} + +/// Moves a file's modification time without changing its content. +fn touch(path: &Path) { + let file = std::fs::File::options().write(true).open(path).unwrap(); + let modified = file.metadata().unwrap().modified().unwrap(); + file.set_modified(modified + Duration::from_secs(2)) + .unwrap(); +} + +/// Lists the OpenVMM processes running on this host, failing if they cannot be listed. +fn openvmm_processes() -> BTreeSet { + let executable = required("NVXHOST_TEST_OPENVMM"); + let name = executable + .file_stem() + .unwrap() + .to_string_lossy() + .into_owned(); + #[cfg(windows)] + let output = Command::new("powershell.exe") + .args([ + "-NoProfile", + "-NonInteractive", + "-Command", + &format!("[Diagnostics.Process]::GetProcessesByName('{name}') | ForEach-Object Id"), + ]) + .output() + .unwrap(); + #[cfg(not(windows))] + let output = Command::new("pgrep").args(["-x", &name]).output().unwrap(); + // pgrep exits with 1 when no process matches. + assert!( + output.status.success() || (cfg!(not(windows)) && output.status.code() == Some(1)), + "cannot list OpenVMM processes: {output:?}" + ); + String::from_utf8_lossy(&output.stdout) + .lines() + .map(str::trim) + .filter(|line| !line.is_empty()) + .map(|line| { + line.parse() + .unwrap_or_else(|_| panic!("unexpected process ID {line:?}")) + }) + .collect() +} + +/// Fails if an OpenVMM process that was not running before the test is still running. +fn assert_no_new_openvmm(before: &BTreeSet) { + let deadline = Instant::now() + Duration::from_secs(10); + loop { + let leaked: Vec = openvmm_processes().difference(before).copied().collect(); + if leaked.is_empty() { + return; + } + assert!( + Instant::now() < deadline, + "OpenVMM processes {leaked:?} outlived their sandboxes" + ); + thread::sleep(Duration::from_millis(100)); + } +} + +/// Terminates a process abruptly, as a crash would. +fn kill_process(pid: u32) { + #[cfg(windows)] + let status = Command::new("powershell.exe") + .args([ + "-NoProfile", + "-NonInteractive", + "-Command", + &format!("Stop-Process -Id {pid} -Force -ErrorAction Stop"), + ]) + .status() + .unwrap(); + #[cfg(not(windows))] + let status = Command::new("kill") + .args(["-KILL", &pid.to_string()]) + .status() + .unwrap(); + assert!(status.success(), "cannot kill process {pid}: {status}"); +} + +/// Stops and removes a sandbox when a test ends, printing its diagnostics if the test failed. +struct Cleanup<'a> { + client: &'a AciEdgeSandbox, + backend: &'a NvxHostBackend, + id: SandboxId, + armed: bool, +} + +impl Cleanup<'_> { + fn release(mut self) -> SandboxId { + self.armed = false; + self.id.clone() + } +} + +impl Drop for Cleanup<'_> { + fn drop(&mut self) { + if !self.armed { + return; + } + if thread::panicking() { + let read = |path: PathBuf| { + std::fs::read_to_string(&path) + .unwrap_or_else(|error| format!("cannot read {}: {error}", path.display())) + }; + eprintln!( + "sandbox {}\nOpenVMM log: {}\nguest console: {}\nOpenVMM outcome: {}", + self.id, + read(self.backend.log_path(&self.id)), + read(self.backend.console_log_path(&self.id)), + read(self.backend.outcome_report_path(&self.id)), + ); + } + let _ = self.client.stop(&self.id); + let _ = self.client.deprovision(&self.id); + } +} + +fn provision<'a>(client: &'a AciEdgeSandbox, backend: &'a NvxHostBackend) -> Cleanup<'a> { + let id = client + .provision(&ProvisionRequest::new()) + .unwrap() + .sandbox_id; + Cleanup { + client, + backend, + id, + armed: true, + } +} + +fn exec(client: &AciEdgeSandbox, id: &SandboxId, request: &ExecRequest) -> ExecOutput { + client + .exec(id, request) + .and_then(|execution| execution.wait_with_output()) + .unwrap_or_else(|error| panic!("{request:?} failed: {error}")) +} + +fn assert_prints(client: &AciEdgeSandbox, id: &SandboxId, text: &str) { + let output = exec( + client, + id, + &ExecRequest::command_line(format!("printf %s {text}")), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(0), "{output:?}"); + assert_eq!(String::from_utf8_lossy(&output.stdout), text); +} + +fn forced(stopped: &StopResult) -> Option { + stopped + .metadata + .as_ref() + .and_then(|metadata| metadata.get("forced")) + .and_then(serde_json::Value::as_bool) +} + +/// Fails if the boot-console capture reported an error. +fn assert_console_captured(stopped: &StopResult) { + assert!( + stopped + .metadata + .as_ref() + .is_none_or(|metadata| !metadata.contains_key("consoleError")), + "{:?}", + stopped.metadata + ); +} + +fn assert_graceful(stopped: Result) { + let stopped = stopped.unwrap(); + assert_eq!(forced(&stopped), Some(false), "{:?}", stopped.metadata); + assert_console_captured(&stopped); +} + +fn failure(result: Result) -> ErrorCode { + match result { + Ok(_) => panic!("the operation unexpectedly succeeded"), + Err(error) => error.code(), + } +} + +/// Fails unless `result` is a `BackendUnavailable` error whose message contains `expected`. +fn assert_unavailable(result: Result, expected: &str) { + match result { + Ok(_) => { + panic!("the operation unexpectedly succeeded instead of failing with {expected:?}") + } + Err(error) => { + assert_eq!(error.code(), ErrorCode::BackendUnavailable, "{error}"); + assert!(error.message().contains(expected), "{error}"); + } + } +} + +/// Starts a sandbox and returns the time that the call spent outside the guest boot. +fn start_overhead(client: &AciEdgeSandbox, id: &SandboxId) -> Duration { + let called = Instant::now(); + let started = client.start(id).unwrap(); + let total = called.elapsed(); + let boot = started + .metadata + .as_ref() + .and_then(|metadata| metadata.get("bootMilliseconds")) + .and_then(serde_json::Value::as_u64) + .unwrap(); + total.saturating_sub(Duration::from_millis(boot)) +} + +fn start_in_helper(state: &Path, id: &SandboxId) -> Child { + Command::new(std::env::current_exe().unwrap()) + .args([ + "helper_starts_a_sandbox_in_another_process", + "--exact", + "--ignored", + "--nocapture", + "--test-threads=1", + ]) + .env(HELPER_STATE, state) + .env(HELPER_SANDBOX, id.as_str()) + .stdin(Stdio::null()) + .spawn() + .unwrap() +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_guest_lifecycle_stops_gracefully_and_cleans_up() { + let state = tempfile::tempdir().unwrap(); + let config = OpenVmmConfig::new( + required("NVXHOST_TEST_OPENVMM"), + required("NVXHOST_TEST_KERNEL"), + required("NVXHOST_TEST_INITRD"), + Hypervisor::Whp, + state.path(), + ); + let backend = Arc::new( + NvxHostBackend::new( + NvxHostConfig::new( + config, + required("NVXHOST_TEST_IMAGE"), + required("NVXHOST_TEST_LIBRARY"), + approved_digest(), + ) + .with_guest_debug(true), + ) + .unwrap(), + ); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox_id = client + .provision(&ProvisionRequest::new()) + .unwrap() + .sandbox_id; + let result = (|| { + let started = client.start(&sandbox_id)?; + if !started + .metadata + .as_ref() + .is_some_and(|metadata| metadata.contains_key("guestBuildId")) + { + return Err(Error::new( + ErrorCode::BackendError, + "the guest did not report a build ID", + )); + } + let executed = client + .exec(&sandbox_id, &ExecRequest::command_line("printf READY"))? + .wait_with_output()?; + if executed.outcome != ExecOutcome::Exited(0) || executed.stdout != b"READY" { + return Err(Error::new( + ErrorCode::BackendError, + format!( + "guest exec outcome {:?}, stdout {:?}, stderr {:?}", + executed.outcome, + String::from_utf8_lossy(&executed.stdout), + String::from_utf8_lossy(&executed.stderr), + ), + )); + } + let logs = backend.guest_logs(&sandbox_id)?; + if !logs.windows(8).any(|window| window == b"execute:") { + return Err(Error::new( + ErrorCode::BackendError, + "guest log stream omitted the executed command", + )); + } + let stopped = client.stop(&sandbox_id)?; + if stopped + .metadata + .as_ref() + .and_then(|metadata| metadata.get("forced")) + .and_then(serde_json::Value::as_bool) + != Some(false) + { + return Err(Error::new( + ErrorCode::BackendError, + format!("guest shutdown was not graceful: {:?}", stopped.metadata), + )); + } + let console = std::fs::read(backend.console_log_path(&sandbox_id)).map_err(|error| { + Error::new( + ErrorCode::BackendError, + "cannot read the guest boot console", + ) + .with_source(error) + })?; + if [b"NVX-EDGE-FATAL".as_slice(), b"NVX-EDGE-SHUTDOWN-ERROR"] + .iter() + .any(|marker| { + console + .windows(marker.len()) + .any(|window| window == *marker) + }) + { + return Err(Error::new( + ErrorCode::BackendError, + "the guest reported a fatal or shutdown error", + )); + } + client.deprovision(&sandbox_id)?; + if state + .path() + .join("nvxhost") + .join(sandbox_id.token()) + .exists() + { + return Err(Error::new( + ErrorCode::BackendError, + "deprovision left sandbox state behind", + )); + } + Ok::<_, aci_edge_sandboxes::Error>(()) + })(); + if let Err(error) = result { + let log = std::fs::read_to_string(backend.log_path(&sandbox_id)) + .unwrap_or_else(|read| format!("cannot read OpenVMM diagnostic log: {read}")); + let console = std::fs::read_to_string(backend.console_log_path(&sandbox_id)) + .unwrap_or_else(|read| format!("cannot read guest boot console: {read}")); + let outcome = std::fs::read_to_string(backend.outcome_report_path(&sandbox_id)) + .unwrap_or_else(|read| format!("cannot read OpenVMM outcome report: {read}")); + let stop = client.stop(&sandbox_id); + let deprovision = client.deprovision(&sandbox_id); + panic!( + "WHP lifecycle failed: {error}; recovery stop: {stop:?}; \ + recovery deprovision: {deprovision:?}; OpenVMM log: {log}; \ + guest console: {console}; OpenVMM outcome: {outcome}" + ); + } +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_exec_reports_exit_codes_streams_timeouts_and_output_limits() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let id = &sandbox.id; + client.start(id).unwrap(); + + let output = exec( + &client, + id, + &ExecRequest::command_line("printf out; printf err >&2; exit 7"), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(7)); + assert_eq!(output.stdout, b"out"); + assert_eq!(output.stderr, b"err"); + + let output = exec( + &client, + id, + &ExecRequest::argv([ + "/bin/sh", + "-c", + "printf '%s|' \"$@\"", + "sh", + "a b", + "$HOME", + ]), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(0)); + assert_eq!(output.stdout, b"a b|$HOME|"); + + let started = Instant::now(); + let output = exec( + &client, + id, + &ExecRequest::command_line("sleep 30").with_timeout(Duration::from_secs(1)), + ); + assert_eq!(output.outcome, ExecOutcome::TimedOut); + assert!( + started.elapsed() < Duration::from_secs(20), + "a one-second timeout took {:?}", + started.elapsed() + ); + + let error = client + .exec(id, &ExecRequest::command_line("yes | head -c 1100000")) + .and_then(|execution| execution.wait_with_output()) + .expect_err("output above the guest limit must fail rather than truncate"); + assert_eq!(error.code(), ErrorCode::BackendError, "{error}"); + assert!(error.message().contains("status 8"), "{error}"); + assert_prints(&client, id, "after-overflow"); + + for index in 0..20 { + assert_prints(&client, id, &format!("command{index}")); + } + for request in [ + ExecRequest::command_line("pwd").with_cwd("/tmp"), + ExecRequest::command_line("env").with_env("NAME=value"), + ExecRequest::command_line("cat").with_stdin(StdinMode::Piped), + ] { + assert_eq!( + failure(client.exec(id, &request)), + ErrorCode::PolicyValidation + ); + } + let logs = backend.guest_logs(id).unwrap(); + assert!(logs.windows(8).any(|window| window == b"execute:")); + assert!(logs.windows(13).any(|window| window == b"output-limit:")); + + assert_graceful(client.stop(id)); + client.deprovision(id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_lifecycle_errors_follow_state_and_a_stopped_sandbox_restarts() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let id = &sandbox.id; + + assert_eq!( + failure(client.exec(id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + client.start(id).unwrap(); + assert_eq!(failure(client.start(id)), ErrorCode::AlreadyStarted); + assert_eq!(failure(client.deprovision(id)), ErrorCode::AlreadyStarted); + assert_prints(&client, id, "first"); + assert_graceful(client.stop(id)); + assert_eq!(failure(client.stop(id)), ErrorCode::AlreadyStopped); + assert_eq!( + failure(client.exec(id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + + client.start(id).unwrap(); + assert_prints(&client, id, "second"); + assert_graceful(client.stop(id)); + client.deprovision(id).unwrap(); + assert_eq!(failure(client.start(id)), ErrorCode::StaleId); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_new_backend_reattaches_to_a_running_guest_and_forces_a_stop() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let id = { + let first = backend(state.path()); + let client = AciEdgeSandbox::from_shared(first.clone()); + let sandbox = provision(&client, &first); + client.start(&sandbox.id).unwrap(); + sandbox.release() + }; + + let second = backend_with(state.path(), false, |config| { + config.stop_timeout = Duration::from_millis(1); + }); + let client = AciEdgeSandbox::from_shared(second.clone()); + let sandbox = Cleanup { + client: &client, + backend: &second, + id, + armed: true, + }; + assert_prints(&client, &sandbox.id, "reattached"); + let logs = second.guest_logs(&sandbox.id).unwrap(); + assert!(logs.windows(8).any(|window| window == b"execute:")); + let stopped = client.stop(&sandbox.id).unwrap(); + assert_eq!(forced(&stopped), Some(true), "{:?}", stopped.metadata); + // The first backend's listener still held the boot console, so the second one waited. + assert_console_captured(&stopped); + assert_eq!( + failure(client.exec(&sandbox.id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + client.deprovision(&sandbox.id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_a_crashed_guest_is_not_running_and_restarts_with_its_console_captured() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend_with(state.path(), true, |_| {}); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let id = &sandbox.id; + client.start(id).unwrap(); + assert_prints(&client, id, "before-crash"); + + let launched: Vec = openvmm_processes().difference(&before).copied().collect(); + assert_eq!(launched.len(), 1, "{launched:?}"); + kill_process(launched[0]); + assert_no_new_openvmm(&before); + assert_eq!( + failure(client.exec(id, &ExecRequest::command_line("true"))), + ErrorCode::NotStarted + ); + assert_eq!(failure(client.stop(id)), ErrorCode::AlreadyStopped); + + let console = backend.console_log_path(id); + let crashed = std::fs::metadata(&console).unwrap().len(); + client.start(id).unwrap(); + assert_prints(&client, id, "after-crash"); + assert_graceful(client.stop(id)); + let restarted = std::fs::metadata(&console).unwrap().len(); + assert!( + restarted > crashed, + "the restarted guest's boot console was not captured ({crashed} -> {restarted} bytes)" + ); + client.deprovision(id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "helper process for whp_guest_survives_its_caller_and_a_killed_start_leaves_no_vm"] +fn helper_starts_a_sandbox_in_another_process() { + let Some(sandbox) = std::env::var_os(HELPER_SANDBOX) else { + return; + }; + let client = AciEdgeSandbox::from_shared(backend(&required(HELPER_STATE))); + let id = SandboxId::parse(&sandbox.to_string_lossy()).unwrap(); + client.start(&id).unwrap(); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_guest_survives_its_caller_and_a_killed_start_leaves_no_vm() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend_with(state.path(), false, |config| { + config.stop_timeout = Duration::from_secs(5); + }); + let client = AciEdgeSandbox::from_shared(backend.clone()); + + let sandbox = provision(&client, &backend); + let status = start_in_helper(state.path(), &sandbox.id).wait().unwrap(); + assert!( + status.success(), + "the helper process could not start the guest: {status}" + ); + assert_prints(&client, &sandbox.id, "survived"); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.id).unwrap(); + drop(sandbox); + assert_no_new_openvmm(&before); + + for delay in [0, 100, 300, 700, 1500] { + let sandbox = provision(&client, &backend); + let mut helper = start_in_helper(state.path(), &sandbox.id); + thread::sleep(Duration::from_millis(delay)); + let _ = helper.kill(); + let _ = helper.wait(); + match client.stop(&sandbox.id) { + Ok(_) => {} + Err(error) if error.code() == ErrorCode::AlreadyStopped => {} + Err(error) => panic!("stop after a start killed at {delay} ms failed: {error}"), + } + client.deprovision(&sandbox.id).unwrap_or_else(|error| { + panic!("deprovision after a start killed at {delay} ms failed: {error}") + }); + drop(sandbox); + assert_no_new_openvmm(&before); + } +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_parallel_sandboxes_and_repeated_restarts_stay_isolated() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + + thread::scope(|scope| { + let workers: Vec<_> = (0..4) + .map(|index| { + let (client, backend) = (&client, &backend); + scope.spawn(move || { + let sandbox = provision(client, backend); + client.start(&sandbox.id).unwrap(); + assert_prints(client, &sandbox.id, &format!("sandbox{index}")); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.id).unwrap(); + }) + }) + .collect(); + for worker in workers { + worker.join().unwrap(); + } + }); + assert_no_new_openvmm(&before); + + let sandbox = provision(&client, &backend); + let mut starts = Vec::new(); + for cycle in 0..8 { + let called = Instant::now(); + let started = client.start(&sandbox.id).unwrap(); + let total = called.elapsed().as_millis(); + let boot = started + .metadata + .as_ref() + .and_then(|metadata| metadata.get("bootMilliseconds")) + .and_then(serde_json::Value::as_u64) + .unwrap(); + starts.push((total, boot)); + assert_prints(&client, &sandbox.id, &format!("cycle{cycle}")); + assert_graceful(client.stop(&sandbox.id)); + } + client.deprovision(&sandbox.id).unwrap(); + eprintln!("(start call, boot) milliseconds across restarts: {starts:?}"); + assert!( + starts + .iter() + .all(|&(total, boot)| total.saturating_sub(u128::from(boot)) + < MAX_START_OVERHEAD.as_millis()), + "starts spent more than {MAX_START_OVERHEAD:?} outside the guest boot: {starts:?}" + ); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_concurrent_commands_share_one_guest() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let backend = backend(state.path()); + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + client.start(&sandbox.id).unwrap(); + + let started = Instant::now(); + thread::scope(|scope| { + let workers: Vec<_> = (0..4) + .map(|index| { + let (client, id) = (&client, &sandbox.id); + scope.spawn(move || { + let output = exec( + client, + id, + &ExecRequest::command_line(format!("sleep 1; printf parallel{index}")), + ); + assert_eq!(output.outcome, ExecOutcome::Exited(0), "{output:?}"); + assert_eq!( + String::from_utf8_lossy(&output.stdout), + format!("parallel{index}") + ); + }) + }) + .collect(); + for worker in workers { + worker.join().unwrap(); + } + }); + eprintln!( + "four concurrent one-second commands took {:?}", + started.elapsed() + ); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.id).unwrap(); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_registered_images_start_without_hashing_and_fail_closed_after_a_change() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let image = copied_image(state.path()); + let created = Instant::now(); + let first = NvxHostBackend::new(native_config(state.path(), &image)).unwrap(); + let hashed = created.elapsed(); + let id = first.image_id(); + let created = Instant::now(); + let backend = Arc::new(NvxHostBackend::new(native_config(state.path(), &image)).unwrap()); + let reused = created.elapsed(); + eprintln!("backend creation: {hashed:?} registering the image, {reused:?} reusing it"); + assert_eq!(backend.image_id(), id); + assert_eq!(backend.runtime_digests(), first.runtime_digests()); + let images = backend.images().unwrap(); + assert_eq!(images.len(), 1, "{images:?}"); + assert_eq!( + ( + images[0].id, + &images[0].path, + images[0].intact, + images[0].trusted + ), + (id, &image, true, false) + ); + + let client = AciEdgeSandbox::from_shared(backend.clone()); + let sandbox = provision(&client, &backend); + let overhead = start_overhead(&client, &sandbox.id); + eprintln!("the first start spent {overhead:?} outside the guest boot"); + assert_prints(&client, &sandbox.id, "registered"); + assert_graceful(client.stop(&sandbox.id)); + + // A changed seal fails closed, without hashing, until the image is registered again. + touch(&image); + assert_unavailable( + client.start(&sandbox.id), + "changed since it was registered; register it again", + ); + assert_unavailable(client.provision(&ProvisionRequest::new()), "changed since"); + assert!(!backend.images().unwrap()[0].intact); + assert_eq!( + failure(backend.unregister_image(&id)), + ErrorCode::PolicyValidation + ); + let registered = backend + .register_image(&image, ImageDigest::Compute) + .unwrap(); + assert_eq!((registered.id, registered.intact), (id, true)); + client.start(&sandbox.id).unwrap(); + assert_prints(&client, &sandbox.id, "reregistered"); + assert_graceful(client.stop(&sandbox.id)); + backend.verify_image(&id).unwrap(); + + client.deprovision(&sandbox.release()).unwrap(); + assert!(backend.unregister_image(&id).unwrap()); + assert!(!backend.unregister_image(&id).unwrap()); + assert!(backend.images().unwrap().is_empty()); + assert_unavailable( + client.provision(&ProvisionRequest::new()), + "no longer registered", + ); + assert!(image.is_file()); + assert_no_new_openvmm(&before); +} + +#[test] +#[ignore = "requires an approved private DLL, edge initramfs, GPT image, and a WHP host"] +fn whp_trusted_digests_and_content_verification_follow_their_contracts() { + let state = tempfile::tempdir().unwrap(); + let before = openvmm_processes(); + let image = copied_image(state.path()); + + // A trusted digest is recorded without hashing, even one that the content does not have. + let claimed = ImageId::from_sha256([7; 32]); + let trusting = NvxHostBackend::new( + native_config(state.path(), &image) + .with_image_digest(ImageDigest::Trusted(*claimed.sha256())), + ) + .unwrap(); + assert_eq!(trusting.image_id(), claimed); + assert!(trusting.images().unwrap()[0].trusted); + assert_unavailable( + trusting.verify_image(&claimed), + "no longer matches its SHA-256", + ); + + // Content verification hashes before every start, so it refuses the misattributed image. + let verifying = Arc::new( + NvxHostBackend::new(native_config(state.path(), &image).with_content_verification(true)) + .unwrap(), + ); + assert_eq!(verifying.image_id(), claimed); + let client = AciEdgeSandbox::from_shared(verifying.clone()); + let sandbox = provision(&client, &verifying); + assert_unavailable(client.start(&sandbox.id), "no longer matches its SHA-256"); + + // Runtime files are checked against approved digests and pinned per sandbox. + let digests = verifying.runtime_digests(); + NvxHostBackend::new(native_config(state.path(), &image).with_runtime_digests(digests)).unwrap(); + let mut unapproved = digests; + unapproved.initrd = [0; 32]; + assert_unavailable( + NvxHostBackend::new(native_config(state.path(), &image).with_runtime_digests(unapproved)), + "does not match the approved SHA-256", + ); + let initrd = state.path().join("initramfs.cpio.gz"); + let mut changed = std::fs::read(required("NVXHOST_TEST_INITRD")).unwrap(); + changed.push(0); + std::fs::write(&initrd, changed).unwrap(); + let mut config = native_config(state.path(), &image); + config.openvmm.initrd = initrd; + let other = AciEdgeSandbox::from_shared(Arc::new(NvxHostBackend::new(config).unwrap())); + assert_unavailable( + other.start(&sandbox.id), + "provisioned with different OpenVMM runtime files", + ); + client.deprovision(&sandbox.release()).unwrap(); + + // Once the image is registered by its actual digest, a verified start succeeds. + assert!(verifying.unregister_image(&claimed).unwrap()); + let actual = verifying + .register_image(&image, ImageDigest::Compute) + .unwrap() + .id; + assert_ne!(actual, claimed); + let verifying = Arc::new( + NvxHostBackend::new(native_config(state.path(), &image).with_content_verification(true)) + .unwrap(), + ); + assert_eq!(verifying.image_id(), actual); + let client = AciEdgeSandbox::from_shared(verifying.clone()); + let sandbox = provision(&client, &verifying); + let overhead = start_overhead(&client, &sandbox.id); + eprintln!("a start with content verification spent {overhead:?} outside the guest boot"); + assert_prints(&client, &sandbox.id, "verified"); + assert_graceful(client.stop(&sandbox.id)); + client.deprovision(&sandbox.release()).unwrap(); + assert_no_new_openvmm(&before); +} diff --git a/openvmm b/openvmm index 4355c010e..952406552 160000 --- a/openvmm +++ b/openvmm @@ -1 +1 @@ -Subproject commit 4355c010e726128c751138e7f74ad969e76a4c19 +Subproject commit 9524065522764c8b80b3a4d48ec7877ddc16090b