From 505b81df1d79ce3f45feb2b14e391b551d8bbb03 Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 29 Sep 2026 22:52:49 +0800 Subject: [PATCH 1/9] Add shared OCI image cache with lazy VM loading Expose the server's OCI image store over a versioned read-only file protocol (`pvisor cache serve`, plus prepare/list/stat/read commands) and mount demand-filled FUSE lowers for VM rootfs images. Clients probe the default per-user Unix socket automatically and fall back to local OCI preparation when no compatible server is running. Also add a `warn` filesystem access level, `--ask` file approval UI and persistent permission rules, a VM filesystem smoke test, and fix macOS virtio-fs passthrough to pin inode paths for FSKit mounts. --- Cargo.lock | 2 + Cargo.toml | 1 + crates/persisting-overlaynet/src/vm.rs | 100 +- crates/persisting-pvisor/Cargo.toml | 6 + crates/persisting-pvisor/src/cache.rs | 1047 +++++++++++++++++ crates/persisting-pvisor/src/cache/lazy.rs | 870 ++++++++++++++ .../persisting-pvisor/src/cache/metadata.rs | 72 ++ .../persisting-pvisor/src/cache/progress.rs | 232 ++++ crates/persisting-pvisor/src/cli/mod.rs | 22 +- crates/persisting-pvisor/src/cli/run.rs | 299 +++-- .../src/cli/run/safe/files.rs | 2 +- crates/persisting-pvisor/src/cli/tui.rs | 4 +- .../src/cli/tui/zellij/audit_ui.rs | 613 +++++++++- .../src/cli/tui/zellij/input.rs | 142 ++- .../src/cli/tui/zellij/runtime.rs | 242 +++- .../src/cli/tui/zellij/status_bar.rs | 152 ++- .../src/cli/tui/zellij/view.rs | 520 ++++++-- crates/persisting-pvisor/src/config.rs | 1 + crates/persisting-pvisor/src/lib.rs | 2 + crates/persisting-pvisor/src/oci.rs | 60 +- crates/persisting-pvisor/src/process.rs | 48 + crates/persisting-pvisor/src/vm/supported.rs | 2 + crates/persisting-pvisor/tests/macos_safe.rs | 112 ++ docs/shared-image-cache.md | 205 ++++ docs/src/en/reference/cli.md | 104 +- docs/src/zh/reference/cases.md | 38 +- docs/src/zh/reference/cli.md | 91 +- scripts/ci/vm_fs_smoke.py | 218 ++++ .../src/virtio/fs/macos/passthrough.rs | 81 +- vendor/krun-devices/src/virtio/fs/overlay.rs | 179 ++- 30 files changed, 4995 insertions(+), 472 deletions(-) create mode 100644 crates/persisting-pvisor/src/cache.rs create mode 100644 crates/persisting-pvisor/src/cache/lazy.rs create mode 100644 crates/persisting-pvisor/src/cache/metadata.rs create mode 100644 crates/persisting-pvisor/src/cache/progress.rs create mode 100644 docs/shared-image-cache.md create mode 100644 scripts/ci/vm_fs_smoke.py diff --git a/Cargo.lock b/Cargo.lock index a6a72f62..021a9abd 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2983,6 +2983,7 @@ dependencies = [ "dirs", "flate2", "fs2", + "fuser", "globset", "libc", "libkrun", @@ -3004,6 +3005,7 @@ dependencies = [ "tokio", "tokio-util", "toml", + "toml_edit 0.22.27", "tracing", "unicode-width", "uuid", diff --git a/Cargo.toml b/Cargo.toml index 18a748ac..e1fb80a8 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -72,6 +72,7 @@ tokio = { version = "1", default-features = false } tokio-stream = "0.1" tokio-util = { version = "0.7", default-features = false } toml = "0.8" +toml_edit = "0.22" tracing = "0.1" url = "2" uuid = "1" diff --git a/crates/persisting-overlaynet/src/vm.rs b/crates/persisting-overlaynet/src/vm.rs index bd39c060..7ef50441 100644 --- a/crates/persisting-overlaynet/src/vm.rs +++ b/crates/persisting-overlaynet/src/vm.rs @@ -9,6 +9,7 @@ use std::collections::{HashMap, VecDeque}; use std::io; use std::net::{IpAddr, Ipv4Addr, SocketAddr}; use std::os::unix::net::UnixStream as StdUnixStream; +use std::sync::Arc; use std::thread; use std::time::Duration as StdDuration; @@ -26,7 +27,7 @@ use smoltcp::wire::{ }; use tokio::io::{AsyncReadExt, AsyncWriteExt}; use tokio::net::{TcpStream, UnixStream}; -use tokio::sync::{mpsc, oneshot}; +use tokio::sync::{OwnedSemaphorePermit, Semaphore, mpsc, oneshot}; use tokio::task::JoinHandle; use crate::egress::{ @@ -188,7 +189,7 @@ struct Flow { phase: FlowPhase, upstream: Option>>, upstream_task: Option>, - inbound: VecDeque>, + inbound: VecDeque<(Vec, Option)>, inbound_offset: usize, dns_input: Vec, remote_eof: bool, @@ -203,6 +204,7 @@ enum FlowEvent { Data { key: FlowKey, bytes: Vec, + permit: OwnedSemaphorePermit, }, Uploaded { key: FlowKey, @@ -561,7 +563,7 @@ fn drive_dns_tcp( let mut framed = Vec::with_capacity(response.len() + 2); framed.extend_from_slice(&(response.len() as u16).to_be_bytes()); framed.extend_from_slice(&response); - flow.inbound.push_back(framed); + flow.inbound.push_back((framed, None)); } flush_inbound(socket, flow); if !socket.may_recv() && flow.inbound.is_empty() && socket.may_send() { @@ -571,7 +573,7 @@ fn drive_dns_tcp( fn flush_inbound(socket: &mut tcp::Socket<'_>, flow: &mut Flow) { while socket.can_send() { - let Some(front) = flow.inbound.front() else { + let Some((front, _permit)) = flow.inbound.front() else { break; }; match socket.send_slice(&front[flow.inbound_offset..]) { @@ -702,7 +704,11 @@ async fn bridge_upstream( }; let download = async { let mut buffer = vec![0; 16 * 1024]; + let credits = Arc::new(Semaphore::new(FLOW_BUFFER_CHUNKS)); loop { + // Bound each flow across both the shared event channel and inbound + // queue. Reading resumes only after the guest socket accepts data. + let permit = credits.clone().acquire_owned().await.unwrap(); let length = read.read(&mut buffer).await?; if length == 0 { let _ = events.send(FlowEvent::RemoteEof(key)).await; @@ -713,6 +719,7 @@ async fn bridge_upstream( .send(FlowEvent::Data { key, bytes: buffer[..length].to_vec(), + permit, }) .await .is_err() @@ -758,21 +765,10 @@ fn apply_flow_event( } } } - FlowEvent::Data { key, bytes } => { + FlowEvent::Data { key, bytes, permit } => { if let Some(flow) = flows.get_mut(&key) { - let queued = flow - .inbound - .iter() - .map(Vec::len) - .sum::() - .saturating_sub(flow.inbound_offset); - if queued.saturating_add(bytes.len()) > TCP_BUFFER_BYTES { - sockets.get_mut::(flow.handle).abort(); - metrics.tcp_connect_failure(); - } else { - metrics.host_to_guest(bytes.len()); - flow.inbound.push_back(bytes); - } + metrics.host_to_guest(bytes.len()); + flow.inbound.push_back((bytes, Some(permit))); } } FlowEvent::Uploaded { key, bytes } => { @@ -1244,6 +1240,74 @@ impl TxToken for FrameTxToken<'_> { mod tests { use super::*; + #[tokio::test] + async fn slow_guest_backpressures_large_download_without_losing_data() { + tokio::time::timeout(StdDuration::from_secs(10), async { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let stream = TcpStream::connect(listener.local_addr().unwrap()) + .await + .unwrap(); + let (mut server, _) = listener.accept().await.unwrap(); + let payload: Vec = (0..1024 * 1024).map(|i| (i % 251) as u8).collect(); + let expected = payload.clone(); + let server_task = tokio::spawn(async move { + server.write_all(&payload).await.unwrap(); + server.shutdown().await.unwrap(); + }); + let key = FlowKey { + guest_port: 49152, + destination_addr: Ipv4Addr::new(198, 18, 0, 1), + destination_port: 80, + }; + let (sender, outbound) = mpsc::channel(FLOW_BUFFER_CHUNKS); + let (events, mut received) = mpsc::channel(TCP_CHANNEL_DEPTH); + let bridge = tokio::spawn(bridge_upstream( + key, + stream, + outbound, + Default::default(), + events, + )); + drop(sender); + let mut bytes = Vec::new(); + let mut held = Vec::new(); + for _ in 0..FLOW_BUFFER_CHUNKS { + match received.recv().await.unwrap() { + FlowEvent::Data { + bytes: chunk, + permit, + .. + } => { + bytes.extend(chunk); + held.push(permit); + } + _ => panic!("expected download data"), + } + } + assert!(bytes.len() <= TCP_BUFFER_BYTES); + // A stalled guest must stop this flow's reader, not abort the TCP + // connection or keep filling the shared event channel. + assert!( + tokio::time::timeout(StdDuration::from_millis(50), received.recv()) + .await + .is_err() + ); + drop(held); + loop { + match received.recv().await.unwrap() { + FlowEvent::Data { bytes: chunk, .. } => bytes.extend(chunk), + FlowEvent::RemoteEof(_) => break, + _ => panic!("download failed while guest resumed"), + } + } + assert_eq!(bytes, expected); + bridge.await.unwrap(); + server_task.await.unwrap(); + }) + .await + .expect("download stalled after guest resumed"); + } + #[test] fn denied_vm_flow_records_its_destination() { let key = FlowKey { diff --git a/crates/persisting-pvisor/Cargo.toml b/crates/persisting-pvisor/Cargo.toml index a4871b3b..1f1ed930 100644 --- a/crates/persisting-pvisor/Cargo.toml +++ b/crates/persisting-pvisor/Cargo.toml @@ -34,6 +34,7 @@ sha2.workspace = true tar.workspace = true thiserror.workspace = true toml.workspace = true +toml_edit.workspace = true tempfile.workspace = true tokio = { workspace = true, features = ["io-util", "macros", "process", "rt-multi-thread", "signal", "sync", "time", "net"] } tokio-util = { workspace = true, features = ["rt"] } @@ -56,3 +57,8 @@ proptest.workspace = true [build-dependencies] libloading.workspace = true +[target.'cfg(target_os = "macos")'.dependencies] +fuser = { workspace = true, features = ["abi-7-19", "libfuse", "macfuse-5"] } + +[target.'cfg(target_os = "linux")'.dependencies] +fuser = { workspace = true, default-features = false, features = ["abi-7-31"] } diff --git a/crates/persisting-pvisor/src/cache.rs b/crates/persisting-pvisor/src/cache.rs new file mode 100644 index 00000000..1e5410a6 --- /dev/null +++ b/crates/persisting-pvisor/src/cache.rs @@ -0,0 +1,1047 @@ +//! Versioned, read-only OCI file service. See docs/shared-image-cache.md. +use anyhow::{Context, bail, ensure}; +use clap::{Args, Subcommand}; +use fs2::FileExt; +use serde::{Deserialize, Serialize, de::DeserializeOwned}; +use sha2::{Digest, Sha256}; +use std::ffi::{CStr, CString, OsStr}; +use std::fs::{self, File, OpenOptions}; +use std::io::{Read, Seek, SeekFrom, Write}; +use std::net::{SocketAddr, TcpListener, TcpStream}; +use std::os::fd::{AsRawFd, FromRawFd, IntoRawFd}; +use std::os::unix::ffi::OsStrExt; +use std::os::unix::fs::{FileTypeExt, OpenOptionsExt, PermissionsExt}; +use std::os::unix::net::{UnixListener, UnixStream}; +use std::path::{Component, Path, PathBuf}; +use std::sync::{Arc, Mutex, mpsc}; +use std::time::Duration; + +use crate::oci::ImageStore; + +mod metadata; +pub(crate) mod progress; +pub use progress::ImageTotals; + +#[cfg(any(target_os = "macos", target_os = "linux"))] +mod lazy; +#[cfg(any(target_os = "macos", target_os = "linux"))] +pub(crate) use lazy::{LazyMount, prepare_image}; + +pub const SERVER_ENV: &str = "PERSISTING_PVISOR_CACHE_SERVER"; +const TOKEN_ENV: &str = "PERSISTING_PVISOR_CACHE_TOKEN"; +const MAX_FRAME: usize = 1024 * 1024; +pub const MAX_READ: u32 = 1024 * 1024; +const TIMEOUT: Duration = Duration::from_secs(300); + +#[derive(Debug, Args)] +pub struct CacheArgs { + #[command(subcommand)] + command: CacheCommand, +} + +#[derive(Debug, Subcommand)] +enum CacheCommand { + /// Serve cached OCI files (foreground; Unix socket by default). + Serve { + /// unix:///absolute/path or tcp://127.0.0.1:PORT. Defaults to CACHE_SERVER. + #[arg(long)] + listen: Option, + /// OCI cache to serve and populate. + #[arg(long, env = "PERSISTING_PVISOR_IMAGE_STORE")] + image_store: Option, + }, + /// Resolve and prepare an image on the server; print its immutable digest. + Prepare { image: String }, + /// List one directory page. Paths are relative to the image root. + List { + digest: String, + path: Option, + #[arg(long, default_value_t = 0)] + offset: usize, + }, + /// Show file attributes without following symlinks. + Stat { digest: String, path: PathBuf }, + /// Stream one regular file to stdout. Does not follow symlinks. + Read { digest: String, path: PathBuf }, +} + +/// One request per connection. All paths are Unix bytes, relative to image root. +#[derive(Debug, Serialize, Deserialize)] +#[serde(tag = "op", rename_all = "snake_case", deny_unknown_fields)] +pub enum Request { + Ping, + Prepare { + image: String, + architecture: String, + }, + List { + digest: String, + path: Vec, + offset: usize, + }, + Stat { + digest: String, + path: Vec, + }, + Read { + digest: String, + path: Vec, + offset: u64, + length: u32, + }, +} + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct Envelope { + version: u32, + token: Option, + request: Request, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(tag = "status", rename_all = "snake_case")] +pub enum Response { + Ready, + Prepared { + #[serde(default)] + metadata_generation: Option, + #[serde(default)] + totals: Option, + digest: String, + architecture: String, + env: std::collections::BTreeMap, + entrypoint: Vec, + cmd: Vec, + }, + Entries { + names: Vec>, + next_offset: Option, + }, + Metadata { + kind: String, + size: u64, + mode: u32, + uid: u32, + gid: u32, + inode: u64, + nlink: u64, + mtime: i64, + mtime_nsec: i64, + target: Option>, + }, + Data { + length: u32, + sha256: String, + }, + Error { + code: String, + message: String, + }, +} + +pub fn default_endpoint() -> anyhow::Result { + let base = dirs::cache_dir().context("cannot find user cache directory")?; + Ok(format!( + "unix://{}", + base.join("persisting/pvisor/cache.sock").display() + )) +} + +fn endpoint_from_env() -> anyhow::Result { + match std::env::var(SERVER_ENV) { + Ok(value) => Ok(value), + Err(std::env::VarError::NotPresent) => default_endpoint(), + Err(error) => Err(error.into()), + } +} + +fn architecture() -> &'static str { + match std::env::consts::ARCH { + "aarch64" => "arm64", + "x86_64" => "amd64", + other => other, + } +} + +enum Endpoint { + Unix(PathBuf), + Tcp(SocketAddr), +} + +fn endpoint(value: &str) -> anyhow::Result { + if let Some(path) = value.strip_prefix("unix://") { + ensure!( + Path::new(path).is_absolute(), + "Unix socket path must be absolute" + ); + return Ok(Endpoint::Unix(path.into())); + } + if let Some(address) = value.strip_prefix("tcp://") { + let address: SocketAddr = address + .parse() + .context("TCP endpoint requires an IP address and port")?; + ensure!( + address.ip().is_loopback(), + "cache TCP is loopback-only; use an SSH tunnel for remote access" + ); + return Ok(Endpoint::Tcp(address)); + } + bail!("expected unix:///absolute/path or tcp://127.0.0.1:PORT") +} + +enum Stream { + Unix(UnixStream), + Tcp(TcpStream), +} + +impl Read for Stream { + fn read(&mut self, buf: &mut [u8]) -> std::io::Result { + match self { + Self::Unix(s) => s.read(buf), + Self::Tcp(s) => s.read(buf), + } + } +} +impl Write for Stream { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + match self { + Self::Unix(s) => s.write(buf), + Self::Tcp(s) => s.write(buf), + } + } + fn flush(&mut self) -> std::io::Result<()> { + match self { + Self::Unix(s) => s.flush(), + Self::Tcp(s) => s.flush(), + } + } +} +impl Stream { + fn timeouts(&self, timeout: Duration) -> std::io::Result<()> { + match self { + Self::Unix(s) => { + s.set_read_timeout(Some(timeout))?; + s.set_write_timeout(Some(timeout)) + } + Self::Tcp(s) => { + s.set_read_timeout(Some(timeout))?; + s.set_write_timeout(Some(timeout)) + } + } + } +} + +fn read_frame(stream: &mut impl Read) -> anyhow::Result { + let mut header = [0; 4]; + stream.read_exact(&mut header)?; + let length = u32::from_be_bytes(header) as usize; + ensure!( + length > 0 && length <= MAX_FRAME, + "invalid cache frame length" + ); + let mut bytes = vec![0; length]; + stream.read_exact(&mut bytes)?; + Ok(serde_json::from_slice(&bytes)?) +} +fn write_frame(stream: &mut impl Write, value: &impl Serialize) -> anyhow::Result<()> { + let bytes = serde_json::to_vec(value)?; + ensure!( + bytes.len() <= MAX_FRAME, + "cache response exceeds frame limit" + ); + stream.write_all(&(bytes.len() as u32).to_be_bytes())?; + stream.write_all(&bytes)?; + Ok(()) +} + +#[derive(Debug, thiserror::Error)] +#[error("cache connection failed: {0}")] +struct CacheConnectError(#[source] std::io::Error); + +/// Blocking client; call from the host side, outside filesystem operation locks. +pub struct CacheClient { + endpoint: String, + token: Option, +} +impl CacheClient { + pub fn from_env() -> anyhow::Result { + Self::new(endpoint_from_env()?, std::env::var(TOKEN_ENV).ok()) + } + pub fn new(address: String, token: Option) -> anyhow::Result { + if matches!(endpoint(&address)?, Endpoint::Tcp(_)) { + ensure!( + token.as_ref().is_some_and(|s| !s.is_empty()), + "TCP requires {TOKEN_ENV}" + ); + } + Ok(Self { + endpoint: address, + token, + }) + } + /// Discover the default socket, or require an explicitly configured service. + pub(crate) fn discover() -> anyhow::Result> { + let explicit = match std::env::var(SERVER_ENV) { + Ok(value) => Some(value), + Err(std::env::VarError::NotPresent) => None, + Err(error) => return Err(error.into()), + }; + if explicit.as_deref() == Some("off") { + return Ok(None); + } + let address = explicit.clone().map_or_else(default_endpoint, Ok)?; + Self::probe(address, std::env::var(TOKEN_ENV).ok(), explicit.is_some()) + } + + fn probe( + address: String, + token: Option, + explicit: bool, + ) -> anyhow::Result> { + let client = Self::new(address, token)?; + match client.request_timeout(Request::Ping, Duration::from_secs(2)) { + Ok((Response::Ready, _)) => Ok(Some(client)), + Ok(_) => bail!("cache server returned an incompatible handshake"), + Err(error) => { + let absent = error.downcast_ref::().is_some_and(|e| { + matches!( + e.0.kind(), + std::io::ErrorKind::NotFound | std::io::ErrorKind::ConnectionRefused + ) + }); + if !explicit && absent { + Ok(None) + } else { + Err(error.context("probe shared image cache")) + } + } + } + } + + /// The returned bytes are present only for `Read`, and are SHA-256 checked. + pub fn request(&self, request: Request) -> anyhow::Result<(Response, Vec)> { + self.request_timeout(request, TIMEOUT) + } + + fn request_timeout( + &self, + request: Request, + timeout: Duration, + ) -> anyhow::Result<(Response, Vec)> { + let expected = match &request { + Request::Read { length, .. } => Some(*length), + _ => None, + }; + let mut stream = match endpoint(&self.endpoint)? { + Endpoint::Unix(path) => Stream::Unix( + UnixStream::connect(path) + .map_err(CacheConnectError) + .with_context(|| { + format!( + "connect cache {}; start `pvisor cache serve`", + self.endpoint + ) + })?, + ), + Endpoint::Tcp(address) => Stream::Tcp( + TcpStream::connect_timeout(&address, Duration::from_secs(10)) + .map_err(CacheConnectError)?, + ), + }; + stream.timeouts(timeout)?; + write_frame( + &mut stream, + &Envelope { + version: 1, + token: self.token.clone(), + request, + }, + )?; + let response: Response = read_frame(&mut stream)?; + let mut body = Vec::new(); + match &response { + Response::Error { code, message } => { + let kind = match code.as_str() { + "not_found" => std::io::ErrorKind::NotFound, + "permission_denied" => std::io::ErrorKind::PermissionDenied, + _ => std::io::ErrorKind::Other, + }; + return Err(std::io::Error::new(kind, format!("cache {code}: {message}")).into()); + } + Response::Data { length, sha256 } => { + ensure!( + expected.is_some_and(|limit| *length <= limit) && *length <= MAX_READ, + "invalid cache data length" + ); + body.resize(*length as usize, 0); + stream.read_exact(&mut body)?; + ensure!(hash(&body) == *sha256, "cache data digest mismatch"); + } + _ => ensure!(expected.is_none(), "expected cache data response"), + } + Ok((response, body)) + } +} +fn hash(bytes: &[u8]) -> String { + format!("sha256:{}", crate::oci::encode_hex(&Sha256::digest(bytes))) +} + +// Every component is opened relative to its parent fd, without following links. +// This remains confined even if a directory is renamed during a request. +fn open_child(parent: &File, name: &OsStr, directory: bool) -> anyhow::Result { + let name = CString::new(name.as_bytes())?; + let flags = libc::O_RDONLY + | libc::O_CLOEXEC + | libc::O_NOFOLLOW + | libc::O_NONBLOCK + | if directory { libc::O_DIRECTORY } else { 0 }; + let fd = unsafe { libc::openat(parent.as_raw_fd(), name.as_ptr(), flags) }; + if fd < 0 { + return Err(std::io::Error::last_os_error().into()); + } + Ok(unsafe { File::from_raw_fd(fd) }) +} + +fn parent(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result<(File, Vec)> { + let digest = crate::oci::digest_hex(digest)?; + ensure!(!path.contains(&0), "NUL in cache path"); + let path = Path::new(OsStr::from_bytes(path)); + let components: Vec<_> = path.components().collect(); + ensure!( + components.iter().all(|c| matches!(c, Component::Normal(_))), + "cache path must be relative without dot or parent components" + ); + let roots = OpenOptions::new() + .read(true) + .custom_flags(libc::O_DIRECTORY | libc::O_NOFOLLOW) + .open(store.root.join("rootfs-v3/sha256"))?; + let mut directory = open_child(&roots, OsStr::new(digest), true)?; + for component in components.iter().take(components.len().saturating_sub(1)) { + directory = open_child(&directory, component.as_os_str(), true)?; + } + let name = components + .last() + .map_or_else(|| b".".to_vec(), |c| c.as_os_str().as_bytes().to_vec()); + Ok((directory, name)) +} + +fn directory_names(directory: File) -> anyhow::Result>> { + let raw = unsafe { libc::fdopendir(directory.as_raw_fd()) }; + if raw.is_null() { + return Err(std::io::Error::last_os_error().into()); + } + let _ = directory.into_raw_fd(); // fdopendir owns the descriptor on success. + struct Directory(*mut libc::DIR); + impl Drop for Directory { + fn drop(&mut self) { + unsafe { + libc::closedir(self.0); + } + } + } + let directory = Directory(raw); + let mut names = Vec::new(); + loop { + #[cfg(target_os = "macos")] + unsafe { + *libc::__error() = 0; + } + #[cfg(target_os = "linux")] + unsafe { + *libc::__errno_location() = 0; + } + let entry = unsafe { libc::readdir(directory.0) }; + if entry.is_null() { + let error = std::io::Error::last_os_error(); + if error.raw_os_error() != Some(0) { + return Err(error.into()); + } + break; + } + let name = unsafe { CStr::from_ptr((*entry).d_name.as_ptr()) }.to_bytes(); + if name != b"." && name != b".." { + names.push(name.to_vec()); + } + } + Ok(names) +} + +#[allow(clippy::unnecessary_cast)] // libc stat field widths differ by platform. +fn metadata_at(directory: &File, name: &[u8]) -> anyhow::Result { + let name = CString::new(name)?; + let mut m: libc::stat = unsafe { std::mem::zeroed() }; + if unsafe { + libc::fstatat( + directory.as_raw_fd(), + name.as_ptr(), + &mut m, + libc::AT_SYMLINK_NOFOLLOW, + ) + } != 0 + { + return Err(std::io::Error::last_os_error().into()); + } + let kind = match m.st_mode & libc::S_IFMT { + libc::S_IFREG => "file", + libc::S_IFDIR => "directory", + libc::S_IFLNK => "symlink", + _ => "special", + }; + let target = if kind == "symlink" { + let mut bytes = vec![0u8; 4096]; + let size = unsafe { + libc::readlinkat( + directory.as_raw_fd(), + name.as_ptr(), + bytes.as_mut_ptr().cast(), + bytes.len(), + ) + }; + if size < 0 { + return Err(std::io::Error::last_os_error().into()); + } + ensure!( + (size as usize) < bytes.len(), + "symlink target exceeds protocol limit" + ); + bytes.truncate(size as usize); + Some(bytes) + } else { + None + }; + Ok(Response::Metadata { + kind: kind.into(), + size: m.st_size as u64, + mode: m.st_mode as u32, + uid: m.st_uid, + gid: m.st_gid, + inode: m.st_ino as u64, + nlink: m.st_nlink as u64, + mtime: m.st_mtime as i64, + mtime_nsec: m.st_mtime_nsec as i64, + target, + }) +} + +fn handle(store: &ImageStore, request: Request) -> anyhow::Result<(Response, Vec)> { + let response = match request { + Request::Ping => Response::Ready, + Request::Prepare { + image, + architecture: requested, + } => { + let image = store.prepare_for_architecture(&image, &requested)?; + Response::Prepared { + metadata_generation: Some(metadata::generation(store, &image.digest)?), + totals: Some(progress::image_totals(store, &image.digest)?), + digest: image.digest, + architecture: requested, + env: image.env, + entrypoint: image.entrypoint, + cmd: image.cmd, + } + } + Request::List { + digest, + path, + offset, + } => { + let names = metadata::directory(store, &digest, &path)?; + ensure!(offset <= names.len(), "directory offset out of range"); + let end = offset.saturating_add(256).min(names.len()); + Response::Entries { + names: names[offset..end].to_vec(), + next_offset: (end < names.len()).then_some(end), + } + } + Request::Stat { digest, path } => metadata::stat(store, &digest, &path)?, + Request::Read { + digest, + path, + offset, + length, + } => { + ensure!( + length > 0 && length <= MAX_READ, + "read length must be 1..={MAX_READ}" + ); + let (directory, name) = parent(store, &digest, &path)?; + let mut file = open_child(&directory, OsStr::from_bytes(&name), false)?; + ensure!(file.metadata()?.is_file(), "only regular files can be read"); + file.seek(SeekFrom::Start(offset))?; + let mut body = Vec::new(); + file.take(length as u64).read_to_end(&mut body)?; + return Ok(( + Response::Data { + length: body.len() as u32, + sha256: hash(&body), + }, + body, + )); + } + }; + Ok((response, Vec::new())) +} + +fn ensure_same_user(socket: &UnixStream) -> anyhow::Result<()> { + #[cfg(target_os = "macos")] + let uid = { + let mut uid = 0; + let mut gid = 0; + if unsafe { libc::getpeereid(socket.as_raw_fd(), &mut uid, &mut gid) } != 0 { + return Err(std::io::Error::last_os_error().into()); + } + uid + }; + #[cfg(target_os = "linux")] + let uid = { + let mut credentials: libc::ucred = unsafe { std::mem::zeroed() }; + let mut length = std::mem::size_of_val(&credentials) as libc::socklen_t; + if unsafe { + libc::getsockopt( + socket.as_raw_fd(), + libc::SOL_SOCKET, + libc::SO_PEERCRED, + (&mut credentials as *mut libc::ucred).cast(), + &mut length, + ) + } != 0 + { + return Err(std::io::Error::last_os_error().into()); + } + credentials.uid + }; + #[cfg(not(any(target_os = "linux", target_os = "macos")))] + anyhow::bail!("cache peer authentication is supported only on Linux and macOS"); + #[cfg(any(target_os = "linux", target_os = "macos"))] + ensure!( + uid == unsafe { libc::geteuid() }, + "cache socket requires the same user" + ); + Ok(()) +} + +fn serve_connection( + mut stream: Stream, + store: &ImageStore, + token: Option<&str>, +) -> anyhow::Result<()> { + stream.timeouts(TIMEOUT)?; + if let Stream::Unix(socket) = &stream { + ensure_same_user(socket)?; + } + let result = (|| { + let envelope: Envelope = read_frame(&mut stream)?; + ensure!(envelope.version == 1, "unsupported cache protocol version"); + ensure!( + token.is_none() || envelope.token.as_deref() == token, + "cache authentication failed" + ); + handle(store, envelope.request) + })(); + let (response, body) = result.unwrap_or_else(|error: anyhow::Error| { + let code = match error.downcast_ref::().map(|e| e.kind()) { + Some(std::io::ErrorKind::NotFound) => "not_found", + Some(std::io::ErrorKind::PermissionDenied) => "permission_denied", + _ => "request_failed", + }; + ( + Response::Error { + code: code.into(), + message: format!("{error:#}"), + }, + Vec::new(), + ) + }); + write_frame(&mut stream, &response)?; + stream.write_all(&body)?; + Ok(()) +} + +fn serve(address: String, store: ImageStore, token: Option) -> anyhow::Result<()> { + let (send, receive) = mpsc::sync_channel::(16); + let receive = Arc::new(Mutex::new(receive)); + let store = Arc::new(store); + let token = Arc::new(token); + let dispatch = |stream| -> anyhow::Result<()> { + // A full queue closes the connection instead of allocating unbounded workers. + send.try_send(stream) + .map_err(|_| anyhow::anyhow!("cache server busy")) + }; + // Bind before starting workers so address conflicts fail without orphan workers. + enum Listener { + Unix(UnixListener, File), + Tcp(TcpListener), + } + let listener = match endpoint(&address)? { + Endpoint::Unix(path) => { + let directory = path.parent().context("socket requires parent directory")?; + fs::create_dir_all(directory)?; + let lock = OpenOptions::new() + .read(true) + .write(true) + .create(true) + .truncate(false) + .mode(0o600) + .custom_flags(libc::O_NOFOLLOW) + .open(path.with_extension("sock.lock"))?; + lock.try_lock_exclusive() + .context("cache server already running (socket lock held)")?; + match fs::symlink_metadata(&path) { + Ok(metadata) => { + ensure!( + metadata.file_type().is_socket(), + "refusing to replace non-socket path" + ); + match UnixStream::connect(&path) { + Ok(_) => bail!("cache socket already in use"), + Err(error) if error.kind() == std::io::ErrorKind::ConnectionRefused => { + fs::remove_file(&path)? + } + Err(error) => return Err(error.into()), + } + } + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Err(error) => return Err(error.into()), + } + let listener = UnixListener::bind(&path)?; + fs::set_permissions(&path, fs::Permissions::from_mode(0o600))?; + Listener::Unix(listener, lock) + } + Endpoint::Tcp(address) => { + ensure!( + token.as_ref().as_ref().is_some_and(|t| !t.is_empty()), + "TCP requires {TOKEN_ENV}" + ); + Listener::Tcp(TcpListener::bind(address)?) + } + }; + for _ in 0..16 { + let receive = receive.clone(); + let store = store.clone(); + let token = token.clone(); + std::thread::spawn(move || { + loop { + let request = receive.lock().unwrap().recv(); + let Ok(stream) = request else { break }; + if let Err(error) = serve_connection(stream, &store, token.as_deref()) { + eprintln!("cache connection: {error}"); + } + } + }); + } + eprintln!("pvisor cache listening on {address}"); + match listener { + Listener::Unix(listener, _lock) => { + for stream in listener.incoming() { + let _ = dispatch(Stream::Unix(stream?)); + } + } + Listener::Tcp(listener) => { + for stream in listener.incoming() { + let _ = dispatch(Stream::Tcp(stream?)); + } + } + } + Ok(()) +} + +pub fn run(args: CacheArgs) -> anyhow::Result<()> { + if let CacheCommand::Serve { + listen, + image_store, + } = args.command + { + return serve( + listen.map_or_else(endpoint_from_env, Ok)?, + ImageStore::new(image_store)?, + std::env::var(TOKEN_ENV).ok(), + ); + } + let client = CacheClient::from_env()?; + let request = match args.command { + CacheCommand::Prepare { image } => Request::Prepare { + image, + architecture: architecture().into(), + }, + CacheCommand::List { + digest, + path, + offset, + } => Request::List { + digest, + path: path.unwrap_or_default().as_os_str().as_bytes().to_vec(), + offset, + }, + CacheCommand::Stat { digest, path } => Request::Stat { + digest, + path: path.as_os_str().as_bytes().to_vec(), + }, + CacheCommand::Read { digest, path } => { + let mut offset = 0; + let mut stdout = std::io::stdout().lock(); + loop { + let (_, body) = client.request(Request::Read { + digest: digest.clone(), + path: path.as_os_str().as_bytes().to_vec(), + offset, + length: MAX_READ, + })?; + stdout.write_all(&body)?; + offset += body.len() as u64; + if body.len() < MAX_READ as usize { + break; + } + } + return Ok(()); + } + CacheCommand::Serve { .. } => unreachable!(), + }; + let (response, _) = client.request(request)?; + println!("{}", serde_json::to_string_pretty(&response)?); + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::os::unix::fs::symlink; + + fn fixture() -> (tempfile::TempDir, ImageStore, String) { + let tmp = tempfile::tempdir().unwrap(); + let store = ImageStore::new(Some(tmp.path().join("store"))).unwrap(); + let digest = format!("sha256:{}", "a".repeat(64)); + let root = store.root.join("rootfs-v3/sha256").join("a".repeat(64)); + fs::create_dir(&root).unwrap(); + fs::write(root.join("hello"), b"hello world").unwrap(); + fs::create_dir(root.join("dir")).unwrap(); + symlink("hello", root.join("alias")).unwrap(); + symlink("/etc", root.join("escape")).unwrap(); + (tmp, store, digest) + } + + #[test] + fn server_metadata_reuses_directory_index_and_invalidates_rebuilt_root() { + let (_tmp, store, digest) = fixture(); + let first = metadata::directory(&store, &digest, b"").unwrap(); + let again = metadata::directory(&store, &digest, b"").unwrap(); + assert!( + Arc::ptr_eq(&first, &again), + "directory must not be scanned again" + ); + let Response::Metadata { size, .. } = metadata::stat(&store, &digest, b"hello").unwrap() + else { + panic!() + }; + assert_eq!(size, 11); + let generation = metadata::generation(&store, &digest).unwrap(); + let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); + fs::rename(&root, root.with_extension("old")).unwrap(); + fs::create_dir(&root).unwrap(); + fs::write(root.join("hello"), b"new").unwrap(); + assert_ne!(generation, metadata::generation(&store, &digest).unwrap()); + assert_eq!( + &*metadata::directory(&store, &digest, b"").unwrap(), + &[b"hello".to_vec()] + ); + let Response::Metadata { size, .. } = metadata::stat(&store, &digest, b"hello").unwrap() + else { + panic!() + }; + assert_eq!(size, 3); + } + + #[test] + fn unix_client_roundtrip_and_parallel_reads() { + let (tmp, store, digest) = fixture(); + let path = tmp.path().join("server.sock"); + let listener = UnixListener::bind(&path).unwrap(); + let server = std::thread::spawn(move || { + std::thread::scope(|scope| { + for connection in listener.incoming().take(8) { + let store = &store; + scope.spawn(move || { + serve_connection(Stream::Unix(connection.unwrap()), store, None).unwrap() + }); + } + }); + }); + std::thread::scope(|scope| { + for _ in 0..8 { + let address = format!("unix://{}", path.display()); + let digest = digest.clone(); + scope.spawn(move || { + let client = CacheClient::new(address, None).unwrap(); + let (_, body) = client + .request(Request::Read { + digest, + path: b"hello".to_vec(), + offset: 6, + length: 5, + }) + .unwrap(); + assert_eq!(body, b"world"); + }); + } + }); + server.join().unwrap(); + } + + #[test] + fn paths_metadata_pagination_and_eof() { + let (_tmp, store, digest) = fixture(); + let (response, _) = handle( + &store, + Request::List { + digest: digest.clone(), + path: vec![], + offset: 0, + }, + ) + .unwrap(); + match response { + Response::Entries { names, next_offset } => { + assert_eq!( + names, + [ + b"alias".to_vec(), + b"dir".to_vec(), + b"escape".to_vec(), + b"hello".to_vec() + ] + ); + assert!(next_offset.is_none()); + } + _ => panic!("expected directory"), + } + let (response, _) = handle( + &store, + Request::Stat { + digest: digest.clone(), + path: b"alias".to_vec(), + }, + ) + .unwrap(); + assert!( + matches!(response, Response::Metadata { target: Some(target), .. } if target == b"hello") + ); + for path in [ + b"../hello".as_slice(), + b"/etc/passwd", + b"escape/passwd", + b"alias", + b"hello\0", + ] { + assert!( + handle( + &store, + Request::Read { + digest: digest.clone(), + path: path.to_vec(), + offset: 0, + length: 10 + } + ) + .is_err() + ); + } + assert!( + handle( + &store, + Request::Read { + digest: "sha256:../../etc".into(), + path: b"passwd".to_vec(), + offset: 0, + length: 10 + } + ) + .is_err() + ); + assert!( + handle( + &store, + Request::Read { + digest: digest.clone(), + path: b"hello".to_vec(), + offset: 0, + length: MAX_READ + 1 + } + ) + .is_err() + ); + let (_, body) = handle( + &store, + Request::Read { + digest, + path: b"hello".to_vec(), + offset: 100, + length: 10, + }, + ) + .unwrap(); + assert!(body.is_empty()); + } + + #[test] + fn rejects_bad_frames_versions_and_tokens() { + assert!(read_frame::(&mut &u32::MAX.to_be_bytes()[..]).is_err()); + assert!(endpoint("tcp://0.0.0.0:9000").is_err()); + assert!(CacheClient::new("tcp://127.0.0.1:9000".into(), None).is_err()); + for (version, token) in [(2, Some("secret")), (1, Some("wrong")), (1, None)] { + let (_tmp, store, digest) = fixture(); + let (mut client, server) = UnixStream::pair().unwrap(); + let worker = std::thread::spawn(move || { + serve_connection(Stream::Unix(server), &store, Some("secret")).unwrap() + }); + write_frame( + &mut client, + &Envelope { + version, + token: token.map(str::to_owned), + request: Request::Stat { + digest, + path: b"hello".to_vec(), + }, + }, + ) + .unwrap(); + assert!(matches!( + read_frame::(&mut client).unwrap(), + Response::Error { .. } + )); + worker.join().unwrap(); + } + } + + #[test] + fn client_rejects_corrupt_content() { + let tmp = tempfile::tempdir().unwrap(); + let path = tmp.path().join("server.sock"); + let listener = UnixListener::bind(&path).unwrap(); + let worker = std::thread::spawn(move || { + let (mut socket, _) = listener.accept().unwrap(); + let _: Envelope = read_frame(&mut socket).unwrap(); + write_frame( + &mut socket, + &Response::Data { + length: 3, + sha256: hash(b"abc"), + }, + ) + .unwrap(); + socket.write_all(b"bad").unwrap(); + }); + let client = CacheClient::new(format!("unix://{}", path.display()), None).unwrap(); + let error = client + .request(Request::Read { + digest: "unused".into(), + path: b"file".to_vec(), + offset: 0, + length: 3, + }) + .unwrap_err(); + assert!(error.to_string().contains("digest mismatch")); + worker.join().unwrap(); + } +} diff --git a/crates/persisting-pvisor/src/cache/lazy.rs b/crates/persisting-pvisor/src/cache/lazy.rs new file mode 100644 index 00000000..6c8186aa --- /dev/null +++ b/crates/persisting-pvisor/src/cache/lazy.rs @@ -0,0 +1,870 @@ +//! A read-only, demand-filled FUSE lower for the existing VM/OCI overlays. +use super::{CacheClient, MAX_READ, Request as CacheRequest, Response, architecture, hash}; +use crate::oci::{ImageStore, PreparedImage}; +use anyhow::{Context, ensure}; +use fs2::FileExt; +use fuser::{ + BackgroundSession, FileAttr, FileType, Filesystem, MountOption, ReplyAttr, ReplyData, + ReplyDirectory, ReplyEntry, ReplyOpen, ReplyStatfs, Request, Session, +}; +use sha2::{Digest, Sha256}; +use std::collections::HashMap; +use std::ffi::{OsStr, OsString}; +use std::fs::{self, OpenOptions}; +use std::io::Write; +use std::os::unix::ffi::{OsStrExt, OsStringExt}; +use std::os::unix::fs::OpenOptionsExt; +#[cfg(target_os = "linux")] +use std::os::unix::fs::PermissionsExt; +use std::path::{Path, PathBuf}; +use std::time::{Duration, UNIX_EPOCH}; + +const TTL: Duration = Duration::from_secs(3600); + +pub(crate) struct LazyMount { + session: Option, + path: PathBuf, +} +impl Drop for LazyMount { + fn drop(&mut self) { + if let Some(session) = self.session.take() + && let Err(error) = session.unmount() + { + crate::cli::diagnostic(format_args!( + "unmount lazy image {}: {error}", + self.path.display() + )); + } + #[cfg(target_os = "linux")] + let _ = fs::remove_dir(&self.path); + } +} + +pub(crate) fn prepare_image( + image: &str, + store: Option, +) -> anyhow::Result<(PreparedImage, Option)> { + let client = CacheClient::discover()?; + let store = ImageStore::new(store)?; + let Some(client) = client else { + return Ok(( + super::progress::loading("resolving and preparing local image", || { + store.prepare(image) + })?, + None, + )); + }; + let downloads = super::progress::Downloads::new(image); + crate::cli::diagnostic(format_args!( + "pVisor image: lazy loading from {}", + client.endpoint + )); + let (response, _) = + super::progress::loading("waiting for cache to resolve and prepare image", || { + client.request(CacheRequest::Prepare { + image: image.into(), + architecture: architecture().into(), + }) + })?; + let Response::Prepared { + metadata_generation, + totals, + digest, + architecture: platform, + env, + entrypoint, + cmd, + } = response + else { + anyhow::bail!("expected cache prepared response"); + }; + ensure!( + platform == architecture(), + "cache returned the wrong image architecture" + ); + crate::oci::digest_hex(&digest)?; + let cache = dirs::cache_dir() + .context("cannot find user cache directory")? + .join("persisting/pvisor/blocks") + .join(&hash(client.endpoint.as_bytes())[7..]) + .join(&digest[7..]); + fs::create_dir_all(&cache)?; + downloads.totals(totals); + if let Some(totals) = totals { + crate::cli::diagnostic(format_args!( + "pVisor image: prepared {digest}; {} files, {:.1} MiB (contents fetched on demand)", + totals.files, + totals.bytes as f64 / (1024.0 * 1024.0) + )); + } + let metadata_cache = metadata_generation.map(|generation| { + dirs::cache_dir() + .expect("cache directory already resolved") + .join("persisting/pvisor/metadata/v1") + .join(&hash(client.endpoint.as_bytes())[7..]) + .join(&digest[7..]) + .join(&hash(generation.as_bytes())[7..]) + }); + let mut filesystem = super::progress::loading("loading root metadata", || { + RemoteFs::new(client, digest.clone(), cache, metadata_cache) + })?; + filesystem.downloads = downloads; + let mount = + super::progress::loading("mounting lazy rootfs", || mount(filesystem, &store.root))?; + Ok(( + PreparedImage { + rootfs: mount.path.clone(), + digest, + env, + entrypoint, + cmd, + }, + Some(mount), + )) +} + +fn mount(filesystem: RemoteFs, _store: &Path) -> anyhow::Result { + #[cfg(target_os = "macos")] + let mountpoint = + PathBuf::from("/Volumes").join(format!("pvisor-image-{}", uuid::Uuid::new_v4())); + #[cfg(target_os = "linux")] + let mountpoint = { + let path = _store.join(format!(".lazy-mount-{}", uuid::Uuid::new_v4())); + fs::create_dir(&path)?; + fs::set_permissions(&path, fs::Permissions::from_mode(0o700))?; + path + }; + #[allow(unused_mut)] + let mut options = vec![ + MountOption::FSName("pvisor-image".into()), + MountOption::RO, + MountOption::NoAtime, + MountOption::DefaultPermissions, + ]; + #[cfg(target_os = "macos")] + options.push(MountOption::CUSTOM("backend=fskit".into())); + let session = Session::new(filesystem, &mountpoint, &options) + .context("mount lazy image lower (FUSE is required); set PERSISTING_PVISOR_CACHE_SERVER=off to use local OCI extraction")?; + let mount = LazyMount { + session: Some(BackgroundSession::new(session)?), + path: mountpoint.clone(), + }; + // FSKit attaches asynchronously after its request loop starts. + #[cfg(target_os = "macos")] + for _ in 0..250 { + if persisting_overlayfs::is_mountpoint(&mountpoint) { + break; + } + if mount + .session + .as_ref() + .is_some_and(|s| s.guard.is_finished()) + { + break; + } + std::thread::sleep(Duration::from_millis(20)); + } + ensure!( + mountpoint.is_dir(), + "lazy image mount did not become available" + ); + Ok(mount) +} + +#[derive(Clone)] +struct Node { + path: Vec, + attr: FileAttr, + target: Option>, +} + +struct RemoteFs { + downloads: super::progress::Downloads, + client: CacheClient, + digest: String, + cache: PathBuf, + metadata_cache: Option, + nodes: HashMap, + paths: HashMap, u64>, + objects: HashMap, + directories: HashMap>, + next_inode: u64, +} +impl RemoteFs { + fn new( + client: CacheClient, + digest: String, + cache: PathBuf, + metadata_cache: Option, + ) -> anyhow::Result { + let mut fs = Self { + downloads: super::progress::Downloads::default(), + client, + digest, + cache, + metadata_cache, + nodes: HashMap::new(), + paths: HashMap::new(), + objects: HashMap::new(), + directories: HashMap::new(), + next_inode: 1, + }; + let root = fs.lookup_path(Vec::new())?; + ensure!( + root.attr.ino == 1 && root.attr.kind == FileType::Directory, + "invalid remote image root" + ); + Ok(fs) + } + + fn metadata_request(&self, request: CacheRequest) -> anyhow::Result { + let Some(directory) = &self.metadata_cache else { + return self.client.request(request).map(|(response, _)| response); + }; + let path = directory.join(&hash(&serde_json::to_vec(&request)?)[7..]); + let cached = fs::read(&path).ok().and_then(|bytes| { + (bytes.len() >= 32 && Sha256::digest(&bytes[32..]).as_slice() == &bytes[..32]) + .then(|| serde_json::from_slice::(&bytes[32..]).ok()) + .flatten() + }); + let response = if let Some(response) = cached { + response + } else { + let response = match self.client.request(request) { + Ok((response, _)) => response, + Err(error) + if error + .downcast_ref::() + .is_some_and(|e| e.kind() == std::io::ErrorKind::NotFound) => + { + Response::Error { + code: "not_found".into(), + message: "path absent from immutable image".into(), + } + } + Err(error) => return Err(error), + }; + fs::create_dir_all(directory)?; + let bytes = serde_json::to_vec(&response)?; + let mut temporary = tempfile::NamedTempFile::new_in(directory)?; + temporary.write_all(&Sha256::digest(&bytes))?; + temporary.write_all(&bytes)?; + temporary.persist(&path)?; + response + }; + match response { + Response::Error { code, message } if code == "not_found" => { + Err(std::io::Error::new(std::io::ErrorKind::NotFound, message).into()) + } + response => Ok(response), + } + } + + fn lookup_path(&mut self, path: Vec) -> anyhow::Result { + if let Some(ino) = self.paths.get(&path) { + return self.node(*ino).cloned(); + } + let response = self.metadata_request(CacheRequest::Stat { + digest: self.digest.clone(), + path: path.clone(), + })?; + let Response::Metadata { + kind, + size, + mode, + inode, + nlink, + mtime, + mtime_nsec, + target, + .. + } = response + else { + anyhow::bail!("expected cache metadata response"); + }; + if let Some(ino) = self.objects.get(&inode) { + self.paths.insert(path, *ino); + return self.node(*ino).cloned(); + } + let ino = self.next_inode; + self.next_inode += 1; + let kind = match kind.as_str() { + "directory" => FileType::Directory, + "file" => FileType::RegularFile, + "symlink" => FileType::Symlink, + _ => match mode & libc::S_IFMT as u32 { + x if x == libc::S_IFIFO as u32 => FileType::NamedPipe, + x if x == libc::S_IFSOCK as u32 => FileType::Socket, + _ => anyhow::bail!("unsupported special file in remote image"), + }, + }; + ensure!( + (0..1_000_000_000).contains(&mtime_nsec), + "invalid remote timestamp" + ); + let time = if mtime >= 0 { + UNIX_EPOCH.checked_add(Duration::new(mtime as u64, mtime_nsec as u32)) + } else { + UNIX_EPOCH.checked_sub(Duration::from_secs(mtime.unsigned_abs())) + } + .context("remote timestamp overflow")?; + let node = Node { + path: path.clone(), + target, + attr: FileAttr { + ino, + size, + blocks: size.div_ceil(512), + atime: time, + mtime: time, + ctime: time, + crtime: time, + kind, + // Host permission checks use the mounting user; the guest receives its + // normal permission semantics through the existing virtio-fs overlay. + perm: if ino == 1 { + 0o700 + } else { + (mode & 0o7777) as u16 + }, + uid: unsafe { libc::geteuid() }, + gid: unsafe { libc::getegid() }, + nlink: nlink.min(u32::MAX as u64) as u32, + rdev: 0, + blksize: 4096, + flags: 0, + }, + }; + self.paths.insert(path, ino); + self.objects.insert(inode, ino); + self.nodes.insert(ino, node.clone()); + Ok(node) + } + + fn node(&self, ino: u64) -> anyhow::Result<&Node> { + self.nodes + .get(&ino) + .ok_or_else(|| std::io::Error::from_raw_os_error(libc::ENOENT).into()) + } + + fn child(&mut self, parent: u64, name: &OsStr) -> anyhow::Result { + let parent = self.node(parent)?; + ensure!(parent.attr.kind == FileType::Directory, "not a directory"); + let name = name.as_bytes(); + ensure!( + !name.is_empty() + && name != b"." + && name != b".." + && !name.contains(&b'/') + && !name.contains(&0), + "invalid remote filename" + ); + let mut path = parent.path.clone(); + if !path.is_empty() { + path.push(b'/'); + } + path.extend_from_slice(name); + self.lookup_path(path) + } + + fn entries(&mut self, ino: u64) -> anyhow::Result<&Vec<(u64, FileType, OsString)>> { + if !self.directories.contains_key(&ino) { + let node = self.node(ino)?.clone(); + ensure!(node.attr.kind == FileType::Directory, "not a directory"); + let parent = Path::new(OsStr::from_bytes(&node.path)) + .parent() + .unwrap_or(Path::new("")); + let parent = self.lookup_path(parent.as_os_str().as_bytes().to_vec())?; + let mut entries = vec![ + (ino, FileType::Directory, ".".into()), + (parent.attr.ino, FileType::Directory, "..".into()), + ]; + let mut offset = 0; + loop { + let response = self.metadata_request(CacheRequest::List { + digest: self.digest.clone(), + path: node.path.clone(), + offset, + })?; + let Response::Entries { names, next_offset } = response else { + anyhow::bail!("expected cache directory response"); + }; + for name in names { + let child = self.child(ino, OsStr::from_bytes(&name))?; + entries.push((child.attr.ino, child.attr.kind, OsString::from_vec(name))); + } + match next_offset { + Some(next) => { + ensure!(next > offset, "invalid cache directory continuation"); + offset = next; + } + None => break, + } + } + self.directories.insert(ino, entries); + } + Ok(self.directories.get(&ino).unwrap()) + } + + fn block(&self, node: &Node, index: u64) -> anyhow::Result> { + let directory = self.cache.join(&hash(&node.path)[7..]); + fs::create_dir_all(&directory)?; + let path = directory.join(index.to_string()); + let lock = OpenOptions::new() + .read(true) + .write(true) + .create(true) + .truncate(false) + .mode(0o600) + .custom_flags(libc::O_NOFOLLOW) + .open(directory.join(format!("{index}.lock")))?; + lock.lock_exclusive()?; + let offset = index + .checked_mul(MAX_READ as u64) + .context("cache block offset overflow")?; + let length = node.attr.size.saturating_sub(offset).min(MAX_READ as u64) as usize; + match fs::read(&path) { + Ok(bytes) + if bytes.len() == length + 32 + && Sha256::digest(&bytes[32..]).as_slice() == &bytes[..32] => + { + self.downloads.cached(&node.path, length); + return Ok(bytes[32..].to_vec()); + } + Ok(_) => { + fs::remove_file(&path)?; + } + Err(e) if e.kind() == std::io::ErrorKind::NotFound => {} + Err(e) => return Err(e.into()), + } + let (_, body) = self.client.request(CacheRequest::Read { + digest: self.digest.clone(), + path: node.path.clone(), + offset, + length: length as u32, + })?; + ensure!( + body.len() == length, + "remote file returned a short block before EOF" + ); + self.downloads.received(&node.path, body.len()); + let mut temporary = tempfile::NamedTempFile::new_in(&directory)?; + temporary.write_all(&Sha256::digest(&body))?; + temporary.write_all(&body)?; + temporary.persist(&path)?; + Ok(body) + } + + fn read_range(&self, ino: u64, offset: u64, size: u32) -> anyhow::Result> { + let node = self.node(ino)?; + ensure!( + node.attr.kind == FileType::RegularFile, + "read requires a regular file" + ); + let end = offset.saturating_add(size as u64).min(node.attr.size); + let mut result = Vec::new(); + let mut cursor = offset; + while cursor < end { + let bytes = self.block(node, cursor / MAX_READ as u64)?; + let begin = (cursor % MAX_READ as u64) as usize; + let count = (end - cursor).min((bytes.len() - begin) as u64) as usize; + result.extend_from_slice(&bytes[begin..begin + count]); + cursor += count as u64; + } + Ok(result) + } +} + +fn errno(error: anyhow::Error) -> i32 { + let code = error + .chain() + .filter_map(|e| e.downcast_ref::()) + .find_map(|e| { + e.raw_os_error().or(match e.kind() { + std::io::ErrorKind::NotFound => Some(libc::ENOENT), + std::io::ErrorKind::PermissionDenied => Some(libc::EACCES), + _ => None, + }) + }) + .unwrap_or(libc::EIO); + if code != libc::ENOENT { + crate::cli::diagnostic(format_args!("lazy image I/O: {error:#}")); + } + code +} +// ponytail: synchronous FUSE reads preserve the existing serialized virtio-fs +// behavior. Introduce queued completions when cache-miss latency warrants it. +impl Filesystem for RemoteFs { + fn lookup(&mut self, _: &Request<'_>, parent: u64, name: &OsStr, reply: ReplyEntry) { + match self.child(parent, name) { + Ok(node) => reply.entry(&TTL, &node.attr, 0), + Err(e) => reply.error(errno(e)), + } + } + fn getattr(&mut self, _: &Request<'_>, ino: u64, _: Option, reply: ReplyAttr) { + match self.node(ino) { + Ok(node) => reply.attr(&TTL, &node.attr), + Err(e) => reply.error(errno(e)), + } + } + fn readlink(&mut self, _: &Request<'_>, ino: u64, reply: ReplyData) { + match self + .node(ino) + .and_then(|node| node.target.as_deref().context("not a symlink")) + { + Ok(target) => reply.data(target), + Err(e) => reply.error(errno(e)), + } + } + fn open(&mut self, _: &Request<'_>, ino: u64, flags: i32, reply: ReplyOpen) { + // FSKit can request O_RDWR for a read on a read-only volume. + if flags & (libc::O_TRUNC | libc::O_APPEND) != 0 { + reply.error(libc::EROFS); + return; + } + match self.node(ino) { + Ok(node) if matches!(node.attr.kind, FileType::RegularFile | FileType::Symlink) => { + reply.opened(ino, 0) + } + _ => reply.error(libc::EINVAL), + } + } + fn read( + &mut self, + _: &Request<'_>, + ino: u64, + _: u64, + offset: i64, + size: u32, + _: i32, + _: Option, + reply: ReplyData, + ) { + if offset < 0 { + reply.error(libc::EINVAL); + return; + } + match self.read_range(ino, offset as u64, size) { + Ok(bytes) => reply.data(&bytes), + Err(e) => reply.error(errno(e)), + } + } + fn opendir(&mut self, _: &Request<'_>, ino: u64, _: i32, reply: ReplyOpen) { + match self.node(ino) { + Ok(node) if node.attr.kind == FileType::Directory => reply.opened(ino, 0), + _ => reply.error(libc::ENOTDIR), + } + } + fn readdir( + &mut self, + _: &Request<'_>, + ino: u64, + _: u64, + offset: i64, + mut reply: ReplyDirectory, + ) { + if offset < 0 { + reply.error(libc::EINVAL); + return; + } + match self.entries(ino) { + Ok(entries) => { + for (index, (ino, kind, name)) in entries.iter().enumerate().skip(offset as usize) { + if reply.add(*ino, (index + 1) as i64, *kind, name) { + break; + } + } + reply.ok(); + } + Err(e) => reply.error(errno(e)), + } + } + fn statfs(&mut self, _: &Request<'_>, _: u64, reply: ReplyStatfs) { + reply.statfs(0, 0, 0, 0, 0, 4096, 255, 4096); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cache::{Envelope, handle, read_frame, write_frame}; + use std::os::unix::net::UnixListener; + use std::sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, + }; + + struct Server { + stop: Arc, + worker: Option>, + reads: Arc, + } + impl Drop for Server { + fn drop(&mut self) { + self.stop.store(true, Ordering::Relaxed); + self.worker.take().unwrap().join().unwrap(); + } + } + fn fixture() -> (tempfile::TempDir, Server, CacheClient, String) { + let temp = tempfile::tempdir().unwrap(); + let store = ImageStore::new(Some(temp.path().join("store"))).unwrap(); + let digest = format!("sha256:{}", "b".repeat(64)); + let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); + fs::create_dir(&root).unwrap(); + fs::write(root.join("large"), vec![42; 3 * MAX_READ as usize]).unwrap(); + std::os::unix::fs::symlink("large", root.join("alias")).unwrap(); + let socket = temp.path().join("s"); + let listener = UnixListener::bind(&socket).unwrap(); + listener.set_nonblocking(true).unwrap(); + let stop = Arc::new(AtomicBool::new(false)); + let reads = Arc::new(AtomicUsize::new(0)); + let worker_stop = stop.clone(); + let worker_reads = reads.clone(); + let worker = std::thread::spawn(move || { + while !worker_stop.load(Ordering::Relaxed) { + let (mut socket, _) = match listener.accept() { + Ok(s) => s, + Err(e) if e.kind() == std::io::ErrorKind::WouldBlock => { + std::thread::sleep(Duration::from_millis(1)); + continue; + } + Err(e) => panic!("{e}"), + }; + socket.set_nonblocking(false).unwrap(); + let envelope: Envelope = read_frame(&mut socket).unwrap(); + if matches!(&envelope.request, CacheRequest::Read { .. }) { + worker_reads.fetch_add(1, Ordering::Relaxed); + } + let (response, bytes) = handle(&store, envelope.request).unwrap_or_else(|e| { + let code = if e + .downcast_ref::() + .is_some_and(|e| e.kind() == std::io::ErrorKind::NotFound) + { + "not_found" + } else { + "request_failed" + }; + ( + Response::Error { + code: code.into(), + message: e.to_string(), + }, + Vec::new(), + ) + }); + write_frame(&mut socket, &response).unwrap(); + socket.write_all(&bytes).unwrap(); + } + }); + let client = CacheClient::new(format!("unix://{}", socket.display()), None).unwrap(); + ( + temp, + Server { + stop, + reads, + worker: Some(worker), + }, + client, + digest, + ) + } + + #[test] + fn persistent_metadata_survives_remount_and_rejects_corruption() { + let (temp, server, client, digest) = fixture(); + let blocks = temp.path().join("blocks"); + let metadata = temp.path().join("metadata"); + let endpoint = client.endpoint.clone(); + let mut cold = RemoteFs::new( + client, + digest.clone(), + blocks.clone(), + Some(metadata.clone()), + ) + .unwrap(); + let expected = cold.child(1, OsStr::new("large")).unwrap().attr.size; + let count = cold.entries(1).unwrap().len(); + assert!(cold.child(1, OsStr::new("missing")).is_err()); + drop(server); + let client = || CacheClient::new(endpoint.clone(), None).unwrap(); + let mut warm = RemoteFs::new( + client(), + digest.clone(), + blocks.clone(), + Some(metadata.clone()), + ) + .unwrap(); + assert_eq!( + warm.child(1, OsStr::new("large")).unwrap().attr.size, + expected + ); + assert_eq!(warm.entries(1).unwrap().len(), count); + let error = warm.child(1, OsStr::new("missing")).err().unwrap(); + assert_eq!( + error.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::NotFound + ); + // A different generation cannot borrow entries from the old snapshot. + assert!( + RemoteFs::new( + client(), + digest.clone(), + blocks.clone(), + Some(temp.path().join("new-generation")) + ) + .is_err() + ); + let root = CacheRequest::Stat { + digest: digest.clone(), + path: vec![], + }; + fs::write( + metadata.join(&hash(&serde_json::to_vec(&root).unwrap())[7..]), + b"corrupt", + ) + .unwrap(); + assert!(RemoteFs::new(client(), digest, blocks, Some(metadata)).is_err()); + } + + #[test] + fn reads_only_requested_blocks_and_reuses_verified_cache() { + let (temp, server, client, digest) = fixture(); + let cache = temp.path().join("client"); + let mut filesystem = RemoteFs::new(client, digest, cache, None).unwrap(); + let file = filesystem.child(1, OsStr::new("large")).unwrap(); + assert_eq!(filesystem.entries(1).unwrap().len(), 4); + assert_eq!( + server.reads.load(Ordering::Relaxed), + 0, + "metadata must not download content" + ); + let offset = MAX_READ as u64 - 4; + assert_eq!( + filesystem.read_range(file.attr.ino, offset, 16).unwrap(), + [42; 16] + ); + assert_eq!( + server.reads.load(Ordering::Relaxed), + 2, + "only two intersecting blocks are fetched" + ); + assert_eq!( + filesystem.read_range(file.attr.ino, offset, 16).unwrap(), + [42; 16] + ); + assert_eq!(server.reads.load(Ordering::Relaxed), 2); + let progress = filesystem.downloads.snapshot(); + assert_eq!(progress.downloaded_files, 1); + assert_eq!(progress.downloaded_bytes, 2 * MAX_READ as u64); + let path = filesystem.cache.join(&hash(&file.path)[7..]).join("0"); + fs::write(path, b"corrupt").unwrap(); + assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); + assert_eq!( + server.reads.load(Ordering::Relaxed), + 3, + "corrupt cache must be replaced" + ); + assert_eq!(filesystem.downloads.snapshot().downloaded_files, 1); + assert_eq!( + filesystem.downloads.snapshot().downloaded_bytes, + 3 * MAX_READ as u64 + ); + let warm_client = CacheClient::new(filesystem.client.endpoint.clone(), None).unwrap(); + let mut warm = RemoteFs::new( + warm_client, + filesystem.digest.clone(), + filesystem.cache.clone(), + None, + ) + .unwrap(); + let warm_file = warm.child(1, OsStr::new("large")).unwrap(); + assert_eq!( + warm.read_range(warm_file.attr.ino, offset, 16).unwrap(), + [42; 16] + ); + assert_eq!(warm.downloads.snapshot().downloaded_files, 0); + assert_eq!(warm.downloads.snapshot().downloaded_bytes, 0); + assert_eq!(warm.downloads.snapshot().cached_files, 1); + assert_eq!(warm.downloads.snapshot().cached_bytes, 2 * MAX_READ as u64); + warm.read_range(warm_file.attr.ino, offset, 16).unwrap(); + assert_eq!(warm.downloads.snapshot().cached_files, 1); + assert_eq!(warm.downloads.snapshot().cached_bytes, 4 * MAX_READ as u64); + drop(server); + assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); + assert!( + filesystem + .read_range(file.attr.ino, 2 * MAX_READ as u64, 1) + .is_err(), + "uncached data must fail, never become zeroes" + ); + } + + #[test] + #[ignore = "requires a working host FUSE installation"] + fn native_mount_reads_lazily_and_unmounts() { + use std::io::Read; + let (temp, server, client, digest) = fixture(); + let filesystem = RemoteFs::new(client, digest, temp.path().join("client"), None).unwrap(); + let mount = mount(filesystem, temp.path()).unwrap(); + assert_eq!( + fs::metadata(mount.path.join("large")).unwrap().len(), + 3 * MAX_READ as u64 + ); + assert_eq!(server.reads.load(Ordering::Relaxed), 0); + assert_eq!( + fs::read_link(mount.path.join("alias")).unwrap(), + Path::new("large") + ); + #[cfg(target_os = "macos")] + { + use std::os::fd::AsRawFd; + let link = OpenOptions::new() + .read(true) + .custom_flags(libc::O_SYMLINK) + .open(mount.path.join("alias")) + .unwrap(); + let mut path = [0u8; libc::PATH_MAX as usize]; + assert_eq!( + unsafe { libc::fcntl(link.as_raw_fd(), libc::F_GETPATH, path.as_mut_ptr()) }, + 0 + ); + } + let mut file = std::fs::File::open(mount.path.join("large")).unwrap(); + let mut bytes = [0u8; 16]; + file.read_exact(&mut bytes).unwrap(); + assert_eq!(bytes, [42; 16]); + assert!(server.reads.load(Ordering::Relaxed) < 3); + drop(file); + let path = mount.path.clone(); + drop(mount); + #[cfg(target_os = "macos")] + assert!(!persisting_overlayfs::is_mountpoint(&path)); + #[cfg(target_os = "linux")] + assert!(!path.exists()); + } + + #[test] + fn auto_probe_distinguishes_absence_from_explicit_failure() { + let temp = tempfile::tempdir().unwrap(); + let address = format!("unix://{}", temp.path().join("missing").display()); + assert!( + CacheClient::probe(address.clone(), None, false) + .unwrap() + .is_none() + ); + assert!(CacheClient::probe(address, None, true).is_err()); + let socket = temp.path().join("stale"); + drop(UnixListener::bind(&socket).unwrap()); + assert!( + CacheClient::probe(format!("unix://{}", socket.display()), None, false) + .unwrap() + .is_none() + ); + let (_temp, _server, client, _) = fixture(); + assert!( + CacheClient::probe(client.endpoint.clone(), None, false) + .unwrap() + .is_some() + ); + } +} diff --git a/crates/persisting-pvisor/src/cache/metadata.rs b/crates/persisting-pvisor/src/cache/metadata.rs new file mode 100644 index 00000000..cb867aff --- /dev/null +++ b/crates/persisting-pvisor/src/cache/metadata.rs @@ -0,0 +1,72 @@ +//! Metadata for immutable, fully prepared image roots. +use super::*; +use std::collections::HashMap; +use std::os::unix::fs::MetadataExt; +use std::sync::OnceLock; + +static STATS: OnceLock>> = OnceLock::new(); +type Names = Arc>>; +static DIRECTORIES: OnceLock>> = OnceLock::new(); + +pub(super) fn generation(store: &ImageStore, digest: &str) -> anyhow::Result { + let root = store + .root + .join("rootfs-v3/sha256") + .join(crate::oci::digest_hex(digest)?); + let m = fs::symlink_metadata(&root)?; + ensure!(m.is_dir(), "image root is not a directory"); + // Include store identity and host inode generation: extracted inode numbers + // can change when the same OCI digest is removed and prepared again. + Ok(hash(&serde_json::to_vec(&( + root.as_os_str().as_bytes(), + m.dev(), + m.ino(), + m.ctime(), + m.ctime_nsec(), + ))?)) +} + +fn key(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result { + Ok(hash(&serde_json::to_vec(&( + generation(store, digest)?, + digest, + path, + ))?)) +} + +fn cached( + cache: &OnceLock>>, + key: String, + capacity: usize, + load: impl FnOnce() -> anyhow::Result, +) -> anyhow::Result { + let mut cache = cache.get_or_init(Default::default).lock().unwrap(); + if let Some(value) = cache.get(&key) { + return Ok(value.clone()); + } + // ponytail: serialize metadata misses and clear at the entry ceiling; + // use per-key loading and an LRU only if cache contention/churn matters. + let value = load()?; + if cache.len() >= capacity { + cache.clear(); + } + cache.insert(key, value.clone()); + Ok(value) +} + +pub(super) fn stat(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result { + cached(&STATS, key(store, digest, path)?, 4096, || { + let (directory, name) = parent(store, digest, path)?; + metadata_at(&directory, &name) + }) +} + +pub(super) fn directory(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result { + cached(&DIRECTORIES, key(store, digest, path)?, 128, || { + let (directory, name) = parent(store, digest, path)?; + let directory = open_child(&directory, OsStr::from_bytes(&name), true)?; + let mut names = directory_names(directory)?; + names.sort(); + Ok(Arc::new(names)) + }) +} diff --git a/crates/persisting-pvisor/src/cache/progress.rs b/crates/persisting-pvisor/src/cache/progress.rs new file mode 100644 index 00000000..f30d6dd4 --- /dev/null +++ b/crates/persisting-pvisor/src/cache/progress.rs @@ -0,0 +1,232 @@ +//! Image size and per-run payload transfers, separate from guest network traffic. +use serde::{Deserialize, Serialize}; +use std::cell::RefCell; +use std::collections::HashSet; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::sync::{OnceLock, mpsc}; +use std::time::{Duration, Instant}; + +#[derive(Clone, Copy, Debug, Default, Serialize, Deserialize, PartialEq, Eq)] +pub struct ImageTotals { + pub files: u64, + pub bytes: u64, +} + +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub(crate) struct ImageProgress { + pub image: String, + pub totals: Option, + pub downloaded_files: u64, + pub downloaded_bytes: u64, + #[serde(default)] + pub cached_files: u64, + #[serde(default)] + pub cached_bytes: u64, +} + +static OUTPUT: OnceLock> = OnceLock::new(); +pub(crate) fn init_output(path: Option) { + let _ = OUTPUT.set(path); +} + +/// Keep blocking startup work visible without changing the cache protocol. +pub(super) fn loading( + label: &str, + work: impl FnOnce() -> anyhow::Result, +) -> anyhow::Result { + crate::cli::diagnostic(format_args!("pVisor image: {label}")); + let started = Instant::now(); + std::thread::scope(|scope| { + let (done, wait) = mpsc::channel::<()>(); + scope.spawn(move || { + while matches!( + wait.recv_timeout(Duration::from_secs(5)), + Err(mpsc::RecvTimeoutError::Timeout) + ) { + crate::cli::diagnostic(format_args!( + "pVisor image: {label}; still waiting ({:.0}s elapsed)", + started.elapsed().as_secs_f64() + )); + } + }); + let result = work(); + drop(done); + match &result { + Ok(_) => crate::cli::diagnostic(format_args!( + "pVisor image: {label}; done ({:.1}s)", + started.elapsed().as_secs_f64() + )), + Err(error) => crate::cli::diagnostic(format_args!( + "pVisor image: {label}; failed ({:.1}s): {error:#}", + started.elapsed().as_secs_f64() + )), + } + result + }) +} + +#[derive(Default)] +pub(crate) struct Downloads { + snapshot: RefCell, + files: RefCell>>, + cached_files: RefCell>>, + output: Option, +} +impl Downloads { + pub fn new(image: &str) -> Self { + let downloads = Self { + snapshot: RefCell::new(ImageProgress { + image: image.into(), + ..Default::default() + }), + files: RefCell::new(HashSet::new()), + cached_files: RefCell::new(HashSet::new()), + output: OUTPUT.get().cloned().flatten(), + }; + downloads.publish(); + downloads + } + pub fn totals(&self, totals: Option) { + self.snapshot.borrow_mut().totals = totals; + self.publish(); + } + pub fn received(&self, path: &[u8], bytes: usize) { + let first = self.files.borrow_mut().insert(path.to_vec()); + let mut snapshot = self.snapshot.borrow_mut(); + snapshot.downloaded_files += u64::from(first); + snapshot.downloaded_bytes += bytes as u64; + crate::cli::diagnostic(format_args!( + "pVisor image: transferred {bytes} bytes from /{} (this run: {} bytes across {} files)", + String::from_utf8_lossy(path).escape_debug(), + snapshot.downloaded_bytes, + snapshot.downloaded_files, + )); + drop(snapshot); + self.publish(); + } + pub fn cached(&self, path: &[u8], bytes: usize) { + let first = self.cached_files.borrow_mut().insert(path.to_vec()); + let mut snapshot = self.snapshot.borrow_mut(); + snapshot.cached_files += u64::from(first); + snapshot.cached_bytes += bytes as u64; + if first { + crate::cli::diagnostic(format_args!( + "pVisor image: cached /{} ({bytes} bytes read; no download)", + String::from_utf8_lossy(path).escape_debug(), + )); + } + drop(snapshot); + self.publish(); + } + #[cfg(test)] + pub fn snapshot(&self) -> ImageProgress { + self.snapshot.borrow().clone() + } + + fn publish(&self) { + if let Some(path) = &self.output { + // This transient UI snapshot needs atomic replacement, not durable fsync. + let result = (|| -> anyhow::Result<()> { + let mut file = tempfile::NamedTempFile::new_in(path.parent().unwrap())?; + serde_json::to_writer(&mut file, &*self.snapshot.borrow())?; + file.flush()?; + file.persist(path)?; + Ok(()) + })(); + if let Err(error) = result { + tracing::debug!("image progress update: {error}"); + } + } + } +} + +pub(super) fn image_totals( + store: &crate::oci::ImageStore, + digest: &str, +) -> anyhow::Result { + let hex = crate::oci::digest_hex(digest)?; + let record = store + .root + .join("metadata/sha256") + .join(format!("{hex}.totals-v1.json")); + if let Ok(bytes) = std::fs::read(&record) + && let Ok(totals) = serde_json::from_slice(&bytes) + { + return Ok(totals); + } + let totals = scan_totals(&store.root.join("rootfs-v3/sha256").join(hex))?; + crate::util::atomic_write(&record, &serde_json::to_vec(&totals)?, 0o600)?; + Ok(totals) +} + +fn scan_totals(root: &Path) -> anyhow::Result { + let mut directories = vec![root.to_path_buf()]; + let mut totals = ImageTotals::default(); + while let Some(directory) = directories.pop() { + for entry in std::fs::read_dir(directory)? { + let entry = entry?; + let metadata = std::fs::symlink_metadata(entry.path())?; + if metadata.is_dir() { + directories.push(entry.path()); + } else if metadata.is_file() { + totals.files += 1; + totals.bytes = totals + .bytes + .checked_add(metadata.len()) + .ok_or_else(|| anyhow::anyhow!("image size overflow"))?; + } + } + } + Ok(totals) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn loading_returns_the_result_and_preserves_error_context() { + assert_eq!(loading("test success", || Ok(42)).unwrap(), 42); + let error = loading::<()>("test failure", || { + Err(anyhow::anyhow!("connection timed out").context("registry request")) + }) + .unwrap_err(); + assert_eq!( + format!("{error:#}"), + "registry request: connection timed out" + ); + } + + #[test] + fn totals_count_regular_paths_without_following_links() { + let root = tempfile::tempdir().unwrap(); + std::fs::create_dir(root.path().join("dir")).unwrap(); + std::fs::write(root.path().join("dir/file"), b"hello").unwrap(); + std::fs::write(root.path().join("empty"), b"").unwrap(); + std::os::unix::fs::symlink("dir", root.path().join("alias")).unwrap(); + std::os::unix::fs::symlink("/", root.path().join("outside")).unwrap(); + assert_eq!( + scan_totals(root.path()).unwrap(), + ImageTotals { files: 2, bytes: 5 } + ); + } + #[test] + fn snapshots_count_each_file_once_but_all_transferred_bytes() { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("image.json"); + let mut downloads = Downloads::new("example:latest"); + downloads.output = Some(path.clone()); + downloads.totals(Some(ImageTotals { + files: 3, + bytes: 100, + })); + downloads.received(b"a", 10); + downloads.received(b"a", 10); + downloads.received(b"b", 20); + let snapshot: ImageProgress = + serde_json::from_slice(&std::fs::read(path).unwrap()).unwrap(); + assert_eq!(snapshot.downloaded_files, 2); + assert_eq!(snapshot.downloaded_bytes, 40); + assert_eq!(snapshot.totals.unwrap().files, 3); + } +} diff --git a/crates/persisting-pvisor/src/cli/mod.rs b/crates/persisting-pvisor/src/cli/mod.rs index a078a189..ff611a44 100644 --- a/crates/persisting-pvisor/src/cli/mod.rs +++ b/crates/persisting-pvisor/src/cli/mod.rs @@ -8,6 +8,8 @@ pub mod runtime; mod trajectory; #[cfg(unix)] mod tui; +#[cfg(unix)] +pub(crate) use tui::diagnostic; use clap::{Parser, Subcommand}; @@ -47,6 +49,9 @@ enum Command { long_about = run::RUN_COMMAND_LONG_ABOUT )] Run(Box), + /// Serve or query the shared OCI file cache. + #[cfg(unix)] + Cache(crate::cache::CacheArgs), /// Apply selected staged changes from a stopped Job. Apply(runtime::ApplyArgs), /// Discard staged changes from a stopped Job. @@ -78,11 +83,11 @@ pub fn main() -> anyhow::Result<()> { if run.tui_requested() || audit { anyhow::ensure!( run.wants_tui(audit), - "--tui/--audit requires inherited stdio and a normal Job" + "--tui/--ask requires inherited stdio and a normal Job" ); anyhow::ensure!( tui::available(), - "--tui/--audit requires an interactive terminal" + "--tui/--ask requires an interactive terminal" ); let code = tui::run(args, audit)?; if code != 0 { @@ -92,6 +97,8 @@ pub fn main() -> anyhow::Result<()> { } } match parsed.command { + #[cfg(unix)] + Command::Cache(args) => crate::cache::run(args)?, Command::Run(args) => { let code = tokio::runtime::Runtime::new()?.block_on(run::run(*args))?; if code != 0 { @@ -138,6 +145,7 @@ fn normalize_default_run(mut args: Vec) -> Vec, - /// Persistent changeset directory for reviewable staged writes. Plain runs - /// without staging write the workspace directly; --safe/--audit stage by - /// default in a temporary stage that is dropped at exit. + /// Changeset directory; defaults to persistent Job storage for review and apply/drop. #[arg(long, value_name = "PATH")] stage: Option, @@ -416,6 +414,7 @@ enum FilesystemLevel { Deny, Ask, Read, + Warn, Stage, Write, } @@ -430,12 +429,10 @@ impl FromStr for FilesystemAccessArg { let level = match level { "deny" => FilesystemLevel::Deny, "ask" => FilesystemLevel::Ask, - "read" => FilesystemLevel::Read, - "stage" => FilesystemLevel::Stage, - "write" => FilesystemLevel::Write, + "warn" => FilesystemLevel::Warn, _ => { return Err(format!( - "invalid access level `{level}`; use deny, ask, read, stage, or write" + "invalid access level `{level}`; use deny, ask, or warn; read-only sharing uses --mount PATH:read" )); } }; @@ -454,10 +451,13 @@ struct OverlayFsOverrides { /// Host path mount: SOURCE[:TARGET]:ACCESS. ACCESS is read, stage, or write. #[arg(long = "mount", value_name = "SOURCE[:TARGET]:ACCESS")] mounts: Vec, - /// Agent-visible path policy: PATH-GLOB:LEVEL. LEVEL is deny, ask, or read; ask opens the audit TUI. + /// Append an overlay-relative rule: PATH-GLOB:deny|ask|warn; ask opens permission prompts. /// Use `--mount` when a path must be staged or writable. #[arg(long = "access", value_name = "PATH-GLOB:LEVEL")] access: Vec, + /// Explicitly remove default/config file rules before adding --access rules. + #[arg(long)] + clear_access: bool, /// Changeset upper-layer backend: `directory` writes plain files, `jujutsu` /// stores content-addressed snapshots in a shared repository. #[arg(long = "overlayfs-backend", value_enum)] @@ -798,77 +798,11 @@ pub async fn run(mut args: RunArgs) -> anyhow::Result { let run_id = format!("run-{}", uuid::Uuid::new_v4()); let mut config = load_run_config(&args, personal_config_root().as_deref(), true)?; apply_run_options(&mut config, args.clone())?; - let stage_limit = config - .overlayfs - .as_ref() - .and_then(|overlay| overlay.stage_size_bytes); - let cleanup_stage = ensure_default_stage(&mut config, &args, &run_id); - let effective_stage = config - .overlayfs - .as_ref() - .and_then(|overlay| overlay.stage.clone()); + normalize_filesystem_config(&mut config)?; if args.run.safe || args.audit { warn_safe_preset(&config, &args); } - let mut result = - execute_config(config, run_id.clone(), args.run.safe || args.audit, None).await; - if result.is_ok() - && let Some(limit) = stage_limit - && let Some(path) = effective_stage - && path.exists() - { - match directory_size_bytes(&path) { - Ok(actual) if actual > limit => { - result = Err(anyhow::anyhow!( - "stage size limit exceeded: {} uses {} bytes (limit {})", - path.display(), - actual, - limit - )); - } - Err(error) => { - result = Err(error).context("measure OverlayFS stage size"); - } - _ => {} - } - } - if let Some(path) = cleanup_stage - && path.exists() - && let Err(error) = std::fs::remove_dir_all(&path) - { - if result.is_ok() { - return Err(error) - .with_context(|| format!("remove temporary stage {}", path.display())); - } - run_log!( - "pVisor warning: failed to remove temporary stage {}: {error}", - path.display() - ); - } - result -} - -fn ensure_default_stage(config: &mut RunConfig, args: &RunArgs, run_id: &str) -> Option { - if !args.run.safe && !args.audit { - return None; - } - let existing = args.stage.clone().or_else(|| { - config - .overlayfs - .as_ref() - .and_then(|filesystem| filesystem.stage.clone()) - }); - let temporary = existing - .is_none() - .then(|| std::env::temp_dir().join(format!("pvisor-stage-{run_id}"))); - let stage = existing - .or_else(|| temporary.clone()) - .expect("safe stage selected"); - config - .overlayfs - .get_or_insert_with(OverlayFsSettings::default) - .stage = Some(stage); - temporary + execute_config(config, run_id, args.run.safe || args.audit, None).await } fn directory_size_bytes(root: &Path) -> anyhow::Result { @@ -900,7 +834,7 @@ fn directory_size_bytes(root: &Path) -> anyhow::Result { async fn run_prepared_spec(args: RunArgs) -> anyhow::Result { anyhow::ensure!( !args.run.safe && !args.audit, - "--safe/--audit cannot modify a prepared JSON RunSpec" + "--safe/--ask cannot modify a prepared JSON RunSpec" ); anyhow::ensure!( args.command.is_empty(), @@ -1229,6 +1163,7 @@ async fn execute_config( ) -> anyhow::Result { normalize_filesystem_config(&mut config)?; resolve_default_vm_rootfs(&mut config)?; + let mut _image_mount: Option = None; let prepared_image = if config.run.executor == RunExecutorKind::Vm && config.vm.rootfs.is_none() { let image = config @@ -1238,11 +1173,11 @@ async fn execute_config( .context("VM image must be explicitly configured")?; let store = config.vm.image_store.clone(); run_log!("pVisor image: resolving {image}"); - let prepared = tokio::task::spawn_blocking(move || { - crate::oci::ImageStore::new(store)?.prepare(&image) - }) - .await - .context("OCI image preparation task failed")??; + let (prepared, mount) = + tokio::task::spawn_blocking(move || crate::cache::prepare_image(&image, store)) + .await + .context("OCI image preparation task failed")??; + _image_mount = mount; run_log!( "pVisor image: {} ({})", prepared.digest, @@ -1810,20 +1745,26 @@ fn normalize_filesystem_config(config: &mut RunConfig) -> anyhow::Result<()> { } } match mount.access { - FilesystemAccessLevel::Deny | FilesystemAccessLevel::Ask => anyhow::bail!( - "filesystem mount access `deny` or `ask` is invalid; use an access rule instead" + FilesystemAccessLevel::Deny + | FilesystemAccessLevel::Ask + | FilesystemAccessLevel::Warn => anyhow::bail!( + "filesystem mounts require read, stage, or write; use --access for deny, ask, or warn" ), - FilesystemAccessLevel::Read | FilesystemAccessLevel::Stage => { + FilesystemAccessLevel::Stage => { filesystem.compose.push(mount.source); } - FilesystemAccessLevel::Write => { + FilesystemAccessLevel::Read | FilesystemAccessLevel::Write => { anyhow::ensure!( target == mount.source, - "write mounts require target to equal source" + "read/write shares require target to equal source" ); config.run.filesystem.push(FilesystemCapability { path: mount.source.display().to_string(), - access: FilesystemAccess::ReadWrite, + access: if mount.access == FilesystemAccessLevel::Read { + FilesystemAccess::Read + } else { + FilesystemAccess::ReadWrite + }, }); } } @@ -1851,7 +1792,7 @@ fn normalize_filesystem_config(config: &mut RunConfig) -> anyhow::Result<()> { filesystem.access_policy.warn().to_vec(), )?; } - FilesystemAccessLevel::Read => { + FilesystemAccessLevel::Warn => { let mut warn = filesystem.access_policy.warn().to_vec(); warn.push(normalize_policy_glob(&rule.path)); filesystem.access_policy = persisting_control::FileAccessPolicy::new_with_ask( @@ -1860,8 +1801,10 @@ fn normalize_filesystem_config(config: &mut RunConfig) -> anyhow::Result<()> { warn, )?; } - FilesystemAccessLevel::Stage | FilesystemAccessLevel::Write => anyhow::bail!( - "filesystem access rules can only reduce access; use a mount for stage or write" + FilesystemAccessLevel::Read + | FilesystemAccessLevel::Stage + | FilesystemAccessLevel::Write => anyhow::bail!( + "filesystem access rules use deny, ask, or warn; use --mount for read, stage, or write" ), } } @@ -1950,22 +1893,6 @@ fn free_loopback_address() -> anyhow::Result { /// Resolve ordinary defaults/config, then the opt-in preset, then explicit CLI values. fn apply_run_options(config: &mut RunConfig, args: RunArgs) -> anyhow::Result<()> { if args.run.safe || args.audit { - let configured_ask = if args.audit { - config - .overlayfs - .as_ref() - .map(|filesystem| { - filesystem - .access - .iter() - .filter(|rule| rule.level == FilesystemAccessLevel::Ask) - .cloned() - .collect::>() - }) - .unwrap_or_default() - } else { - Vec::new() - }; // Resolve the actual command/executor/routes first, without mistaking --name for an Agent. let mut requested = config.clone(); apply_cli(&mut requested, args.clone())?; @@ -1985,9 +1912,6 @@ fn apply_run_options(config: &mut RunConfig, args: RunArgs) -> anyhow::Result<() // installs the synthetic root/Landlock or Seatbelt write controls. // An explicit --filesystem value in `args` still wins below. config.filesystem = FilesystemMode::Sandbox; - if let Some(filesystem) = config.overlayfs.as_mut() { - filesystem.access.extend(configured_ask); - } } apply_cli(config, args.clone())?; apply_safe_defaults(config)?; @@ -2242,6 +2166,7 @@ fn apply_cli(config: &mut RunConfig, args: RunArgs) -> anyhow::Result<()> { let enables_overlayfs = !args.overlayfs.mounts.is_empty() || !args.overlayfs.access.is_empty() + || args.overlayfs.clear_access || args.overlayfs.backend.is_some() || args.overlayfs.max_size.is_some() || args.stage.is_some(); @@ -2249,23 +2174,26 @@ fn apply_cli(config: &mut RunConfig, args: RunArgs) -> anyhow::Result<()> { let overlayfs = config .overlayfs .get_or_insert_with(OverlayFsSettings::default); - if !args.overlayfs.access.is_empty() { + if args.overlayfs.clear_access { overlayfs.access_policy = Default::default(); - overlayfs.access = args - .overlayfs + overlayfs.access.clear(); + } + if !args.overlayfs.access.is_empty() { + overlayfs .access - .into_iter() - .map(|access| crate::config::FilesystemAccessRule { - path: access.path, - level: match access.level { - FilesystemLevel::Deny => FilesystemAccessLevel::Deny, - FilesystemLevel::Ask => FilesystemAccessLevel::Ask, - FilesystemLevel::Read => FilesystemAccessLevel::Read, - FilesystemLevel::Stage => FilesystemAccessLevel::Stage, - FilesystemLevel::Write => FilesystemAccessLevel::Write, - }, - }) - .collect(); + .extend(args.overlayfs.access.into_iter().map(|access| { + crate::config::FilesystemAccessRule { + path: access.path, + level: match access.level { + FilesystemLevel::Deny => FilesystemAccessLevel::Deny, + FilesystemLevel::Ask => FilesystemAccessLevel::Ask, + FilesystemLevel::Read => FilesystemAccessLevel::Read, + FilesystemLevel::Warn => FilesystemAccessLevel::Warn, + FilesystemLevel::Stage => FilesystemAccessLevel::Stage, + FilesystemLevel::Write => FilesystemAccessLevel::Write, + }, + } + })); } if !args.overlayfs.mounts.is_empty() { overlayfs.compose.clear(); @@ -2280,6 +2208,7 @@ fn apply_cli(config: &mut RunConfig, args: RunArgs) -> anyhow::Result<()> { FilesystemLevel::Deny => FilesystemAccessLevel::Deny, FilesystemLevel::Ask => FilesystemAccessLevel::Ask, FilesystemLevel::Read => FilesystemAccessLevel::Read, + FilesystemLevel::Warn => FilesystemAccessLevel::Warn, FilesystemLevel::Stage => FilesystemAccessLevel::Stage, FilesystemLevel::Write => FilesystemAccessLevel::Write, }, @@ -2449,6 +2378,14 @@ fn validate_vm_rootfs_platform(config: &RunConfig) -> anyhow::Result<()> { } fn validate(config: &RunConfig, safe: bool) -> anyhow::Result<()> { + anyhow::ensure!( + safe || !config + .run + .filesystem + .iter() + .any(|grant| grant.access == FilesystemAccess::Read), + "read-only shares require --safe or --ask" + ); if let Some(filesystem) = &config.overlayfs { anyhow::ensure!( filesystem.compose.is_empty() || filesystem.commit != OverlayFsCommit::Apply, @@ -2901,36 +2838,75 @@ mod tests { } #[test] - fn safe_and_audit_runs_create_an_implicit_workspace_stage() { - let mut normal = RunConfig::default(); - let normal_args = preset_args(&["--", "bash"]); - assert!(ensure_default_stage(&mut normal, &normal_args, "normal").is_none()); - assert!(normal.overlayfs.is_none(), "stage is opt-in"); - assert_eq!(normal.filesystem, FilesystemMode::Host); - - let mut safe = RunConfig::default(); - let safe_args = preset_args(&["--safe", "--", "bash"]); - let temporary = ensure_default_stage(&mut safe, &safe_args, "safe").unwrap(); - assert_eq!(safe.overlayfs.unwrap().stage, Some(temporary)); - - let mut audit = RunConfig::default(); - let audit_args = preset_args(&["--audit", "--", "codex"]); - assert!(audit_args.tui_requested()); - assert!(audit_args.wants_tui(true)); - let temporary = ensure_default_stage(&mut audit, &audit_args, "audit").unwrap(); - assert_eq!(audit.overlayfs.unwrap().stage, Some(temporary)); + fn safe_and_ask_runs_use_persistent_job_storage() { + let temporary = tempfile::tempdir().unwrap(); + let workspace = temporary.path().join("workspace"); + let storage = temporary.path().join("job"); + std::fs::create_dir(&workspace).unwrap(); + for flag in ["--safe", "--ask"] { + let mut config = RunConfig::default(); + apply_run_options(&mut config, preset_args(&[flag, "--", "bash"])).unwrap(); + normalize_filesystem_config(&mut config).unwrap(); + let overlay = resolve_overlay(&config, &workspace, &storage, "job") + .unwrap() + .unwrap(); + assert_eq!(overlay.stage_dir, Some(storage.canonicalize().unwrap())); + assert!(!overlay.auto_discard); + assert!( + config + .overlayfs + .unwrap() + .access_policy + .denied(Path::new(".ssh/key")) + ); + } + } + + #[test] + fn ask_enables_permission_prompts_not_an_agent_command() { + let args = preset_args(&["--tui", "--ask", "--", "bash"]); + assert!(args.audit_requested().unwrap()); + assert!(args.audit); + assert_eq!(args.command, ["bash"]); + } + + #[test] + fn replacing_file_defaults_requires_an_explicit_clear() { let mut config = RunConfig::default(); - apply_run_options(&mut config, audit_args).unwrap(); - assert_eq!(config.overlaynet.policy, OverlayNetPolicy::Allowlist); + apply_run_options( + &mut config, + preset_args(&[ + "--ask", + "--clear-access", + "--access", + "custom:ask", + "--", + "bash", + ]), + ) + .unwrap(); normalize_filesystem_config(&mut config).unwrap(); - assert!(!config.overlayfs.unwrap().access_policy.ask().is_empty()); + let policy = &config.overlayfs.unwrap().access_policy; + assert_eq!(policy.ask(), ["custom"]); + assert!(policy.deny().is_empty()); + assert!("secret:read".parse::().is_err()); + assert!("secret:warn".parse::().is_ok()); + } - let mut retained = RunConfig::default(); - let directory = tempfile::tempdir().unwrap(); - let stage = directory.path().join("retained-stage"); - let args = preset_args(&["--safe", "--stage", stage.to_str().unwrap(), "--", "bash"]); - assert!(ensure_default_stage(&mut retained, &args, "retained").is_none()); - assert_eq!(retained.overlayfs.unwrap().stage, Some(stage)); + #[test] + fn read_shares_are_read_only_capabilities_not_overlay_layers() { + let mut config = RunConfig::default(); + apply_run_options( + &mut config, + preset_args(&["--safe", "--mount", "/reference:read", "--", "bash"]), + ) + .unwrap(); + normalize_filesystem_config(&mut config).unwrap(); + assert!(config.overlayfs.as_ref().unwrap().compose.is_empty()); + assert_eq!(config.run.filesystem[0].access, FilesystemAccess::Read); + assert_eq!(config.run.filesystem[0].path, "/reference"); + assert!(validate(&config, false).is_err()); + assert!(validate(&config, true).is_ok()); } #[test] @@ -3156,14 +3132,14 @@ path = "configured-secret" level = "deny" [[overlayfs.access]] path = "configured-warning" -level = "read" +level = "warn" "#; let mut config: RunConfig = toml::from_str(source).unwrap(); apply_run_options(&mut config, preset_args(&["--safe", "--", "codex"])).unwrap(); normalize_filesystem_config(&mut config).unwrap(); let policy = &config.overlayfs.as_ref().unwrap().access_policy; assert!(policy.deny().contains(&"**/.ssh".into())); - assert!(!policy.deny().contains(&"configured-secret".into())); + assert!(policy.deny().contains(&"configured-secret".into())); let mut config: RunConfig = toml::from_str(source).unwrap(); apply_run_options( @@ -3171,7 +3147,7 @@ level = "read" preset_args(&[ "--safe", "--access", - "custom/*.pem:read", + "custom/*.pem:warn", "--access", "private/**:deny", "--", @@ -3181,8 +3157,9 @@ level = "read" .unwrap(); normalize_filesystem_config(&mut config).unwrap(); let policy = &config.overlayfs.as_ref().unwrap().access_policy; - assert_eq!(policy.deny(), ["private/**"]); - assert_eq!(policy.warn(), ["custom/*.pem"]); + assert!(policy.deny().contains(&"private/**".into())); + assert!(policy.denied(Path::new(".ssh/key"))); + assert!(policy.warn().contains(&"custom/*.pem".into())); let overlay = resolve_overlay( &config, Path::new("."), @@ -3204,12 +3181,14 @@ level = "read" let mut config = RunConfig::default(); apply_run_options( &mut config, - preset_args(&["--audit", "--access", "secrets/*.pem:ask", "--", "codex"]), + preset_args(&["--ask", "--access", "secrets/*.pem:ask", "--", "codex"]), ) .unwrap(); normalize_filesystem_config(&mut config).unwrap(); let policy = &config.overlayfs.as_ref().unwrap().access_policy; - assert_eq!(policy.ask(), ["secrets/*.pem"]); + assert!(policy.ask().contains(&"secrets/*.pem".into())); + assert!(policy.ask().contains(&"**/.env".into())); + assert!(policy.denied(Path::new(".ssh/key"))); assert!(policy.warn().is_empty()); } diff --git a/crates/persisting-pvisor/src/cli/run/safe/files.rs b/crates/persisting-pvisor/src/cli/run/safe/files.rs index 3b4fe01b..aa1f6aef 100644 --- a/crates/persisting-pvisor/src/cli/run/safe/files.rs +++ b/crates/persisting-pvisor/src/cli/run/safe/files.rs @@ -26,7 +26,7 @@ pub(super) fn patch(audit: bool) -> Vec { ] { args.extend([ "--access".into(), - format!("{pattern}:{}", if audit { "ask" } else { "read" }), + format!("{pattern}:{}", if audit { "ask" } else { "warn" }), ]); } args diff --git a/crates/persisting-pvisor/src/cli/tui.rs b/crates/persisting-pvisor/src/cli/tui.rs index 2a37716a..7e298418 100644 --- a/crates/persisting-pvisor/src/cli/tui.rs +++ b/crates/persisting-pvisor/src/cli/tui.rs @@ -2,4 +2,6 @@ mod zellij; -pub(super) use zellij::{announce_stage, available, diagnostic, init_child_context, is_child, run}; +pub(super) use zellij::{announce_stage, available, init_child_context, is_child, run}; + +pub(crate) use zellij::diagnostic; diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs b/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs index 96f8fb69..fe9c9499 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs @@ -9,7 +9,7 @@ use std::collections::VecDeque; use std::io::{BufRead, BufReader, Read, Write}; use std::net::IpAddr; use std::os::unix::net::{UnixListener, UnixStream}; -use std::path::Path; +use std::path::{Path, PathBuf}; const POLICY_FILE: &str = "audit-policy.json"; @@ -152,11 +152,13 @@ impl SessionPolicy { let mut bytes = Vec::new(); source.read_to_end(&mut bytes)?; let policy: Self = serde_json::from_slice(&bytes)?; - anyhow::ensure!( - policy.schema_version == 1, - "unsupported audit policy version" - ); - for rule in &policy.rules { + policy.validate()?; + Ok(policy) + } + + fn validate(&self) -> Result<()> { + anyhow::ensure!(self.schema_version == 1, "unsupported audit policy version"); + for rule in &self.rules { let valid = match rule.kind { AuditKind::File => { matches!(rule.scope, Scope::Exact | Scope::Directory | Scope::Suffix) @@ -175,7 +177,7 @@ impl SessionPolicy { }; anyhow::ensure!(valid, "invalid session audit rule"); } - Ok(policy) + Ok(()) } pub fn persist(&self, storage: &Path) -> Result<()> { @@ -203,6 +205,13 @@ impl SessionPolicy { ) -> Option { let rule = SessionRule::from_request(request, scope, decision)?; let mut next = self.clone(); + next.rules.retain(|old| { + !(old.kind == rule.kind + && old.scope == rule.scope + && old.value == rule.value + && old.port == rule.port + && old.transport == rule.transport) + }); next.rules.push(rule); Some(next) } @@ -212,6 +221,265 @@ impl SessionPolicy { } } +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub(super) enum Lifetime { + #[default] + Session, + Workspace, + User, +} + +impl Lifetime { + pub fn label(self) -> &'static str { + match self { + Self::Session => "session", + Self::Workspace => "workspace", + Self::User => "user", + } + } + + pub fn key(byte: u8) -> Option { + match byte { + b's' => Some(Self::Session), + b'w' => Some(Self::Workspace), + b'u' => Some(Self::User), + _ => None, + } + } +} + +pub(super) fn permissions_config_path() -> Result { + let root = std::env::var_os("XDG_CONFIG_HOME") + .map(PathBuf::from) + .filter(|path| path.is_absolute()) + .or_else(|| dirs::home_dir().map(|home| home.join(".config"))) + .context("personal configuration directory unavailable")?; + Ok(root.join("pvisor/config.toml")) +} + +#[derive(Default, Serialize, Deserialize)] +#[serde(default, deny_unknown_fields)] +struct SavedPermissions { + user: SessionPolicy, + workspaces: std::collections::BTreeMap, +} + +impl SavedPermissions { + fn read(path: &Path) -> Result<(toml_edit::DocumentMut, Self)> { + let source = match std::fs::read_to_string(path) { + Ok(source) => source, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => String::new(), + Err(error) => return Err(error.into()), + }; + let doc = source.parse::()?; + let config: toml::Table = toml::from_str(&source)?; + let saved: Self = config + .get("permissions") + .cloned() + .map(toml::Value::try_into) + .transpose()? + .unwrap_or_default(); + saved.user.validate()?; + for policy in saved.workspaces.values() { + policy.validate()?; + } + Ok((doc, saved)) + } +} + +pub(super) struct Permissions { + session: SessionPolicy, + saved: SavedPermissions, + config: PathBuf, + workspace: String, + file_root: PathBuf, +} + +impl Permissions { + pub fn load( + storage: &Path, + workspace: &Path, + file_root: &Path, + config: PathBuf, + ) -> Result { + Ok(Self { + session: SessionPolicy::load(storage)?, + saved: SavedPermissions::read(&config)?.1, + config, + workspace: workspace + .canonicalize()? + .to_str() + .context("workspace must be UTF-8")? + .into(), + file_root: file_root.canonicalize()?, + }) + } + + // Ask paths are relative to the overlay target, not necessarily the cwd. + // Persist original absolute paths so a user rule cannot approve a same-named + // file in an unrelated workspace merely because its relative path matches. + fn persistent_request(&self, request: &AuditRequest) -> Result { + let mut request = request.clone(); + if request.kind == AuditKind::File { + let path = Path::new(&request.target); + anyhow::ensure!( + !path + .components() + .any(|c| matches!(c, std::path::Component::ParentDir)), + "audit path contains parent traversal" + ); + request.target = self + .file_root + .join(path) + .to_str() + .context("audit path must be UTF-8")? + .into(); + } + Ok(request) + } + + pub fn resolve(&self, request: &AuditRequest) -> Option<(AuditDecision, Scope, Lifetime)> { + if let Some((decision, scope)) = self.session.resolve(request) { + return Some((decision, scope, Lifetime::Session)); + } + let request = self.persistent_request(request).ok()?; + if let Some((decision, scope)) = self + .saved + .workspaces + .get(&self.workspace) + .and_then(|p| p.resolve(&request)) + { + return Some((decision, scope, Lifetime::Workspace)); + } + self.saved + .user + .resolve(&request) + .map(|(d, s)| (d, s, Lifetime::User)) + } + + pub fn remember( + &mut self, + storage: &Path, + request: &AuditRequest, + scope: Scope, + decision: AuditDecision, + lifetime: Lifetime, + ) -> Result<()> { + if lifetime == Lifetime::Session { + let next = self + .session + .with_decision(request, scope, decision) + .context("invalid audit scope")?; + next.persist(storage)?; + self.session = next; + return Ok(()); + } + let request = self.persistent_request(request)?; + self.update_saved(|saved, workspace| { + let policy = match lifetime { + Lifetime::Workspace => saved.workspaces.entry(workspace.into()).or_default(), + Lifetime::User => &mut saved.user, + Lifetime::Session => unreachable!(), + }; + *policy = policy + .with_decision(&request, scope, decision) + .context("invalid audit scope")?; + Ok(()) + }) + } + + fn update_saved( + &mut self, + update: impl FnOnce(&mut SavedPermissions, &str) -> Result<()>, + ) -> Result<()> { + let parent = self + .config + .parent() + .context("configuration path has no parent")?; + std::fs::create_dir_all(parent)?; + use std::os::unix::fs::OpenOptionsExt; + let lock = std::fs::OpenOptions::new() + .create(true) + .truncate(false) + .read(true) + .write(true) + .mode(0o600) + .custom_flags(libc::O_NOFOLLOW | libc::O_CLOEXEC) + .open(parent.join("permissions.lock"))?; + fs2::FileExt::lock_exclusive(&lock)?; + // Re-read under the lock: simultaneous TUIs must not overwrite each other. + let (mut doc, mut saved) = SavedPermissions::read(&self.config)?; + update(&mut saved, &self.workspace)?; + let encoded = toml::to_string(&saved)?.parse::()?; + doc["permissions"] = toml_edit::Item::Table(encoded.as_table().clone()); + let mut file = tempfile::NamedTempFile::new_in(parent)?; + file.write_all(doc.to_string().as_bytes())?; + file.as_file().sync_all()?; + file.persist(&self.config)?; + std::fs::File::open(parent)?.sync_all()?; + self.saved = saved; + Ok(()) + } + + fn rule_entries(&self) -> Vec<(Lifetime, SessionRule)> { + [ + (Lifetime::Session, Some(&self.session)), + ( + Lifetime::Workspace, + self.saved.workspaces.get(&self.workspace), + ), + (Lifetime::User, Some(&self.saved.user)), + ] + .into_iter() + .flat_map(|(lifetime, policy)| { + policy + .into_iter() + .flat_map(move |p| p.rules.iter().rev().cloned().map(move |r| (lifetime, r))) + }) + .collect() + } + + pub fn rule_labels(&self) -> Vec { + self.rule_entries() + .iter() + .map(|(lifetime, rule)| format!("[{}] {}", lifetime.label(), rule.label())) + .collect() + } + + /// Remove the selected decision, not whatever occupies its index after another writer saves. + pub fn forget(&mut self, storage: &Path, index: usize) -> Result<()> { + let (lifetime, rule) = self + .rule_entries() + .get(index) + .cloned() + .context("no decision selected")?; + if lifetime == Lifetime::Session { + let mut next = self.session.clone(); + next.rules.retain(|r| r != &rule); + next.persist(storage)?; + self.session = next; + return Ok(()); + } + self.update_saved(|saved, workspace| { + let policy = match lifetime { + Lifetime::Workspace => saved.workspaces.get_mut(workspace), + Lifetime::User => Some(&mut saved.user), + Lifetime::Session => unreachable!(), + }; + if let Some(policy) = policy { + policy.rules.retain(|r| r != &rule); + } + Ok(()) + }) + } + + pub fn display_request(&self, request: &AuditRequest) -> AuditRequest { + self.persistent_request(request) + .unwrap_or_else(|_| request.clone()) + } +} + pub(super) fn choice(request: &AuditRequest, byte: u8) -> Option<(Scope, AuditDecision)> { let scope = match byte { b'1' | b'a' | b'A' => Scope::Exact, @@ -225,6 +493,95 @@ pub(super) fn choice(request: &AuditRequest, byte: u8) -> Option<(Scope, AuditDe .map(|_| (scope, AuditDecision::Allow)) } +#[derive(Default)] +pub(super) struct Prompt { + pub lifetime: Lifetime, + pub scope: Option, + pub focus: u8, + escape: u8, +} + +impl Prompt { + pub fn input(&mut self, request: &AuditRequest, byte: u8) -> Option<(Scope, AuditDecision)> { + // Consume complete cursor sequences, even when reads split their bytes. + if self.escape != 0 { + if self.escape == 1 && matches!(byte, b'[' | b'O') { + self.escape = 2; + return None; + } + let cursor = self.escape == 2; + self.escape = 0; + if cursor { + match byte { + b'A' => self.focus = (self.focus + 2) % 3, + b'B' => self.focus = (self.focus + 1) % 3, + b'C' | b'D' => { + let step = if byte == b'C' { 1 } else { -1 }; + match self.focus { + 0 => { + let scopes: Vec<_> = (*b"123") + .into_iter() + .filter_map(|key| choice(request, key).map(|v| v.0)) + .collect(); + let index = scopes + .iter() + .position(|s| Some(*s) == self.scope) + .unwrap_or(0); + self.scope = Some( + scopes[(index as isize + step).rem_euclid(scopes.len() as isize) + as usize], + ); + } + 1 => { + let index = match self.lifetime { + Lifetime::Session => 0, + Lifetime::Workspace => 1, + Lifetime::User => 2, + }; + self.lifetime = + [Lifetime::Session, Lifetime::Workspace, Lifetime::User] + [(index + step).rem_euclid(3) as usize]; + } + _ => { + self.scope = if self.scope.is_some() { + None + } else { + Some(Scope::Exact) + } + } + } + } + _ => {} + } + return None; + } + } + match byte { + 0x1b => { + self.escape = 1; + } + b'\t' => self.focus = (self.focus + 1) % 3, + b'\r' | b'\n' if self.focus != 2 => self.focus += 1, + b'\r' | b'\n' => { + return Some(self.scope.map_or((Scope::Exact, AuditDecision::Deny), |s| { + (s, AuditDecision::Allow) + })); + } + b'd' | b'D' => return Some((Scope::Exact, AuditDecision::Deny)), + _ => { + if let Some(lifetime) = Lifetime::key(byte) { + self.lifetime = lifetime; + } + if let Some((scope, AuditDecision::Allow)) = choice(request, byte) { + self.scope = Some(scope); + self.focus = 2; + } + } + } + None + } +} + struct Pending { request: AuditRequest, stream: UnixStream, @@ -352,6 +709,35 @@ mod tests { assert_eq!(pending.read(&mut byte).unwrap(), 0); } + #[test] + fn prompt_navigation_never_grants_without_confirming_the_button() { + let request = file("private/one.txt"); + let mut prompt = Prompt::default(); + for byte in b"\x1b[C\t\x1b[C" { + assert_eq!(prompt.input(&request, *byte), None); + } + assert_eq!(prompt.scope, Some(Scope::Directory)); + assert_eq!(prompt.lifetime, Lifetime::Workspace); + assert_eq!(prompt.input(&request, b'\t'), None); + assert_eq!( + prompt.input(&request, b'\r'), + Some((Scope::Directory, AuditDecision::Allow)) + ); + let mut prompt = Prompt::default(); + for byte in b"\t\t" { + assert_eq!(prompt.input(&request, *byte), None); + } + assert_eq!( + prompt.input(&request, b'\r'), + Some((Scope::Exact, AuditDecision::Deny)) + ); + assert_eq!(prompt.input(&request, b'2'), None); + assert_eq!( + prompt.input(&request, b'\r'), + Some((Scope::Directory, AuditDecision::Allow)) + ); + } + fn file(target: &str) -> AuditRequest { AuditRequest { kind: AuditKind::File, @@ -374,6 +760,219 @@ mod tests { } } + #[test] + fn persistent_permissions_isolate_workspaces_and_keep_user_paths_absolute() { + let temp = tempfile::tempdir().unwrap(); + let a = temp.path().join("a"); + let b = temp.path().join("b"); + let first = temp.path().join("first"); + let second = temp.path().join("second"); + for path in [&a, &b, &first, &second] { + std::fs::create_dir(path).unwrap(); + } + let config = temp.path().join("config.toml"); + std::fs::write( + &config, + "# keep my settings\n[run]\nname = 'example' # keep this too\n", + ) + .unwrap(); + let mut policy = Permissions::load(&first, &a, &a, config.clone()).unwrap(); + policy + .remember( + &first, + &file("private/key.txt"), + Scope::Directory, + AuditDecision::Allow, + Lifetime::Workspace, + ) + .unwrap(); + policy + .remember( + &first, + &file("user.txt"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::User, + ) + .unwrap(); + policy + .remember( + &first, + &file("session.txt"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::Session, + ) + .unwrap(); + let same = Permissions::load(&second, &a, &a, config.clone()).unwrap(); + assert_eq!( + same.resolve(&file("private/next.txt")), + Some((AuditDecision::Allow, Scope::Directory, Lifetime::Workspace)) + ); + assert!(same.resolve(&file("session.txt")).is_none()); + let other = Permissions::load(&second, &b, &b, config.clone()).unwrap(); + assert!(other.resolve(&file("private/next.txt")).is_none()); + assert!(other.resolve(&file("user.txt")).is_none()); + let absolute = a.canonicalize().unwrap().join("user.txt"); + assert_eq!( + other.resolve(&file(absolute.to_str().unwrap())), + Some((AuditDecision::Allow, Scope::Exact, Lifetime::User)) + ); + let saved = std::fs::read_to_string(&config).unwrap(); + assert!(saved.contains("# keep my settings")); + assert!(saved.contains("name = 'example' # keep this too")); + use std::os::unix::fs::PermissionsExt; + assert_eq!( + std::fs::metadata(config).unwrap().permissions().mode() & 0o777, + 0o600 + ); + } + + #[test] + fn persistent_permissions_merge_writers_and_prefer_narrower_lifetimes() { + let temp = tempfile::tempdir().unwrap(); + let first = temp.path().join("first"); + let second = temp.path().join("second"); + std::fs::create_dir(&first).unwrap(); + std::fs::create_dir(&second).unwrap(); + let config = temp.path().join("config.toml"); + let mut one = Permissions::load(&first, temp.path(), temp.path(), config.clone()).unwrap(); + let mut two = Permissions::load(&second, temp.path(), temp.path(), config.clone()).unwrap(); + one.remember( + &first, + &file("one.txt"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::User, + ) + .unwrap(); + two.remember( + &second, + &file("two.txt"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::Workspace, + ) + .unwrap(); + let mut loaded = Permissions::load(&first, temp.path(), temp.path(), config).unwrap(); + assert!(loaded.resolve(&file("one.txt")).is_some()); + assert!(loaded.resolve(&file("two.txt")).is_some()); + loaded + .remember( + &first, + &file("one.txt"), + Scope::Exact, + AuditDecision::Deny, + Lifetime::Workspace, + ) + .unwrap(); + assert_eq!( + loaded.resolve(&file("one.txt")).unwrap().0, + AuditDecision::Deny + ); + loaded + .remember( + &first, + &file("one.txt"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::Session, + ) + .unwrap(); + assert_eq!( + loaded.resolve(&file("one.txt")).unwrap().2, + Lifetime::Session + ); + assert!( + loaded + .remember( + &first, + &file("../escape"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::User + ) + .is_err() + ); + } + + #[test] + fn malformed_persistent_permissions_are_not_overwritten_or_applied() { + let temp = tempfile::tempdir().unwrap(); + let config = temp.path().join("config.toml"); + let mut policy = + Permissions::load(temp.path(), temp.path(), temp.path(), config.clone()).unwrap(); + for bad in [ + "[broken", + "[permissions.user]\nschema_version = 999\nrules = []\n", + ] { + std::fs::write(&config, bad).unwrap(); + assert!( + Permissions::load(temp.path(), temp.path(), temp.path(), config.clone()).is_err() + ); + assert!( + policy + .remember( + temp.path(), + &file("secret"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::User + ) + .is_err() + ); + assert!(policy.resolve(&file("secret")).is_none()); + assert_eq!(std::fs::read_to_string(&config).unwrap(), bad); + } + } + + #[test] + fn forgetting_a_selected_decision_keeps_other_writers_and_scopes() { + let temp = tempfile::tempdir().unwrap(); + let config = temp.path().join("config.toml"); + let mut policy = + Permissions::load(temp.path(), temp.path(), temp.path(), config.clone()).unwrap(); + for (target, lifetime) in [ + ("session", Lifetime::Session), + ("workspace", Lifetime::Workspace), + ("user", Lifetime::User), + ] { + policy + .remember( + temp.path(), + &file(target), + Scope::Exact, + AuditDecision::Allow, + lifetime, + ) + .unwrap(); + } + let mut other = + Permissions::load(temp.path(), temp.path(), temp.path(), config.clone()).unwrap(); + other + .remember( + temp.path(), + &file("other"), + Scope::Exact, + AuditDecision::Allow, + Lifetime::User, + ) + .unwrap(); + policy.forget(temp.path(), 0).unwrap(); + assert!(policy.resolve(&file("session")).is_none()); + policy.forget(temp.path(), 0).unwrap(); + assert!(policy.resolve(&file("workspace")).is_none()); + let user = policy + .rule_entries() + .iter() + .position(|(_, r)| r.value.ends_with("/user")) + .unwrap(); + policy.forget(temp.path(), user).unwrap(); + let reloaded = Permissions::load(temp.path(), temp.path(), temp.path(), config).unwrap(); + assert!(reloaded.resolve(&file("user")).is_none()); + assert!(reloaded.resolve(&file("other")).is_some()); + } + #[test] fn file_choices_are_scoped_and_persisted_for_this_job() { let request = file("config/dev/token.pem"); diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/input.rs b/crates/persisting-pvisor/src/cli/tui/zellij/input.rs index fe75d638..d7118b97 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/input.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/input.rs @@ -1,6 +1,8 @@ //! Mode-scoped key bindings following Zellij's input-mode/action pattern. //! pVisor binds only keys it implements; Agent mode forwards everything else. +use std::time::{Duration, Instant}; + const PREFIX: u8 = 0x1d; // Ctrl-] #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -63,6 +65,10 @@ enum Action { NextPanel, ScrollUp, ScrollDown, + PageUp, + PageDown, + First, + Last, SendPrefix, } @@ -116,7 +122,11 @@ const BINDINGS: &[Binding] = &[ bind!(Command, b'5', Action::Open(Panel::Log), ""), bind!(Command, b'6', Action::Open(Panel::Permissions), ""), bind!(Panel, b'\t', Action::NextPanel, "Tab View"), - bind!(Panel, b'j', Action::ScrollDown, "j/k Scroll"), + bind!(Panel, b'j', Action::ScrollDown, "↑/↓/Wheel Scroll"), + bind!(Panel, 0x02, Action::PageUp, "PgUp/PgDn Page"), + bind!(Panel, 0x06, Action::PageDown, ""), + bind!(Panel, b'g', Action::First, "Home/End Jump"), + bind!(Panel, b'G', Action::Last, ""), bind!(Panel, b'k', Action::ScrollUp, ""), bind!(Panel, b'1', Action::Open(Panel::Overview), "1-6 Select"), bind!(Panel, b'2', Action::Open(Panel::Files), ""), @@ -165,6 +175,12 @@ pub(super) struct UiState { pub mode: Mode, pub panel: Panel, pub scroll: usize, + pub page_rows: usize, + pub max_scroll: usize, + pub escape: Vec, + pub escape_started: Option, + pub permission: usize, + pub forget_pending: bool, } impl Default for UiState { @@ -173,6 +189,12 @@ impl Default for UiState { mode: Mode::Agent, panel: Panel::Overview, scroll: 0, + page_rows: 1, + max_scroll: usize::MAX, + escape: Vec::new(), + escape_started: None, + permission: 0, + forget_pending: false, } } } @@ -228,8 +250,83 @@ impl UiState { } } + // Escape sequences can arrive in separate stdin reads. A lone Esc closes + // Review only after a short timeout; arrow/mouse bytes never reach the PTY. + pub fn expire_escape(&mut self, now: Instant) -> bool { + if self + .escape_started + .is_some_and(|at| now.duration_since(at) >= Duration::from_millis(150)) + { + if self.escape == [0x1b] { + self.mode = Mode::Agent; + } + self.escape.clear(); + self.escape_started = None; + return true; + } + false + } + + fn navigation(&mut self, action: Action) { + let offset = if self.panel == Panel::Permissions { + &mut self.permission + } else { + &mut self.scroll + }; + *offset = match action { + Action::ScrollUp => offset.saturating_sub(1), + Action::ScrollDown => offset.saturating_add(1), + Action::PageUp => offset.saturating_sub(self.page_rows), + Action::PageDown => offset.saturating_add(self.page_rows), + Action::First => 0, + Action::Last => self.max_scroll, + _ => return, + }; + if self.panel != Panel::Permissions { + *offset = (*offset).min(self.max_scroll); + } + } + /// Returns a byte for the Agent PTY, or consumes it as a UI binding. pub fn input(&mut self, byte: u8) -> Option { + self.forget_pending = false; + if self.panel_open() && (byte == 0x1b || !self.escape.is_empty()) { + if self.escape.is_empty() { + self.escape_started = Some(Instant::now()); + } + self.escape.push(byte); + if self.escape.len() == 1 || (self.escape.len() == 2 && matches!(byte, b'[' | b'O')) { + return None; + } + if self.escape.len() > 32 || (self.escape.len() == 2 && !matches!(byte, b'[' | b'O')) { + self.escape.clear(); + self.escape_started = None; + return None; + } + if (0x40..=0x7e).contains(&byte) { + let action = match self.escape.as_slice() { + b"\x1b[A" | b"\x1bOA" => Some(Action::ScrollUp), + b"\x1b[B" | b"\x1bOB" => Some(Action::ScrollDown), + b"\x1b[5~" => Some(Action::PageUp), + b"\x1b[6~" => Some(Action::PageDown), + b"\x1b[H" | b"\x1bOH" | b"\x1b[1~" | b"\x1b[7~" => Some(Action::First), + b"\x1b[F" | b"\x1bOF" | b"\x1b[4~" | b"\x1b[8~" => Some(Action::Last), + mouse if mouse.starts_with(b"\x1b[<64;") && byte == b'M' => { + Some(Action::ScrollUp) + } + mouse if mouse.starts_with(b"\x1b[<65;") && byte == b'M' => { + Some(Action::ScrollDown) + } + _ => None, + }; + self.escape.clear(); + self.escape_started = None; + if let Some(action) = action { + self.navigation(action); + } + } + return None; + } let Some(binding) = BINDINGS .iter() .find(|binding| binding.mode == self.mode && binding.key == byte) @@ -247,8 +344,12 @@ impl UiState { self.panel = self.panel.next(); self.scroll = 0; } - Action::ScrollUp => self.scroll = self.scroll.saturating_sub(1), - Action::ScrollDown => self.scroll = self.scroll.saturating_add(1), + action @ (Action::ScrollUp + | Action::ScrollDown + | Action::PageUp + | Action::PageDown + | Action::First + | Action::Last) => self.navigation(action), Action::SendPrefix => { self.mode = Mode::Agent; return Some(PREFIX); @@ -262,6 +363,39 @@ impl UiState { mod tests { use super::*; + #[test] + fn review_navigation_consumes_sequences_and_obeys_viewport_bounds() { + let mut state = UiState::default(); + state.input(PREFIX); + state.input(b'l'); + state.page_rows = 10; + state.max_scroll = 23; + for (sequence, expected) in [ + (b"\x1b[B".as_slice(), 1), + (b"\x1b[6~".as_slice(), 11), + (b"\x1b[F".as_slice(), 23), + (b"j".as_slice(), 23), + (b"\x1b[5~".as_slice(), 13), + (b"\x1b[<64;30;12M".as_slice(), 12), + (b"\x1b[<65;30;12M".as_slice(), 13), + (b"\x1b[H".as_slice(), 0), + (b"\x1b[A".as_slice(), 0), + (b"\x1b[99~".as_slice(), 0), + ] { + for byte in sequence { + assert_eq!(state.input(*byte), None); + } + assert_eq!(state.scroll, expected); + assert_eq!(state.mode, Mode::Panel); + } + state.input(0x1b); + assert!(!state.expire_escape(Instant::now())); + assert!(state.expire_escape(Instant::now() + Duration::from_millis(200))); + for byte in b"\x1b[B" { + assert_eq!(state.input(*byte), Some(*byte)); + } + } + #[test] fn bindings_are_unique_within_each_mode() { for (index, binding) in BINDINGS.iter().enumerate() { @@ -287,6 +421,7 @@ mod tests { assert_eq!(state.input(b'j'), None); assert_eq!(state.scroll, 1); assert_eq!(state.input(0x1b), None); + state.expire_escape(Instant::now() + Duration::from_millis(200)); assert_eq!(state.mode, Mode::Agent); assert_eq!(state.input(b'f'), Some(b'f')); } @@ -321,6 +456,7 @@ mod tests { assert_eq!(state.input(b'5'), None); assert_eq!(state.panel, Panel::Log); assert_eq!(state.input(0x1b), None); + state.expire_escape(Instant::now() + Duration::from_millis(200)); assert_eq!(state.input(b'l'), Some(b'l')); } } diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs b/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs index c22d3d14..30e264ee 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs @@ -1,7 +1,7 @@ //! PTY ownership and event loop for the native pane UI. use super::{ - audit_ui::{self, AuditServer, Scope, SessionPolicy}, + audit_ui::{self, AuditServer, Lifetime, Permissions, Prompt, Scope}, input, view, }; use crate::runtime::{RunRecord, control_observations}; @@ -53,6 +53,11 @@ pub(crate) fn init_child_context() { persisting_control::audit::init(socket); } let _ = CHILD_CONTEXT.set(path); + crate::cache::progress::init_output( + log_path + .as_ref() + .map(|path| path.with_extension("image.json")), + ); let _ = LOG_CONTEXT.set(log_path); } @@ -127,6 +132,7 @@ impl Drop for ChildCleanup { #[derive(Default)] pub(super) struct Snapshot { + pub(super) image: Option, pub(super) stage: Option, pub(super) record: Option, pub(super) filesystem: Option, @@ -134,10 +140,16 @@ pub(super) struct Snapshot { pub(super) log: Vec, pub(super) audit: Option, pub(super) audit_rules: Vec, + pub(super) audit_prompt: Prompt, } impl Snapshot { fn refresh(&mut self, stage_file: &Path, log_file: &Path) { + if let Ok(bytes) = std::fs::read(log_file.with_extension("image.json")) + && let Ok(image) = serde_json::from_slice(&bytes) + { + self.image = Some(image); + } if let Ok(contents) = std::fs::read_to_string(log_file) { self.log = contents.lines().map(str::to_owned).collect(); } @@ -229,6 +241,7 @@ fn persist_audit_decision( decision: AuditDecision, scope: Scope, automatic: bool, + lifetime: Lifetime, ) -> Result<()> { let mut file = OpenOptions::new() .create(true) @@ -252,6 +265,7 @@ fn persist_audit_decision( "decision": decision, "scope": scope, "automatic": automatic, + "lifetime": lifetime, }), )?; file.write_all(b"\n")?; @@ -363,9 +377,9 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { stdout.flush()?; let mut parser = vt100::Parser::new(layout.agent_rows, layout.agent_cols, 2000); let mut input_modes = HostInputModes::default(); + let mut review_mouse = false; let mut snapshot = Snapshot::default(); - let mut audit_policy = SessionPolicy::default(); - let mut audit_policy_loaded = false; + let mut audit_policy: Option = None; let started = Instant::now(); let mut last_draw = Instant::now() - Duration::from_secs(1); let mut last_refresh = Instant::now() - Duration::from_secs(1); @@ -381,28 +395,59 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { dirty = true; } if audit.is_some() - && !audit_policy_loaded + && audit_policy.is_none() && let Some(record) = snapshot.record.as_ref() { - audit_policy = - SessionPolicy::load(&record.storage).context("load Job session audit policy")?; - audit_policy_loaded = true; + let workspace = record.workspace.clone().unwrap_or(std::env::current_dir()?); + let file_root = record + .overlay + .as_ref() + .map(|o| o.target.as_path()) + .unwrap_or(&workspace); + audit_policy = Some( + Permissions::load( + &record.storage, + &workspace, + file_root, + audit_ui::permissions_config_path()?, + ) + .context("load audit permissions")?, + ); } if let Some(server) = audit.as_mut() { while let Some(request) = server.active().cloned() { - let Some((decision, scope)) = audit_policy.resolve(&request) else { + let Some((decision, scope, lifetime)) = + audit_policy.as_ref().and_then(|p| p.resolve(&request)) + else { break; }; if let Some(record) = snapshot.record.as_ref() { - let _ = - persist_audit_decision(&record.storage, &request, decision, scope, true); + let _ = persist_audit_decision( + &record.storage, + &request, + decision, + scope, + true, + lifetime, + ); } let _ = server.decide(decision); + snapshot.audit_prompt = Prompt::default(); dirty = true; } } - snapshot.audit_rules = audit_policy.rule_labels(); - snapshot.audit = audit.as_ref().and_then(AuditServer::active).cloned(); + snapshot.audit_rules = audit_policy + .as_ref() + .map(Permissions::rule_labels) + .unwrap_or_default(); + snapshot.audit = audit.as_ref().and_then(AuditServer::active).map(|r| { + audit_policy + .as_ref() + .map_or_else(|| r.clone(), |p| p.display_request(r)) + }); + state.permission = state + .permission + .min(snapshot.audit_rules.len().saturating_sub(1)); let next_size = terminal_size(); if next_size.ws_row != size.ws_row || next_size.ws_col != size.ws_col { size = next_size; @@ -418,11 +463,22 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { last_refresh = Instant::now(); dirty = true; } + dirty |= state.expire_escape(Instant::now()); + let mouse = state.panel_open() && snapshot.audit.is_none(); + if mouse != review_mouse { + stdout.write_all(if mouse { + b"\x1b[?1000h\x1b[?1006h" + } else { + b"\x1b[?1000l\x1b[?1006l" + })?; + review_mouse = mouse; + dirty = true; + } if dirty || last_draw.elapsed() >= Duration::from_secs(1) { view::render( &mut stdout, layout, - &state, + &mut state, parser.screen(), &snapshot, started, @@ -488,29 +544,32 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { if let Some(server) = audit.as_mut() && server.active().is_some() { + if (size.ws_col < 60 || size.ws_row < 16) + && !matches!(*byte, b'd' | b'D' | 0x1b) + { + continue; + } if let Some((scope, decision)) = server .active() - .and_then(|request| audit_ui::choice(request, *byte)) + .and_then(|request| snapshot.audit_prompt.input(request, *byte)) && let Some(request) = server.active().cloned() { - let next = audit_policy.with_decision(&request, scope, decision); + let lifetime = snapshot.audit_prompt.lifetime; let save = snapshot .record .as_ref() .ok_or_else(|| anyhow::anyhow!("Job record is not ready")) .and_then(|record| { - next.as_ref() - .ok_or_else(|| anyhow::anyhow!("invalid audit scope"))? - .persist(&record.storage) + audit_policy + .as_mut() + .context("audit permissions are not loaded")? + .remember(&record.storage, &request, scope, decision, lifetime) }); let actual = if save.is_ok() { decision } else { AuditDecision::Deny }; - if let (Ok(()), Some(next)) = (&save, next) { - audit_policy = next; - } let _ = server.decide(actual); let mut file = OpenOptions::new().append(true).open(&log_file)?; if let Err(error) = save { @@ -523,6 +582,7 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { actual, scope, false, + lifetime, ) { writeln!(file, "audit journal unavailable: {error:#}")?; @@ -532,9 +592,40 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { "audit {:?} {:?}: {:?} {:?} ({:?})", actual, scope, request.kind, request.target, request.reason )?; - snapshot.audit_rules = audit_policy.rule_labels(); + snapshot.audit_rules = audit_policy + .as_ref() + .map(Permissions::rule_labels) + .unwrap_or_default(); snapshot.audit = server.active().cloned(); + snapshot.audit_prompt = Prompt::default(); + } + dirty = true; + continue; + } + if state.panel_open() && state.panel == input::Panel::Permissions && *byte == b'x' { + if state.forget_pending { + let result = snapshot + .record + .as_ref() + .context("Job record is not ready") + .and_then(|record| { + audit_policy + .as_mut() + .context("Permissions are not loaded")? + .forget(&record.storage, state.permission) + }); + if let Err(error) = result { + writeln!( + OpenOptions::new().append(true).open(&log_file)?, + "Cannot forget permission: {error:#}" + )?; + } + snapshot.audit_rules = audit_policy + .as_ref() + .map(Permissions::rule_labels) + .unwrap_or_default(); } + state.forget_pending = !state.forget_pending; dirty = true; continue; } @@ -556,23 +647,122 @@ pub(crate) fn run(args: Vec, audit_enabled: bool) -> Result { } } let status = exited.unwrap_or(child.0.wait()?); + snapshot.refresh(&stage_file, &log_file); // Leave the outer alternate screen before printing the persistent review // location: the Run pane itself disappears with the TUI. - let review_path = snapshot.stage.clone(); drop(stdout); drop(terminal); - if let Some(path) = review_path { - eprintln!("Review: pvisor status --review {}", path.display()); - } + write_exit_report( + &mut std::io::stderr().lock(), + status, + &snapshot, + parser.screen(), + )?; Ok(status .code() .unwrap_or_else(|| 128 + status.signal().unwrap_or(1))) } +fn write_exit_report( + output: &mut impl Write, + status: std::process::ExitStatus, + snapshot: &Snapshot, + screen: &vt100::Screen, +) -> std::io::Result<()> { + if !status.success() { + for line in &snapshot.log { + writeln!(output, "{line}")?; + } + let contents = screen.contents(); + if !contents.trim().is_empty() { + writeln!(output, "{contents}")?; + } + writeln!(output, "pVisor Job failed: {status}")?; + } + if let Some(path) = &snapshot.stage { + writeln!(output, "Review: pvisor status --review {}", path.display())?; + if snapshot.record.as_ref().is_some_and(|r| { + r.overlay + .as_ref() + .is_some_and(|o| !o.auto_discard && !o.auto_apply) + }) { + writeln!( + output, + "Changes retained. Apply: pvisor apply {} | Discard: pvisor drop {}", + path.display(), + path.display() + )?; + } + } + Ok(()) +} + #[cfg(test)] mod tests { use super::*; + #[test] + fn image_transfers_reach_log_panel_without_throttling() { + let directory = tempfile::tempdir().unwrap(); + let log = directory.path().join("diagnostics.log"); + File::create(&log).unwrap(); + LOG_CONTEXT.set(Some(log.clone())).unwrap(); + let downloads = crate::cache::progress::Downloads::new("example:latest"); + downloads.received(b"transfer-log-test/file\nname", 10); + downloads.received(b"transfer-log-test/file\nname", 20); + let mut snapshot = Snapshot::default(); + snapshot.refresh(&directory.path().join("missing"), &log); + let transfers: Vec<_> = snapshot + .log + .iter() + .filter(|line| line.contains("transfer-log-test")) + .collect(); + assert_eq!(transfers.len(), 2); + assert!(transfers[0].contains("transferred 10 bytes from /transfer-log-test/file\\nname")); + assert!(transfers[1].contains("this run: 30 bytes across 1 files")); + } + + #[test] + fn image_progress_is_visible_before_stage_is_announced() { + let directory = tempfile::tempdir().unwrap(); + let log = directory.path().join("diagnostics.log"); + let path = log.with_extension("image.json"); + std::fs::write(&path, br#"{"image":"ubuntu:latest","totals":{"files":100,"bytes":2000},"downloaded_files":2,"downloaded_bytes":10}"#).unwrap(); + let mut snapshot = Snapshot::default(); + snapshot.refresh(&directory.path().join("missing"), &log); + assert!(snapshot.stage.is_none()); + assert_eq!(snapshot.image.as_ref().unwrap().downloaded_files, 2); + std::fs::write(&path, b"invalid").unwrap(); + snapshot.refresh(&directory.path().join("missing"), &log); + assert_eq!(snapshot.image.as_ref().unwrap().downloaded_bytes, 10); + } + + #[test] + fn fast_failure_preserves_diagnostics_and_review_location() { + let temporary = tempfile::tempdir().unwrap(); + let stage_file = temporary.path().join("stage"); + let log_file = temporary.path().join("diagnostics.log"); + std::fs::write(&stage_file, b"/missing/run").unwrap(); + std::fs::write(&log_file, b"resolve Agent executable: missing\n").unwrap(); + let mut snapshot = Snapshot::default(); + snapshot.refresh(&stage_file, &log_file); + let mut parser = vt100::Parser::new(24, 80, 0); + parser.process(b"Error: startup failed\r\n"); + let mut output = Vec::new(); + write_exit_report( + &mut output, + std::process::ExitStatus::from_raw(1 << 8), + &snapshot, + parser.screen(), + ) + .unwrap(); + let output = String::from_utf8(output).unwrap(); + assert!(output.contains("resolve Agent executable: missing")); + assert!(output.contains("Error: startup failed")); + assert!(output.contains("pVisor Job failed: exit status: 1")); + assert!(output.contains("Review: pvisor status --review /missing/run")); + } + #[test] fn bracketed_paste_follows_agent_focus() { let mut parser = vt100::Parser::new(12, 80, 0); diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs b/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs index 6967d6eb..a4b64b5b 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs @@ -17,8 +17,8 @@ const LIGHT: Rgb = (224, 230, 221); const AMBER: Rgb = (255, 174, 102); const RED: Rgb = (255, 111, 111); const DARK: Rgb = (18, 23, 19); -const ARROW: &str = ""; -const LEFT_ARROW: &str = ""; +const ARROW: &str = " "; +const LEFT_ARROW: &str = " "; fn move_to(buf: &mut Vec, row: u16, col: u16) { write!(buf, "\x1b[{row};{col}H").unwrap(); @@ -60,33 +60,80 @@ fn elapsed(started: Instant) -> String { } } -fn metrics(snapshot: &Snapshot, elapsed: &str, available: usize) -> String { - let (hits, effects, denied, failed) = snapshot.file_totals(); - let (allowed, net_denied, net_failed) = snapshot.network_totals(); - let logs = snapshot.log.len(); - let candidates = [ - format!( - " {elapsed} │ FILES {hits} hits · {effects} effects · {denied} denied · {failed} failed │ NET {allowed} allowed · {net_denied} denied · {net_failed} failed │ LOG {logs}" - ), - format!( - " {elapsed} │ FILES {effects} eff/{denied} deny/{failed} fail │ NET {allowed} ok/{net_denied} deny/{net_failed} fail │ LOG {logs}" - ), - format!( - " {elapsed} │ F {effects}e/{denied}d/{failed}f │ N {allowed}a/{net_denied}d/{net_failed}f │ L {logs}" - ), - format!(" {elapsed} F{effects}/{denied} N{allowed}/{net_denied} L{logs}"), - ]; - candidates - .into_iter() - .find(|candidate| UnicodeWidthStr::width(candidate.as_str()) <= available) - .unwrap_or_else(|| { - let fallback = format!(" {elapsed}"); - if UnicodeWidthStr::width(fallback.as_str()) <= available { - fallback +pub(super) fn format_bytes(bytes: u64) -> String { + let mut value = bytes as f64; + for unit in ["B", "KiB", "MiB", "GiB", "TiB"] { + if value < 1024.0 || unit == "TiB" { + return if unit == "B" { + format!("{bytes}B") } else { - String::new() - } - }) + format!("{value:.1}{unit}") + }; + } + value /= 1024.0; + } + unreachable!() +} + +fn image_summary( + image: &crate::cache::progress::ImageProgress, + available: usize, +) -> Option { + let files = image + .totals + .map_or_else(|| "?".into(), |total| total.files.to_string()); + let total = image + .totals + .map_or_else(|| "?".into(), |total| format_bytes(total.bytes)); + let bytes = format_bytes(image.downloaded_bytes); + let cached = format_bytes(image.cached_bytes); + let downloaded_files = image.downloaded_files; + let cached_files = image.cached_files; + [ + format!(" Cached {cached_files} files {cached} | Transferred {downloaded_files} files {bytes} | Total {files} files {total}"), + format!(" Cached {cached_files}/{cached} | Transferred {downloaded_files}/{bytes} | Total {files}/{total}"), + format!(" C:{cached_files}/{} X:{downloaded_files}/{} T:{files}/{}", cached.replace("iB", ""), bytes.replace("iB", ""), total.replace("iB", "")), + ] + .into_iter() + .find(|text| text.width() <= available) +} + +fn metrics(snapshot: &Snapshot, elapsed: &str, available: usize) -> String { + let (_, _, denied, failed) = snapshot.file_totals(); + let (_, net_denied, net_failed) = snapshot.network_totals(); + let mut parts = if let Some(image) = snapshot + .image + .as_ref() + .and_then(|image| image_summary(image, available)) + { + vec![image, elapsed.to_string()] + } else { + vec![format!(" {elapsed}")] + }; + if snapshot.audit.is_some() { + parts.push("Waiting for permission".into()); + } + let paths = snapshot.filesystem.as_ref().map_or(0, |fs| { + fs.paths + .values() + .filter(|ops| ops.values().any(|c| c.effects > 0)) + .count() + }); + if paths > 0 { + parts.push(format!("{paths} touched paths")); + } + if denied + net_denied > 0 { + parts.push(format!("{} denied", denied + net_denied)); + } + if failed + net_failed > 0 { + parts.push(format!("{} failed", failed + net_failed)); + } + while parts.join(" | ").width() > available { + if parts.pop().is_none() { + break; + } + } + parts.join(" | ") } fn state_color(state: &str) -> Rgb { @@ -114,7 +161,11 @@ fn render_metrics(buf: &mut Vec, row: u16, cols: u16, snapshot: &Snapshot, s buf.extend_from_slice(chip.as_bytes()); style(buf, color, BASE, false); buf.extend_from_slice(ARROW.as_bytes()); - let hint = " Ctrl-] MENU "; + let hint = if cols >= 100 { + " Ctrl-] Files / Permissions / Log " + } else { + " Ctrl-] Menu " + }; let hint_width = UnicodeWidthStr::width(hint) + 1; let remaining = usize::from(cols).saturating_sub(chip_width + 1 + hint_width + 1); let summary = metrics(snapshot, &elapsed(started), remaining); @@ -123,7 +174,7 @@ fn render_metrics(buf: &mut Vec, row: u16, cols: u16, snapshot: &Snapshot, s move_to(buf, row, cols - hint_width as u16 + 1); style(buf, TILE_A, BASE, false); buf.extend_from_slice(LEFT_ARROW.as_bytes()); - style(buf, DARK, TILE_A, true); + style(buf, LIGHT, TILE_A, true); buf.extend_from_slice(hint.as_bytes()); buf.extend_from_slice(b"\x1b[0m"); } @@ -233,16 +284,47 @@ mod tests { fn metrics_choose_readable_detail_for_terminal_width() { let snapshot = Snapshot::default(); let wide = metrics(&snapshot, "01:23", 140); - assert!(wide.contains("hits")); - assert!(wide.contains("failed")); + assert_eq!(wide, " 01:23"); let medium = metrics(&snapshot, "01:23", 70); - assert!(medium.contains("FILES")); - assert!(medium.contains("NET")); - assert!(medium.contains("LOG")); + assert_eq!(medium, " 01:23"); let narrow = metrics(&snapshot, "01:23", 27); assert!(UnicodeWidthStr::width(narrow.as_str()) <= 27); } + #[test] + fn image_metrics_fit_and_keep_unknown_totals_distinct_from_zero() { + let mut snapshot = Snapshot { + image: Some(crate::cache::progress::ImageProgress { + cached_files: 12, + cached_bytes: 6 * 1024 * 1024, + downloaded_files: 27, + downloaded_bytes: 4 * 1024 * 1024, + totals: Some(crate::cache::ImageTotals { + files: 1234, + bytes: 80 * 1024 * 1024, + }), + ..Default::default() + }), + ..Default::default() + }; + for width in [40, 55, 100, 160] { + let text = metrics(&snapshot, "01:23", width); + assert!(text.width() <= width); + for value in ["12", "6.0", "27", "4.0", "1234", "80.0"] { + assert!(text.contains(value), "{text}"); + } + } + let wide = metrics(&snapshot, "01:23", 160); + assert!(wide.contains("Cached 12 files 6.0MiB")); + assert!(wide.contains("Transferred 27 files 4.0MiB")); + assert!(wide.contains("Total 1234 files 80.0MiB")); + snapshot.image.as_mut().unwrap().downloaded_files = 0; + snapshot.image.as_mut().unwrap().downloaded_bytes = 0; + assert!(metrics(&snapshot, "01:23", 160).contains("Transferred 0 files 0B")); + snapshot.image.as_mut().unwrap().totals = None; + assert!(metrics(&snapshot, "01:23", 160).contains("Total ? files ?")); + } + #[test] fn shortcut_ribbon_never_exceeds_terminal_width() { for cols in [30, 45, 80, 120] { diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/view.rs b/crates/persisting-pvisor/src/cli/tui/zellij/view.rs index 713f6577..d18917cd 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/view.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/view.rs @@ -4,7 +4,7 @@ use super::{LineStyle, border_glyphs, status_bar}; use anyhow::Result; use std::io::Write; use std::time::Instant; -use unicode_width::{UnicodeWidthChar, UnicodeWidthStr}; +use unicode_width::UnicodeWidthChar; use vt100::{Cell, Color, Screen}; const BAR: &str = "\x1b[48;2;33;36;37;38;2;220;224;220m"; @@ -181,30 +181,63 @@ fn panel_lines(snapshot: &Snapshot, panel: Panel, started: Instant, width: usize let (allowed, net_denied, net_failed) = snapshot.network_totals(); let record = snapshot.record.as_ref(); match panel { - Panel::Overview => vec![ - format!( - "State {}", - record.map_or("starting", |run| run.state.as_str()) - ), - format!("Elapsed {}s", started.elapsed().as_secs()), - format!( - "Agent {}", - record.map_or("pending", |run| run.agent.as_str()) - ), - String::new(), - "FILESYSTEM".into(), - format!(" {effects} effects {denied} denied"), - format!(" {hits} hits {failed} failed"), - String::new(), - "NETWORK".into(), - format!(" {allowed} allowed {net_denied} denied"), - format!(" {net_failed} failed"), - String::new(), - "WORKSPACE".into(), - record - .and_then(|run| run.workspace.as_ref()) - .map_or("pending".into(), |path| path.display().to_string()), - ], + Panel::Overview => { + let mut lines = vec![ + format!( + "State {}", + record.map_or("starting", |run| run.state.as_str()) + ), + format!("Elapsed {}s", started.elapsed().as_secs()), + format!( + "Agent {}", + record.map_or("pending", |run| run.agent.as_str()) + ), + String::new(), + "FILESYSTEM".into(), + format!(" {effects} effects {denied} denied"), + format!(" {hits} hits {failed} failed"), + String::new(), + "NETWORK".into(), + format!(" {allowed} allowed {net_denied} denied"), + format!(" {net_failed} failed"), + String::new(), + "WORKSPACE".into(), + record + .and_then(|run| run.workspace.as_ref()) + .map_or("pending".into(), |path| path.display().to_string()), + ]; + if let Some(image) = &snapshot.image { + let total_files = image + .totals + .map_or_else(|| "?".into(), |t| t.files.to_string()); + let total_bytes = image + .totals + .map_or_else(|| "?".into(), |t| t.bytes.to_string()); + let mut image_lines = vec![ + "IMAGE CONTENT (this run)".into(), + image.image.clone(), + format!(" {:<13} {:>10} {:>16}", "", "Files", "Bytes"), + format!( + " {:<13} {:>10} {:>16}", + "Cached", image.cached_files, image.cached_bytes + ), + format!( + " {:<13} {:>10} {:>16}", + "Transferred", image.downloaded_files, image.downloaded_bytes + ), + format!(" {:<13} {:>10} {:>16}", "Total", total_files, total_bytes), + " Cached: local block reads, including repeated reads".into(), + " Transferred: verified bytes received from the image server".into(), + " Total: regular files and logical bytes in the whole image".into(), + " Files are distinct paths; partial reads count; kernel cache hits excluded" + .into(), + String::new(), + ]; + image_lines.extend(lines.drain(4..)); + lines.extend(image_lines); + } + lines + } Panel::Files => { let mut lines = vec![ format!("{effects} effects {denied} denied {failed} failed"), @@ -312,30 +345,50 @@ fn panel_lines(snapshot: &Snapshot, panel: Panel, started: Instant, width: usize wrap_log_lines(&snapshot.log, width) } } - Panel::Permissions => { - let mut lines = vec!["SESSION PERMISSIONS".into(), String::new()]; - if let Some(record) = snapshot.record.as_ref() { - lines.push(format!( - "Stored at {}/audit-policy.json", - record.storage.display() - )); - lines.push(String::new()); - } - if snapshot.audit_rules.is_empty() { - lines.push("No decisions yet. Ask rules pause the matching access.".into()); - } else { - lines.extend(snapshot.audit_rules.iter().cloned()); - } - lines - } + Panel::Permissions => permission_lines(snapshot, 0, false), Panel::Keys => super::input::help_lines(), } } +fn permission_lines(snapshot: &Snapshot, selected: usize, confirm: bool) -> Vec { + let mut lines = vec![ + if confirm { + "Press x again to forget; broader rules may apply." + } else { + "j/k Select decision x Forget (asks for confirmation)" + } + .into(), + "Saved decisions apply to ask rules; explicit deny wins.".into(), + ]; + if let Some(overlay) = snapshot.record.as_ref().and_then(|r| r.overlay.as_ref()) { + lines.push(format!( + "File rules: {} deny / {} ask / {} warn", + overlay.access_policy.deny().len(), + overlay.access_policy.ask().len(), + overlay.access_policy.warn().len() + )); + lines.push(format!("Rule root: {}", overlay.target.display())); + } + lines.push("DECISIONS (session > workspace > user)".into()); + if snapshot.audit_rules.is_empty() { + lines.push("No saved decisions. Matching sensitive files still ask.".into()); + } else { + lines.extend( + snapshot + .audit_rules + .iter() + .enumerate() + .skip(selected) + .map(|(i, line)| format!("{} {line}", if i == selected { ">" } else { " " })), + ); + } + lines +} + fn floating_panel( buf: &mut Vec, layout: Layout, - state: &UiState, + state: &mut UiState, snapshot: &Snapshot, started: Instant, ) { @@ -373,7 +426,22 @@ fn floating_panel( } else { "1 Overview 2 Files 3 Network 4 Job 5 Log 6 Permissions" }; - let lines = panel_lines(snapshot, state.panel, started, (width - 4) as usize); + let lines = if state.panel == Panel::Permissions { + permission_lines(snapshot, state.permission, state.forget_pending) + } else { + panel_lines(snapshot, state.panel, started, (width - 4) as usize) + }; + let lines = wrap_log_lines(&lines, usize::from(width.saturating_sub(4)).max(1)); + state.page_rows = usize::from(height.saturating_sub(4)).max(1); + state.max_scroll = lines.len().saturating_sub(state.page_rows); + if state.panel == Panel::Permissions { + state.max_scroll = snapshot.audit_rules.len().saturating_sub(1); + state.scroll = 0; + } else { + state.scroll = state.scroll.min(state.max_scroll); + } + let first = state.scroll; + let last = (first + state.page_rows).min(lines.len()); for inner in 0..height - 2 { let row = y + inner + 1; move_to(buf, row, x); @@ -424,77 +492,237 @@ fn floating_panel( ) .as_bytes(), ); + if state.panel != Panel::Permissions && !lines.is_empty() { + move_to(buf, y + height - 1, x + 2); + print_clipped( + buf, + &format!( + " {}–{} / {} ↑↓ Scroll · PgUp/PgDn · Home/End ", + first + 1, + last, + lines.len() + ), + width - 4, + ); + } buf.extend_from_slice(b"\x1b[0m"); } +fn boundary_label(snapshot: &Snapshot) -> String { + let Some(run) = &snapshot.record else { + return "Preparing runtime…".into(); + }; + let files = match &run.overlay { + None => "Files: host paths", + Some(overlay) if overlay.auto_apply => "Files: overlay (auto-apply)", + Some(overlay) if overlay.auto_discard => "Files: overlay (discard on exit)", + Some(_) => "Files: overlay (review to apply)", + }; + let network = match &run.network_interception { + Some(profile) if profile.is_enforcing() => "Net: enforced interception", + Some(_) => "Net: cooperative proxy", + None => "Net: see Review for policy", + }; + format!("{files} | {network}") +} + fn audit_dialog( buf: &mut Vec, layout: Layout, request: &persisting_control::audit::AuditRequest, + prompt: &super::audit_ui::Prompt, ) { + use super::audit_ui::{Lifetime, Scope, choice}; + use persisting_control::audit::AuditKind; + const BODY: &str = "\x1b[0;48;2;28;32;40;38;2;232;235;240m"; + const AMBER: &str = "\x1b[0;48;2;28;32;40;38;2;255;190;80m"; + const SELECTED: &str = "\x1b[1;48;2;255;190;80;38;2;24;28;34m"; let width = layout.cols.saturating_sub(4).clamp(10, 88); - let height = layout.agent_rows.clamp(4, 12); + let height = layout.rows.saturating_sub(2).min(16); let x = (layout.cols - width) / 2 + 1; - let y = 3 + (layout.agent_rows - height) / 2; + let y = (layout.rows - height) / 2 + 1; let title = match request.kind { - persisting_control::audit::AuditKind::File => "FILE ACCESS PAUSED", - persisting_control::audit::AuditKind::Network => "NETWORK ACCESS PAUSED", + AuditKind::File => " FILE ACCESS PAUSED ", + AuditKind::Network => " NETWORK ACCESS PAUSED ", }; - let mut lines = wrap_log_lines( - std::slice::from_ref(&request.target), - usize::from(width.saturating_sub(4)), - ); - lines.push(String::new()); - lines.extend(wrap_log_lines( - std::slice::from_ref(&request.reason), - usize::from(width.saturating_sub(4)), - )); for row in 0..height { move_to(buf, y + row, x); - buf.extend_from_slice(ACTIVE.as_bytes()); + buf.extend_from_slice(AMBER.as_bytes()); let (left, right) = if row == 0 { - ("┌", "┐") + ("╭", "╮") } else if row == height - 1 { - ("└", "┘") + ("╰", "╯") } else { ("│", "│") }; - buf.extend_from_slice(left.as_bytes()); - if row == 0 || row == height - 1 { - buf.extend_from_slice("─".repeat((width - 2) as usize).as_bytes()); + write!( + buf, + "{left}{}{right}", + if row == 0 || row == height - 1 { + "─" + } else { + " " + } + .repeat((width - 2) as usize) + ) + .unwrap(); + } + let mut line = |row: u16, text: &str, style: &str| { + move_to(buf, y + row, x + 2); + buf.extend_from_slice(style.as_bytes()); + print_clipped(buf, text, width - 4); + }; + line(0, title, SELECTED); + if height < 14 || width < 56 { + line(1, "Resize to review access", BODY); + line(height - 2, "[ d Deny ]", SELECTED); + return; + } + line( + 1, + if request.kind == AuditKind::File { + "Allow file access (not just reading)" } else { - buf.extend_from_slice(b"\x1b[48;2;15;19;16m"); - buf.extend_from_slice(" ".repeat((width - 2) as usize).as_bytes()); - } - move_to(buf, y + row, x + width - 1); - buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice(right.as_bytes()); - if row == 0 { - move_to(buf, y, x + 2); - buf.extend_from_slice(b"\x1b[38;2;167;230;54m"); - print_clipped(buf, &format!(" {title} "), width - 4); - } else if row > 0 && row < height - 1 { - move_to(buf, y + row, x + 2); - buf.extend_from_slice(b"\x1b[48;2;15;19;16;38;2;220;224;220m"); - let line = if row == height - 2 { - match request.kind { - persisting_control::audit::AuditKind::File if width < 55 => { - "1 File 2 Dir 3 Ext d Deny" - } - persisting_control::audit::AuditKind::File => { - "1 This file 2 Same directory 3 Same suffix d Deny" - } - persisting_control::audit::AuditKind::Network if width < 55 => { - "1 Target 2 Domain d Deny" - } - persisting_control::audit::AuditKind::Network => { - "1 This target 2 Host and subdomains (same port) d Deny" - } - } + "This connection needs your permission" + }, + BODY, + ); + let target = wrap_log_lines(std::slice::from_ref(&request.target), (width - 4) as usize); + for (index, text) in target.iter().take(2).enumerate() { + line(2 + index as u16, text, AMBER); + } + line( + 4, + if request.reason == "not-in-allowlist" { + "No matching permission has been saved." + } else { + &request.reason + }, + BODY, + ); + line( + 5, + if request.kind == AuditKind::Network { + "Limited to this port and transport." + } else { + "Includes inspect, read, change, and delete in the view." + }, + BODY, + ); + line( + 6, + if prompt.focus == 0 { + "> ALLOW ACCESS TO (Left / Right)" + } else { + " ALLOW ACCESS TO" + }, + AMBER, + ); + line( + 8, + if prompt.focus == 1 { + "> REMEMBER FOR (Left / Right)" + } else { + " REMEMBER FOR" + }, + AMBER, + ); + let broad = prompt.lifetime == Lifetime::User && prompt.scope == Some(Scope::Suffix); + line( + 10, + if broad { + "All workspaces: files with this suffix will be allowed." + } else if prompt.lifetime == Lifetime::User { + "Saved for this user, available across workspaces." + } else if prompt.lifetime == Lifetime::Workspace { + "Saved for future sessions in this workspace." + } else { + "Applies only to this session." + }, + BODY, + ); + line( + height - 2, + "Tab / Up / Down Move Enter Confirm d Deny", + BODY, + ); + let scope_options = match request.kind { + AuditKind::File => [ + (b'1', "This file"), + (b'2', "Same folder"), + (b'3', "Same suffix"), + ] + .to_vec(), + AuditKind::Network => [(b'1', "This target"), (b'2', "Host + subdomains")].to_vec(), + }; + for (row, options, focus) in [ + ( + 7, + scope_options + .into_iter() + .filter_map(|(key, label)| { + choice(request, key).map(|(scope, _)| { + ( + format!("{} {label}", key as char), + prompt.scope.unwrap_or(Scope::Exact) == scope, + ) + }) + }) + .collect::>(), + 0, + ), + ( + 9, + [ + ("s Session", Lifetime::Session), + ("w Workspace", Lifetime::Workspace), + ("u User", Lifetime::User), + ] + .into_iter() + .map(|(label, lifetime)| (label.to_string(), lifetime == prompt.lifetime)) + .collect(), + 1, + ), + ( + height - 3, + vec![ + ("Deny".into(), prompt.scope.is_none()), + ("Allow & remember".into(), prompt.scope.is_some()), + ], + 2, + ), + ] { + move_to(buf, y + row, x + 2); + let mut remaining = width - 4; + for (label, selected) in options { + let label = if width < 72 { + label + .replace("This file", "File") + .replace("Same folder", "Folder") + .replace("Same suffix", "Suffix") + .replace("This target", "Target") + .replace("Host + subdomains", "Subdomains") } else { - lines.get((row - 1) as usize).map_or("", String::as_str) + label }; - print_clipped(buf, line, width - 4); + let text = format!( + "{}[{} {}] ", + if prompt.focus == focus && selected { + ">" + } else { + " " + }, + if selected { "●" } else { "○" }, + label + ); + buf.extend_from_slice(if selected { + SELECTED.as_bytes() + } else { + BODY.as_bytes() + }); + print_clipped(buf, &text, remaining); + remaining = remaining + .saturating_sub(unicode_width::UnicodeWidthStr::width(text.as_str()) as u16); } } buf.extend_from_slice(b"\x1b[0m"); @@ -503,7 +731,7 @@ fn audit_dialog( pub(super) fn render( stdout: &mut impl Write, layout: Layout, - state: &UiState, + state: &mut UiState, screen: &Screen, snapshot: &Snapshot, started: Instant, @@ -518,31 +746,16 @@ pub(super) fn render( .record .as_ref() .and_then(|run| run.workspace.as_ref()) - .and_then(|path| path.file_name()) - .and_then(|name| name.to_str()) - .unwrap_or("workspace"); - let mut session = format!( - " pVisor ({}) ", - workspace.chars().take(16).collect::() - ); - if UnicodeWidthStr::width(session.as_str()) + 11 > usize::from(layout.cols) { - session = " pVisor ".into(); - } - bar_line(&mut buf, 1, layout.cols, &session); - move_to( + .map_or_else( + || "Loading workspace…".into(), + |path| path.display().to_string(), + ); + bar_line( &mut buf, 1, - UnicodeWidthStr::width(session.as_str()) as u16 + 1, + layout.cols, + &format!(" pVisor | {agent} | {workspace}"), ); - buf.extend_from_slice(b"\x1b[48;2;167;230;54;38;2;18;22;17;1m"); - print_clipped( - &mut buf, - " ❯ Job #1 ", - layout - .cols - .saturating_sub(UnicodeWidthStr::width(session.as_str()) as u16), - ); - buf.extend_from_slice(b"\x1b[0m"); move_to(&mut buf, 2, 1); buf.extend_from_slice(ACTIVE.as_bytes()); @@ -571,7 +784,7 @@ pub(super) fn render( move_to(&mut buf, 2, 3); print_clipped( &mut buf, - &format!(" {agent} "), + &format!(" {} ", boundary_label(snapshot)), layout.cols.saturating_sub(6), ); @@ -612,7 +825,7 @@ pub(super) fn render( floating_panel(&mut buf, layout, state, snapshot, started); } if let Some(request) = &snapshot.audit { - audit_dialog(&mut buf, layout, request); + audit_dialog(&mut buf, layout, request, &snapshot.audit_prompt); } status_bar::render(&mut buf, layout.cols, layout.rows, state, snapshot, started); if state.agent_input_active() && snapshot.audit.is_none() && !screen.hide_cursor() { @@ -720,7 +933,7 @@ mod tests { render( &mut output, layout, - &UiState::default(), + &mut UiState::default(), screen.screen(), &snapshot, Instant::now(), @@ -730,8 +943,36 @@ mod tests { assert!(output.contains("NETWORK ACCESS PAUSED")); assert!(output.contains("unexpected.example:443")); assert!(output.contains("1 This target")); - assert!(output.contains("2 Host and subdomains")); + assert!(output.contains("2 Host + subdomains")); + assert!(output.contains("s Session")); + assert!(output.contains("● Deny")); + assert!(output.contains("Enter Confirm")); + assert!(output.contains("w Workspace")); + assert!(output.contains("u User")); assert!(!output.contains("\x1b[?25h")); + let mut terminal = vt100::Parser::new(24, 80, 0); + terminal.process(output.as_bytes()); + assert!(terminal.screen().contents().contains("Enter Confirm")); + for (cols, rows) in [(30, 8), (60, 14), (120, 30)] { + let layout = Layout::new(size(cols, rows), &UiState::default()); + let mut bytes = Vec::new(); + render( + &mut bytes, + layout, + &mut UiState::default(), + screen.screen(), + &snapshot, + Instant::now(), + ) + .unwrap(); + let mut terminal = vt100::Parser::new(rows, cols, 0); + terminal.process(&bytes); + let content = terminal.screen().contents(); + assert!(content.contains("Deny"), "{cols}x{rows}: {content}"); + if cols == 30 { + assert!(content.contains("Resize to review")); + } + } } #[test] @@ -744,9 +985,50 @@ mod tests { assert_eq!(state.input(b'f'), None); assert_eq!(state.panel, Panel::Files); assert_eq!(state.input(0x1b), None); + state.expire_escape(Instant::now() + std::time::Duration::from_millis(200)); assert_eq!(state.input(b'x'), Some(b'x')); } + #[test] + fn review_scroll_reaches_last_line_and_clamps_after_resize() { + let mut state = UiState::default(); + state.input(0x1d); + state.input(b'l'); + let snapshot = Snapshot { + log: (0..100).map(|i| format!("entry-{i:03}")).collect(), + ..Snapshot::default() + }; + let layout = Layout::new(size(80, 24), &state); + let mut bytes = Vec::new(); + floating_panel(&mut bytes, layout, &mut state, &snapshot, Instant::now()); + for byte in b"\x1b[F" { + state.input(*byte); + } + bytes.clear(); + floating_panel(&mut bytes, layout, &mut state, &snapshot, Instant::now()); + let mut terminal = vt100::Parser::new(24, 80, 0); + terminal.process(&bytes); + let content = terminal.screen().contents(); + assert!(content.contains("entry-099"), "{content}"); + assert!(content.contains("/ 100")); + assert!(!content.contains("entry-000")); + let layout = Layout::new(size(120, 60), &state); + floating_panel( + &mut Vec::new(), + layout, + &mut state, + &snapshot, + Instant::now(), + ); + assert_eq!(state.scroll, state.max_scroll); + let short = Snapshot { + log: vec!["only line".into()], + ..Snapshot::default() + }; + floating_panel(&mut Vec::new(), layout, &mut state, &short, Instant::now()); + assert_eq!(state.scroll, 0); + } + #[test] fn long_diagnostics_wrap_within_the_log_panel() { let wrapped = wrap_log_lines(&["startup message with detail".into()], 10); diff --git a/crates/persisting-pvisor/src/config.rs b/crates/persisting-pvisor/src/config.rs index 1b35b452..6540973b 100644 --- a/crates/persisting-pvisor/src/config.rs +++ b/crates/persisting-pvisor/src/config.rs @@ -311,6 +311,7 @@ pub enum FilesystemAccessLevel { Deny, Ask, Read, + Warn, Stage, Write, } diff --git a/crates/persisting-pvisor/src/lib.rs b/crates/persisting-pvisor/src/lib.rs index 3a0920c7..6796cc84 100644 --- a/crates/persisting-pvisor/src/lib.rs +++ b/crates/persisting-pvisor/src/lib.rs @@ -6,6 +6,8 @@ #![cfg_attr(all(target_os = "macos", target_arch = "x86_64"), allow(dead_code))] +#[cfg(unix)] +pub mod cache; pub mod cli; pub mod core; mod runtime; diff --git a/crates/persisting-pvisor/src/oci.rs b/crates/persisting-pvisor/src/oci.rs index df11de3b..97c1fbc9 100644 --- a/crates/persisting-pvisor/src/oci.rs +++ b/crates/persisting-pvisor/src/oci.rs @@ -41,7 +41,7 @@ pub struct PreparedImage { #[derive(Debug, Clone)] pub struct ImageStore { - root: PathBuf, + pub(crate) root: PathBuf, client: Client, } @@ -77,6 +77,10 @@ struct ImageManifest { #[derive(Debug, Default, Deserialize)] struct ImageConfiguration { + #[serde(default)] + architecture: String, + #[serde(default)] + os: String, #[serde(default)] config: RuntimeConfiguration, } @@ -124,6 +128,18 @@ impl ImageStore { } pub fn prepare(&self, image: &str) -> anyhow::Result { + self.prepare_for_architecture(image, host_architecture()?) + } + + pub(crate) fn prepare_for_architecture( + &self, + image: &str, + architecture: &str, + ) -> anyhow::Result { + anyhow::ensure!( + matches!(architecture, "arm64" | "amd64"), + "unsupported image architecture {architecture}" + ); let image_ref = ImageReference::parse(image)?; let mut registry = RegistryClient::new(&self.client, image_ref.clone()); let (mut body, mut manifest_digest) = registry.fetch_manifest(&image_ref.reference)?; @@ -131,7 +147,7 @@ impl ImageStore { .with_context(|| format!("decode OCI manifest for {image}"))?; if value.get("manifests").is_some() { let index: ImageIndex = serde_json::from_value(value)?; - let descriptor = select_platform(&index.manifests)?; + let descriptor = select_platform(&index.manifests, architecture)?; let fetched = registry.fetch_manifest(&descriptor.digest)?; body = fetched.0; manifest_digest = fetched.1; @@ -143,6 +159,12 @@ impl ImageStore { let config: ImageConfiguration = serde_json::from_reader(File::open(config_path)?) .with_context(|| format!("decode image configuration for {image}"))?; + anyhow::ensure!( + config.os == "linux" && config.architecture == architecture, + "image platform is {}/{}, expected linux/{architecture}", + config.os, + config.architecture + ); let digest_hex = digest_hex(&manifest_digest)?; let rootfs = self.root.join("rootfs-v3/sha256").join(digest_hex); let lock_path = self.root.join("locks").join(format!("{digest_hex}.lock")); @@ -388,12 +410,18 @@ fn default_store_dir() -> anyhow::Result { .join("persisting/pvisor/images")) } -fn select_platform(manifests: &[Descriptor]) -> anyhow::Result<&Descriptor> { - let architecture = match std::env::consts::ARCH { +fn host_architecture() -> anyhow::Result<&'static str> { + Ok(match std::env::consts::ARCH { "aarch64" => "arm64", "x86_64" => "amd64", other => bail!("libkrun OCI images are unsupported on host architecture {other}"), - }; + }) +} + +fn select_platform<'a>( + manifests: &'a [Descriptor], + architecture: &str, +) -> anyhow::Result<&'a Descriptor> { manifests .iter() .find(|descriptor| { @@ -436,7 +464,7 @@ fn parse_bearer_challenge(value: &str) -> anyhow::Result anyhow::Result<&str> { +pub(crate) fn digest_hex(digest: &str) -> anyhow::Result<&str> { let value = digest .strip_prefix("sha256:") .context("pVisor v1 only supports sha256 OCI digests")?; @@ -478,7 +506,7 @@ fn verify_file_digest(expected: &str, path: &Path) -> anyhow::Result<()> { Ok(()) } -fn encode_hex(bytes: &[u8]) -> String { +pub(crate) fn encode_hex(bytes: &[u8]) -> String { const HEX: &[u8; 16] = b"0123456789abcdef"; let mut output = String::with_capacity(bytes.len() * 2); for byte in bytes { @@ -784,6 +812,24 @@ mod tests { ); } + #[test] + fn platform_selection_uses_client_architecture() { + let descriptors: Vec = serde_json::from_value(serde_json::json!([ + {"digest": "amd", "platform": {"architecture": "amd64", "os": "linux"}}, + {"digest": "arm", "platform": {"architecture": "arm64", "os": "linux", "variant": "v8"}} + ])) + .unwrap(); + assert_eq!( + select_platform(&descriptors, "amd64").unwrap().digest, + "amd" + ); + assert_eq!( + select_platform(&descriptors, "arm64").unwrap().digest, + "arm" + ); + assert!(select_platform(&descriptors, "riscv64").is_err()); + } + #[test] fn nullable_docker_runtime_fields_decode_as_empty() { let config: ImageConfiguration = diff --git a/crates/persisting-pvisor/src/process.rs b/crates/persisting-pvisor/src/process.rs index a09ffc28..de1dfacf 100644 --- a/crates/persisting-pvisor/src/process.rs +++ b/crates/persisting-pvisor/src/process.rs @@ -26,6 +26,28 @@ use std::process::Stdio; use tokio::io::{AsyncRead, AsyncReadExt}; use tokio::process::{Child, Command}; +#[cfg(any(target_os = "linux", target_os = "macos"))] +fn check_read_only_grants(spec: &RunSpec, cwd: &Path, writable: &[PathBuf]) -> std::io::Result<()> { + for grant in spec + .capabilities + .filesystem + .iter() + .filter(|g| g.access == FilesystemAccess::Read) + { + let path = cwd.join(&grant.path).canonicalize()?; + for writable in writable { + let writable = writable.canonicalize()?; + if path.starts_with(&writable) || writable.starts_with(&path) { + return Err(std::io::Error::other(format!( + "read-only share overlaps a writable runtime path: {}", + path.display() + ))); + } + } + } + Ok(()) +} + #[cfg(target_os = "linux")] struct ResourceCgroup { path: PathBuf, @@ -842,6 +864,8 @@ fn platform_launcher_command( } } + check_read_only_grants(spec, &cwd, &writable_paths)?; + let network = network_isolation(spec)?; let (allowed_unix_sockets, local_socket_roots) = if network.is_loopback_only() { ( @@ -1113,6 +1137,8 @@ fn rootless_plan( } } + check_read_only_grants(spec, &cwd, &read_write)?; + read_only.sort_unstable(); read_only.dedup(); read_write.sort_unstable(); @@ -1824,6 +1850,28 @@ mod tests { } } + #[cfg(any(target_os = "linux", target_os = "macos"))] + #[test] + fn read_only_grants_reject_writable_ancestors_and_symlink_aliases() { + let temp = tempfile::tempdir().unwrap(); + let shared = temp.path().join("shared"); + let separate = temp.path().join("separate"); + std::fs::create_dir(&shared).unwrap(); + std::fs::create_dir(&separate).unwrap(); + let alias = temp.path().join("alias"); + std::os::unix::fs::symlink(&shared, &alias).unwrap(); + let mut spec = RunSpec::process("run", "agent", "/bin/true"); + spec.capabilities + .filesystem + .push(persisting_control::FilesystemCapability { + path: alias.display().to_string(), + access: FilesystemAccess::Read, + }); + assert!(check_read_only_grants(&spec, temp.path(), &[separate]).is_ok()); + assert!(check_read_only_grants(&spec, temp.path(), &[shared]).is_err()); + assert!(check_read_only_grants(&spec, temp.path(), &[temp.path().to_owned()]).is_err()); + } + #[cfg(target_os = "linux")] #[test] fn rootless_plan_writes_through_normally_and_stages_state_for_safe_shells() { diff --git a/crates/persisting-pvisor/src/vm/supported.rs b/crates/persisting-pvisor/src/vm/supported.rs index 9acf0d8c..aa51dcd8 100644 --- a/crates/persisting-pvisor/src/vm/supported.rs +++ b/crates/persisting-pvisor/src/vm/supported.rs @@ -372,6 +372,8 @@ impl RunExecutor for VmExecutor { BTreeMap::new() }; for key in [ + crate::cache::SERVER_ENV, + "PERSISTING_PVISOR_CACHE_TOKEN", crate::AGENTCTL_ENDPOINT_ENV, crate::AGENTCTL_TOKEN_ENV, crate::AGENTCTL_TRANSPORT_ENV, diff --git a/crates/persisting-pvisor/tests/macos_safe.rs b/crates/persisting-pvisor/tests/macos_safe.rs index 008909b2..a08928f1 100644 --- a/crates/persisting-pvisor/tests/macos_safe.rs +++ b/crates/persisting-pvisor/tests/macos_safe.rs @@ -104,6 +104,8 @@ fn safe_profile_stages_reviews_and_applies_on_macos() { let outside = temporary.path().join("outside.txt"); let outside_secret = temporary.path().join("outside-secret.txt"); fs::create_dir(&workspace).unwrap(); + fs::create_dir(workspace.join("existing")).unwrap(); + fs::write(workspace.join("existing/from-cwd.txt"), "cwd-lower").unwrap(); fs::write(&outside_secret, "read-compatible").unwrap(); let mut command = Command::new(env!("CARGO_BIN_EXE_pvisor")); @@ -117,6 +119,7 @@ fn safe_profile_stages_reviews_and_applies_on_macos() { "/bin/sh", "-c", r#" + test "$(cat existing/from-cwd.txt)" = cwd-lower || exit 37 test "$PERSISTING_SANDBOX_FILESYSTEM" = seatbelt-write || exit 38 test "$PERSISTING_SANDBOX_NETWORK" = ambient || exit 39 test "$(cat "$2")" = read-compatible || exit 40 @@ -474,3 +477,112 @@ print('required-sandbox-ok') ); assert!(!workspace.join("result.txt").exists()); } + +#[test] +fn ask_preserves_default_rules_read_only_shares_and_job_changes() { + use std::io::{BufRead, BufReader, Write}; + use std::process::Stdio; + use std::time::{Duration, Instant}; + if !macfuse_is_installed() { + return; + } + let temp = tempfile::Builder::new() + .prefix("pvask") + .tempdir_in("/tmp") + .unwrap(); + let workspace = temp.path().join("workspace"); + let reference = temp.path().join("reference"); + let runs = temp.path().join("runs"); + fs::create_dir_all(workspace.join(".ssh")).unwrap(); + fs::create_dir(&reference).unwrap(); + fs::write(workspace.join(".env"), "secret").unwrap(); + fs::write(workspace.join(".ssh/key"), "private key").unwrap(); + fs::write(reference.join("reference.txt"), "reference").unwrap(); + let socket = temp.path().join("ask.sock"); + let listener = UnixListener::bind(&socket).unwrap(); + listener.set_nonblocking(true).unwrap(); + let log = temp.path().join("log"); + fs::write(&log, "").unwrap(); + let mut child = Command::new(env!("CARGO_BIN_EXE_pvisor")) + .current_dir(&workspace) + .env("PERSISTING_RUN_HOME", &runs) + .env("PVISOR_UI_CHILD", "1") + .env("PVISOR_UI_STAGE_FILE", temp.path().join("stage")) + .env("PVISOR_UI_LOG_FILE", &log) + .env("PVISOR_UI_AUDIT_SOCKET", &socket) + .args(["--ask", "--no-config", "--access", "custom:ask", "--mount"]) + .arg(format!("{}:read", reference.display())) + .args([ + "--", + "/bin/sh", + "-c", + r#" + if cat .env; then exit 11; fi + if cat .ssh/key; then exit 12; fi + test "$(cat "$1/reference.txt")" = reference || exit 13 + if echo changed > "$1/reference.txt"; then exit 14; fi + if rm "$1/reference.txt"; then exit 15; fi + printf retained > result.txt + "#, + "sh", + ]) + .arg(&reference) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + let deadline = Instant::now() + Duration::from_secs(40); + let mut asked = Vec::new(); + while child.try_wait().unwrap().is_none() { + if Instant::now() > deadline { + child.kill().unwrap(); + panic!("ask test timed out"); + } + match listener.accept() { + Ok((mut stream, _)) => { + stream + .set_read_timeout(Some(Duration::from_secs(2))) + .unwrap(); + let mut line = String::new(); + BufReader::new(&stream).read_line(&mut line).unwrap(); + let request: persisting_control::audit::AuditRequest = + serde_json::from_str(&line).unwrap(); + asked.push(request.target); + stream.write_all(b"\"deny\"\n").unwrap(); + } + Err(e) if e.kind() == std::io::ErrorKind::WouldBlock => { + std::thread::sleep(Duration::from_millis(10)) + } + Err(e) => panic!("{e}"), + } + } + let output = child.wait_with_output().unwrap(); + assert!( + output.status.success(), + "{}\n{}", + String::from_utf8_lossy(&output.stderr), + fs::read_to_string(&log).unwrap() + ); + assert!(asked.iter().any(|path| path == ".env")); + assert!(!workspace.join("result.txt").exists()); + assert_eq!( + fs::read_to_string(reference.join("reference.txt")).unwrap(), + "reference" + ); + let job = only_run(&runs); + assert!(RunBundle::read(&job).unwrap().filesystem.is_some()); + let output = Command::new(env!("CARGO_BIN_EXE_pvisor")) + .arg("apply") + .arg(&job) + .output() + .unwrap(); + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!( + fs::read_to_string(workspace.join("result.txt")).unwrap(), + "retained" + ); +} diff --git a/docs/shared-image-cache.md b/docs/shared-image-cache.md new file mode 100644 index 00000000..b6b500dc --- /dev/null +++ b/docs/shared-image-cache.md @@ -0,0 +1,205 @@ +# Shared OCI file cache, protocol v1 + +`pvisor cache serve` exposes the server's OCI image store as a read-only file +service. Clients never receive host filesystem paths. An image is prepared once +on the server, then addressed by its resolved platform manifest SHA-256 digest. +The existing store performs blob verification, layer application and whiteout +handling. File queries do not contact the registry. + +## Usage + +```sh +# Terminal 1: foreground server, default per-user Unix socket and OCI store +pvisor cache serve + +# Terminal 2: use the same default socket +pvisor cache prepare alpine:latest +# Copy the digest from the JSON result: +pvisor cache list sha256:YOUR_MANIFEST_DIGEST +pvisor cache stat sha256:YOUR_MANIFEST_DIGEST etc/os-release +pvisor cache read sha256:YOUR_MANIFEST_DIGEST etc/os-release +``` + +`PERSISTING_PVISOR_CACHE_SERVER` selects the endpoint for both clients and server. +`cache serve --listen` overrides it on the server. With no override the endpoint +is `unix:///persisting/pvisor/cache.sock`: + +- macOS: `~/Library/Caches/persisting/pvisor/cache.sock` +- Linux: `$XDG_CACHE_HOME/persisting/pvisor/cache.sock`, ordinarily + `~/.cache/persisting/pvisor/cache.sock` + +The server accepts `--image-store DIR` or `PERSISTING_PVISOR_IMAGE_STORE` for its +existing OCI store. It does not auto-start. + +## Automatic VM lazy loading + +When `pvisor --rootfs image=IMAGE -- COMMAND` prepares an OCI image for a VM, +it probes the default socket with a two-second `ping` handshake. A live compatible +server selects lazy loading automatically; a missing socket or refused connection +(stale socket) uses the existing local OCI preparation path. Authentication, +protocol and timeout errors are reported, not silently bypassed. + +An explicit `PERSISTING_PVISOR_CACHE_SERVER` requires that service to work. +Set `PERSISTING_PVISOR_CACHE_SERVER=off` to force local preparation. Explicit +rootfs directories and native container execution retain their existing behavior. + +The client mounts an immutable read-only FUSE lower (macFUSE FSKit on macOS; +FUSE on Linux), retaining the existing VM writable upper. Metadata is fetched on demand and retained in memory during the mount. With a +server that advertises `metadata_generation`, verified stat responses (including +missing paths) and directory pages are also persisted under +`/persisting/pvisor/metadata/v1////`. +They survive VM exits; corrupt entries are fetched again. Older servers without +a generation keep the previous in-memory-only behavior. Generation includes the +server root directory identity and change time, so rebuilding the extracted root +invalidates metadata containing old host inode numbers. Prepared roots must stay +immutable; in-place edits below the root are not supported. Content is fetched in 1 MiB +blocks into `/persisting/pvisor/blocks///`, +with per-file/per-block keys. On macOS, `` is `~/Library/Caches`; +on Linux it is `$XDG_CACHE_HOME`, ordinarily `~/.cache`. This block cache is +independent of `--image-store` and `PERSISTING_PVISOR_IMAGE_STORE`. Blocks are checksum-verified, published atomically and +shared across local processes using file locks. Cache hits are verified before +use; corrupted blocks are fetched again. No sparse placeholder files are exposed. +Normal kernel readahead may fetch adjacent bytes, and copy-up may read a whole +individual file. The client does not extract the full image. + +The FUSE mount lives until the VM run finishes and is then unmounted; cached +blocks remain. Cached data can still be read after service failure, but missing +blocks fail with I/O errors. The digest and endpoint stay fixed for the run; +there is no mid-run registry fallback. The cache endpoint/token are removed from +implicitly inherited guest environment variables. + +The server still fully prepares an uncached image before answering `prepare`. +This is client-side lazy loading, not lazy OCI layer extraction on the server. +The FUSE adapter and existing virtio-fs worker currently process requests +synchronously: a cache miss can delay unrelated filesystem requests. No explicit +vCPU pause is used. Cache quotas/eviction, original OCI xattrs and asynchronous +virtio-fs completions are not added by this implementation. + +The public Rust client is `persisting_pvisor::cache::CacheClient::from_env()`. +It is blocking. `cache prepare/list/stat/read` continue to be explicit service +commands and do not use the VM's local fallback policy. + +Unix sockets are mode 0600 and require the same effective user at both ends. +A lock prevents two servers from owning the socket; a stale socket is reclaimed +on restart, but a regular file, symlink or live listener is never removed. +Use a socket in a directory owned by the serving user. Stopping the foreground +server with Ctrl-C can leave a stale socket; no manual cleanup is needed. + +For a remote server, use authenticated loopback TCP inside an SSH tunnel: + +```sh +# On the server: set a strong shared secret through your secret manager/shell. +export PERSISTING_PVISOR_CACHE_TOKEN='YOUR_RANDOM_SECRET' +pvisor cache serve --listen tcp://127.0.0.1:7447 + +# On the client machine, keep this tunnel running: +ssh -N -L 7447:127.0.0.1:7447 your-server + +# Client shell, using the same secret: +export PERSISTING_PVISOR_CACHE_TOKEN='YOUR_RANDOM_SECRET' +export PERSISTING_PVISOR_CACHE_SERVER=tcp://127.0.0.1:7447 +pvisor cache prepare alpine:latest +``` + +TCP requires a nonempty token and accepts only literal loopback IP endpoints. +There is no built-in TLS; use SSH for transport encryption. A token grants all +cache operations, including preparation of new images, so this is a trusted +shared service, not a public multi-tenant API. If a token is configured for a +Unix server, Unix clients must supply it too. + +## TUI download statistics + +With `--tui`, lazy image runs show three groups, each with file count and size: +`Cached` (local cache reads), `Transferred` (content received from the image +server, whether local or remote), and `Total` (the whole image). Narrow status +bars abbreviate them to `C / X / T`; Overview shows all three rows with exact +byte counts. The Log panel records every +verified block transfer with its file path, byte count and cumulative run totals. +The status bar and Overview also show local cache reads: distinct file paths +and cumulative verified block bytes, including repeated reads. These are separate +from downloads; they exclude reads satisfied by the host or guest kernel page +cache. Each file's first local cache hit appears in Log as `no download`. +Directory listings only fetch metadata and do not count as content reads. Image startup and I/O diagnostics +also go to this panel instead of the guest terminal; without TUI they use stderr. Downloads count this run's +verified content received from the cache server, excluding local cache hits, +protocol metadata and guest network traffic. A partially downloaded file counts +once; repeated transfers add bytes again. A warm run can therefore show zero +downloads even while using the image. + +Totals count regular file paths and their uncompressed logical sizes in the +server's extracted image, including empty files and each hard-link name. +Directories and symlinks are excluded. These totals are cached per manifest +digest after a metadata-only scan. They describe the whole image, not compressed +OCI layer sizes. Older servers without totals display `?`. + +## Wire format + +Each connection carries exactly one request and one response, then closes. +A frame is a four-byte unsigned big-endian JSON byte length followed by UTF-8 +JSON. JSON frames are limited to 1 MiB. A successful `read` response frame is +followed immediately by exactly `length` raw bytes. Other responses have no +binary body. There are no unsolicited messages or compression. + +Request envelope: + +```json +{"version":1,"token":null,"request":{"op":"read","digest":"sha256:...","path":[101,116,99,47,111,115,45,114,101,108,101,97,115,101],"offset":0,"length":1048576}} +``` + +Paths and directory names are JSON arrays of Unix filename bytes, preserving +non-UTF-8 filenames. Paths are relative to the image root; an empty path means +the root directory. Absolute paths, parent traversal and NUL are rejected. +Symlinks are returned as metadata, never followed by server path resolution. +A future filesystem client must resolve guest symlinks within the guest tree. + +| `op` | Fields | Response `status` | +| --- | --- | --- | +| `ping` | none | `ready` (protocol v1) | +| `prepare` | `image`, `architecture` (`amd64` or `arm64`) | `prepared`: `digest`, `architecture`, `env`, `entrypoint`, `cmd`, optional `totals` (`files`, `bytes`), optional `metadata_generation` | +| `list` | `digest`, `path`, `offset` (entry index, start at 0) | `entries`: sorted `names` (up to 256), `next_offset` (null when done) | +| `stat` | `digest`, `path` | `metadata`: `kind`, `size`, `mode`, `uid`, `gid`, `inode`, `nlink`, `mtime`, `mtime_nsec`, `target` | +| `read` | `digest`, `path`, `offset` (byte offset), `length` (1..1048576) | `data`: `length`, `sha256`, followed by raw bytes | + +`prepare` requests Linux images for the client's architecture, independent of +the server architecture. It resolves mutable tags on each call; the returned +platform manifest digest pins subsequent requests. Preparation can populate an +uncached image and uses existing per-image extraction locks. It may still need +the registry even when an extracted root already exists. `read`, `stat` and +`list` require an already prepared digest; they never implicitly pull an image. + +`kind` is `file`, `directory`, `symlink` or `special`. `mode` includes Unix type +and permission bits. `target` contains symlink bytes or null. Attributes reflect +the extracted server filesystem; v1 does not reconstruct original tar ownership, +provide xattrs, or define portable inode IDs across servers. Only regular files +can be read. A short read, including zero bytes, indicates EOF. Clients must +check the body length and SHA-256 before admitting bytes to a cache. The supplied +hash detects transfer corruption; it is not an independent proof against an +untrusted server. The server and its local image store are trusted. + +Errors are frames such as: + +```json +{"status":"error","code":"not_found","message":"..."} +``` + +Codes are `not_found`, `permission_denied`, and `request_failed` (including +invalid arguments, unsupported protocol versions and failed authentication). +Malformed framing may also cause disconnection. Clients must treat an early +close, truncated body or bad checksum as failure, never as a missing file or +zero-filled content. Error messages are explanatory, not machine-stable. + +The server shares in-memory caches for up to 4096 stat responses and 128 sorted +directory indexes across requests. Directory pagination reuses the same index +instead of rescanning and sorting on every page. Each cache clears at its entry +ceiling and is rebuilt lazily after a server restart. These caches cover metadata, +not mutable tag resolution; `prepare` still contacts the registry. + +The server has 16 workers and at most 16 queued connections. Excess connections +are closed; clients may retry. Socket reads/writes have a 300-second inactivity +timeout, and TCP connects have a 10-second timeout. Long-running preparation may +outlive a disconnected client; retrying is safe. Shutdown does not cancel +individual OCI downloads gracefully. Registry download limits and cache eviction +remain those of the existing image store; v1 adds neither quotas nor eviction. + +No vCPU pause/resume messages exist. Downloads happen in the host-side FUSE +server, outside the sandboxed VM runner process. diff --git a/docs/src/en/reference/cli.md b/docs/src/en/reference/cli.md index 0f7796b6..17e13a01 100644 --- a/docs/src/en/reference/cli.md +++ b/docs/src/en/reference/cli.md @@ -40,16 +40,46 @@ pvisor run --tui -- bash ``` The shell or Agent keeps the terminal width and receives keyboard input by -default. The bottom bar shows Job state, elapsed time, filesystem and network -counts, log count, and the `Ctrl-]` command hint. Pressing the prefix replaces +default. The header shows the workspace and file/network boundaries. The bottom +bar shows state, elapsed time, pending authorization and nonzero activity or +errors; detailed counters remain in the panels. `Ctrl-]` opens the menu. Pressing the prefix replaces the same bar with the available shortcuts. pVisor's own startup diagnostics appear in the Log panel instead of the Agent terminal. Press `Ctrl-]` to enter command mode, then `r`, `f`, `n`, -`u`, or `l` to open the Overview, Files, Network, Job, or Log panel. Press `?` -for key help. In a panel, use Tab or `1`–`5` to switch views, `j`/`k` to scroll, +`u`, `l`, or `p` to open the Overview, Files, Network, Job, Log, or Permissions panel. Press `?` +for key help. In a panel, use Tab or `1`–`6` to switch views, `j`/`k` to scroll, and Esc or `Ctrl-]` to return to the Agent. Press `Ctrl-]` twice to send a literal Ctrl-] to the Agent. +Use `pvisor --ask -- bash` to ask before accessing files covered by `ask` +rules or unlisted proxy destinations; `--ask` also enables `--tui` and `--safe`. +File approval covers inspection, reading, modification, and deletion in the +allowed view, not read-only access. Saved decisions are consulted only for +matching ask rules, and cannot override explicit deny or sandbox boundaries. +Ask prompts build reusable permissions as you work. `--access 'private/*:ask'` +automatically opens the audit TUI. In a prompt, choose where to remember the +answer with `s` (session, the default), `w` (workspace), or `u` (user), then +press `1` to select the exact target and Enter to confirm, or `d` to deny it. +Use Tab or Up/Down to move between scope, lifetime and action buttons; +Left/Right changes the selection. Enter on an action button confirms it. +Deny is selected by default; `d` denies immediately. Selection alone never grants access. For files, `2` allows +files in the same directory and `3` allows the same suffix. For network +requests, `2` allows the hostname and its subdomains on the same port and +transport. + +Session rules live in the Job's `audit-policy.json`. Workspace and user rules +live under `permissions` in `~/.config/pvisor/config.toml` (or +`$XDG_CONFIG_HOME/pvisor/config.toml` when set to an absolute directory), and +load when a new TUI Job starts. Workspace rules use the canonical workspace +path. Precedence is session > workspace > user, with the last matching rule +winning within each scope. Explicit deny policies still take precedence over +ask approvals. Persistent file rules use original absolute paths; user-wide +suffix grants can match files across workspaces. Review rules in Permissions; +use `j`/`k` to select a saved decision and `x`, then `x` again, to forget it. +Removing a decision restores matching broader rules or a new prompt; it does +not close existing file handles. Other running TUIs reload on their next launch. +Other configuration settings and comments are preserved when saving. + The reference that follows is organized by lifecycle. Options that affect the same Job are intentionally described together so that a copied command has a clear verification step. @@ -76,9 +106,11 @@ pvisor status --review last Host execution preserves the host filesystem view by default. `--filesystem sandbox` opts into pVisor's synthetic-root/Landlock or Seatbelt filesystem access policy. -`--stage ` independently opts into an OverlayFS stage for the current workspace, creates an independent Run -and writable stage at the supplied path, -retains changes for manual review, and writes `run-bundle.json` with mode `0600`. +`--safe` stages workspace writes and keeps writable home +state in a private copy-on-write view, including Codex launched from a shell. +`--safe` and `--ask` retain workspace changes and `run-bundle.json` in Job +storage by default. Use `status --review`, `apply`, or `drop` after exit. +`--stage ` chooses another storage directory; it is not required to retain changes. `--strict` fails closed before command start unless every requested capability dimension has non-bypassable enforcement evidence. Today host, container, and @@ -265,18 +297,25 @@ Bulk-read and tool-call attribution monitoring are not provided. Additional shar ## File access rules ```bash -pvisor run --mount /opt/tool:read --access '**/.ssh:deny' --access '**/.env*:read' -- my-agent +pvisor run --safe --mount /opt/tool:read --access '**/.ssh:deny' --access '**/.env*:warn' -- my-agent ``` -`--mount SOURCE[:TARGET]:read|stage|write` declares a host path. When TARGET is omitted, -it equals SOURCE. `write` adds a direct persistent host write grant; `read` and -`stage` currently both become lower layers of the copy-on-write view. A `read` -mount is therefore **not** an enforced read-only boundary for that path. The -current overlay accepts only one nonidentity TARGET for a Run; `write` requires -TARGET to equal SOURCE. -`--access PATH-GLOB:deny|read` defines overlay rules. `deny` hides a matching -path and blocks access. `read` currently maps to a warning on access; it does -not block writes. Rules cannot raise access to `stage` or `write`. +`--mount SOURCE:read` grants a host executor read-only access to that absolute +path, and requires `--safe` or `--ask`. `SOURCE:write` grants direct persistent +host writes. Both keep the original path, cannot be remapped, and must not +overlap the workspace, Job storage, or writable runtime paths. Read-only +shares under the private `/tmp` are unsupported on Linux. These explicit +shares are outside overlay-relative ask rules. +`--mount SOURCE[:TARGET]:stage` adds a copy-on-write lower layer to the workspace +view; it is not an independent directory mount. A Run accepts only one +nonidentity TARGET for this view. Use stage layers for VM composition; +read/write host grants are not supported by the VM or container executor. +`--access PATH-GLOB:deny|ask|warn` adds overlay rules: deny blocks access, ask +pauses matching access for approval, and warn allows access with a diagnostic. +The former `:read` warning spelling is rejected; use `:warn` or a read-only share. +Rules accumulate across config, presets, and CLI. `--clear-access` explicitly +removes config and preset file rules before applying CLI rules. It removes the +default sensitive-file protection too. Precedence is deny > ask > warn. Globs are relative to the mount root: `*` stays within one component, `**` crosses directories, and a matching directory covers all descendants. Matching @@ -294,10 +333,10 @@ cannot identify every private key; add rules for custom names. [filesystem] stage = "../stage-001" backend = "directory" -mount = [{ source = "/opt/tool", access = "read" }] +mount = [{ source = "/opt/tool", access = "stage" }] access = [ { path = "**/.ssh", level = "deny" }, - { path = "**/.env", level = "read" }, + { path = "**/.env", level = "warn" }, ] ``` @@ -422,7 +461,7 @@ overrides. pVisor does not discover a hidden project file. pvisor run \ --name my-agent \ --stage ../stage-001 \ - --mount /opt/tool:read \ + --mount /opt/tool:stage \ --access '**/.ssh:deny' \ --overlayfs-backend directory \ --overlaynet-allow api.openai.com:443 \ @@ -458,7 +497,7 @@ command = ["my-agent"] [filesystem] stage = "../stage-001" -mount = [{ source = "/opt/tool", access = "read" }] +mount = [{ source = "/opt/tool", access = "stage" }] access = [{ path = "**/.ssh", level = "deny" }] backend = "directory" @@ -543,10 +582,9 @@ On Linux, `--rootfs host` selects the host `/` as the VM rootfs lower and selects the VM executor when `--executor` is omitted. `--rootfs ` selects a prepared directory and `--rootfs image=` selects an OCI image or image path. These forms are mutually exclusive, and host rootfs is rejected on macOS. -Use `--mount SOURCE[:TARGET]:read|stage|write` for additional host paths in the -guest; the current workspace is the implicit bottom layer. Omitting TARGET -preserves the source path. Workspace changes use the configured stage, or a -temporary stage that is discarded at Run exit. Writes elsewhere in the VM root +Use `--mount SOURCE[:TARGET]:stage` for additional lower layers in the guest +workspace view; the current workspace is the implicit bottom layer. Workspace +changes are retained in the configured stage or default Job storage. Writes elsewhere in the VM root use a temporary upper and are discarded when the VM exits. The merged rootfs is guest `/`, and `/workspace` becomes the guest cwd. On both @@ -639,3 +677,19 @@ retains compact Run/Overlay metadata, the apply ledger, and capture artifacts. - [Execution environments](../guides/execution.md) for choosing a provider. - [Review and apply changes](../guides/review-apply.md) for filtered, repeatable apply. - [Network control](../guides/network.md) and [capture](../guides/capture.md) for other Effect dimensions. + +### Shared image file cache + +`pvisor cache serve` runs the OCI file service in the foreground. Use +`cache prepare IMAGE`, `cache list DIGEST [PATH]`, `cache stat DIGEST PATH`, and +`cache read DIGEST PATH` to query it. `PERSISTING_PVISOR_CACHE_SERVER` selects the +endpoint; the default is `persisting/pvisor/cache.sock` under the user's cache +directory. The server accepts `--image-store DIR` for an existing OCI store. +Reads support byte ranges and SHA-256 transfer verification. + +VM image runs automatically probe the default socket and use a read-only FUSE +lower with persistent 1 MiB block caching when a compatible server is available. +A missing/stale default socket retains local OCI preparation. An explicitly +configured server must work; `PERSISTING_PVISOR_CACHE_SERVER=off` forces local +preparation. Explicit rootfs directories and native containers are unchanged. +See the [protocol and remote-access guide](../../../shared-image-cache.md). diff --git a/docs/src/zh/reference/cases.md b/docs/src/zh/reference/cases.md index 7026bb89..76e298ce 100644 --- a/docs/src/zh/reference/cases.md +++ b/docs/src/zh/reference/cases.md @@ -375,7 +375,7 @@ Linux host stage 示例需要可用的 user/mount namespace。VM 示例需要可 ### C. Stage 与 whole-rootfs -当你希望 Agent 可以自由修改文件、但不污染当前 workspace 时使用这一组。C01 是最常用的持久模式;C02 使用 `--safe` 自动创建并清理临时 stage,C03 演示对持久 stage 显式执行 `drop`。 +当你希望 Agent 可以自由修改文件、但不污染当前 workspace 时使用这一组。C01 是最常用的持久模式;C02 使用 `--safe` 自动选择并保留 stage,C03 演示对持久 stage 显式执行 `drop`。 - [ ] **C01:持久 stage** @@ -406,15 +406,15 @@ Linux host stage 示例需要可用的 user/mount namespace。VM 示例需要可 -- [ ] **C02:自动临时 stage** +- [ ] **C02:默认保留 stage** 建议场景:适合隔离文件变更、保留 stage 或验证 whole-rootfs 的任务。 - 用途:运行一次不需要保留改动的任务。`--safe` 在没有指定 `--stage` 时自动创建系统临时 stage,退出后删除该目录。 + 用途:无需手写存储路径。`--safe` 在没有指定 `--stage` 时使用持久 Job 存储,退出后保留改动。 准备:Linux user/mount namespace 或 macOS Seatbelt 可用。 - 预期:命令成功,日志中给出的临时存储目录已删除,原 workspace 也没有新建的文件。 + 预期:命令成功,日志中给出的存储目录及 Run Bundle 保留,原 workspace 没有新建的文件。 ```bash pvisor --safe -- /bin/sh -c 'printf changed > result.txt' @@ -428,7 +428,7 @@ Linux host stage 示例需要可用的 user/mount namespace。VM 示例需要可 ```bash storage=$(dirname "$(grep -m1 '^Run Bundle: ' "$PVISOR_CASE_STDOUT" | cut -d' ' -f3-)") test -n "$storage" - test ! -e "$storage" + test -f "$storage/run-bundle.json" test ! -e result.txt ``` @@ -551,14 +551,20 @@ Linux host stage 示例需要可用的 user/mount namespace。VM 示例需要可 ### D. OverlayFS 与 Host 安全边界 D01 讲视图层组合,D02/D03 讲 host executor,D04–D06 讲拒绝、显式写入和交互授权。 -文件权限级别为 `deny`、`ask`、`read`、`stage`、`write`:`deny` 直接拒绝, -`ask` 暂停命中的文件操作并询问用户,`read` 目前只记录访问警告; -`stage` 和 `write` 用于 `--mount`,其中 `write` 直接写入宿主 lower。 -`--access PATH-GLOB:ask` 会自动启用审计 TUI 和 safe 暂存视图,无需另加 `--audit`。 +文件规则为 `deny`、`ask`、`warn`:分别表示拒绝、询问、放行并警告;默认累加。 +`--mount` 的 `read` 授予只读宿主共享(要求 host executor 加 `--safe`/`--ask`), +`stage` 组合写时复制底层,`write` 直接写入宿主 lower。 +`--access PATH-GLOB:ask` 会自动启用审计 TUI 和 safe 暂存视图,无需另加 `--ask`。 文件弹窗的 `1` 仅允许此文件,`2` 允许同级目录中的文件,`3` 允许相同后缀的文件; `d` 拒绝此目标。明确的 `deny` 规则仍直接拒绝,不弹窗。 -选择写入当前 Job 的 `audit-policy.json`,后续命中同一范围时自动应用; -`audit.jsonl` 记录人工及自动决策。显式指定 `--stage PATH` 才能在 Job 结束后保留这些记录。 +弹窗默认仅对当前 session 生效;先按 `s`、`w`、`u`,分别选择 session、workspace、user, +再按数字选择授权范围,按 Enter 确认;默认选中拒绝按钮,`d` 直接拒绝。session 规则写入当前 Job 的 `audit-policy.json`; +workspace 和 user 规则写入 `~/.config/pvisor/config.toml` 的 `permissions` 部分 +(设置了绝对路径 `XDG_CONFIG_HOME` 时使用该目录)。workspace 按规范化后的工作目录区分。 +后续 TUI Job 加载这些规则,优先级为 session > workspace > user,同层最后匹配的规则生效。 +持久化文件路径使用原始绝对路径;用户级后缀规则可覆盖其他工作区,授权时应注意范围。 +可在 Permissions 面板查看规则,用 j/k 选择后按两次 x 移除决定;随后可能命中其他规则或重新询问。 +`audit.jsonl` 记录人工及自动决策和保存范围。Job 记录默认保留;`--stage PATH` 可指定位置。 - [ ] **D01:高级 OverlayFS 组合** @@ -719,9 +725,9 @@ D01 讲视图层组合,D02/D03 讲 host executor,D04–D06 讲拒绝、显 - [ ] **D06:`ask` 弹窗与当前 Job 的目录授权** - 用途:用 `--access 'private/*.txt:ask'` 启动审计 TUI;第一次读取时按 `2` 授权同级目录,再读取另一文件,验证规则自动复用。 + 用途:用 `--access 'private/*.txt:ask'` 启动审计 TUI;第一次读取时按 `2`、Enter 授权同级目录,再读取另一文件,验证规则自动复用。 - 准备:Linux user/mount namespace 和 Python 3 可用。示例用伪终端自动输入 `2`;手工运行时在弹窗中按该键。 + 准备:Linux user/mount namespace 和 Python 3 可用。示例用伪终端自动输入 `2`、Enter;手工运行时在弹窗中选择后按 Enter 确认。 预期:只出现一次文件授权弹窗,两个文件均可读取;`audit-policy.json` 保存目录规则,`audit.jsonl` 记录第二次自动允许。`--stage` 保留当前 Job 的审计记录,不会把选择变成全局配置。 @@ -756,7 +762,7 @@ D01 讲视图层组合,D02/D03 讲 host executor,D04–D06 讲拒绝、显 except OSError: pass if not prompted and b'FILE ACCESS PAUSED' in screen: - os.write(master, b'2') + os.write(master, b'2\r') prompted = True ended, result = os.waitpid(pid, os.WNOHANG) if ended: @@ -1124,9 +1130,9 @@ D01 讲视图层组合,D02/D03 讲 host executor,D04–D06 讲拒绝、显 ### G. OverlayNet 这一组只讨论网络边界。proxy 适合需要 host Gateway 的协作式访问,VM auto 和 host deny-all 才适合需要更强网络边界的场景。 -使用 `--audit` 时,未列入规则的代理网络目标会暂停并弹窗:`1` 仅允许当前目标, +使用 `--ask` 时,未列入规则的代理网络目标会暂停并弹窗:`1` 仅允许当前目标, `2` 允许当前主机名及其子域名,范围仍限于相同端口和传输协议;IP 地址没有域名选项, -`d` 拒绝当前目标。选择同样只保存在当前 Job 的 `audit-policy.json` 中。 +`d` 拒绝当前目标。与文件授权一样,先按 `s` / `w` / `u` 选择 session / workspace / user 保存范围。 显式拒绝规则不进入弹窗;未经代理的直接 socket 连接也不会触发此审计。 - [ ] **G01:启用默认 proxy** diff --git a/docs/src/zh/reference/cli.md b/docs/src/zh/reference/cli.md index 83ae39dc..aa10f133 100644 --- a/docs/src/zh/reference/cli.md +++ b/docs/src/zh/reference/cli.md @@ -30,8 +30,9 @@ pvisor apply last --path src pvisor run --tui -- bash ``` -默认键盘输入交给 shell 或 Agent,终端始终保持完整宽度。底栏显示 Job 状态、运行时间、 -文件与网络计数、日志数量和 `Ctrl-]` 引导提示;按下引导键后,同一行切换为完整快捷键。 +默认键盘输入交给 shell 或 Agent,终端始终保持完整宽度。顶部显示工作目录及文件、网络边界; +底栏显示运行状态、时间和非零的访问异常,授权等待会突出提示。详细计数保留在面板中。 +按 `Ctrl-]` 打开菜单,同一行切换为快捷键。 pVisor 自身的启动信息显示在 Log 面板,不混入 Agent 终端。按 `Ctrl-]` 进入命令模式,再按 `r`、`f`、`n`、 `u`、`l`、`p` 打开概览、文件、网络、Job、Log 或 Permissions 面板,按 `?` 查看按键帮助。在面板中用 @@ -42,31 +43,41 @@ Tab 或 `1`–`6` 切换视图,用 `j`/`k` 滚动,按 Esc 或 `Ctrl-]` 返 ### 文件系统参数 -普通 host Job 默认将工作区写入直接透传到 lower。`--safe` 会为工作区创建临时 -changeset,Job 结束后自动丢弃;指定 `--stage PATH` 则保留 changeset, -由 `status --review`、`apply` 或 `drop` 手动处理。 +macOS 的 macFUSE 临时工作区默认以启动 pVisor 时的当前目录作为 lower; +`/Volumes/pvisor-*` 是合并视图的挂载点,包含当前目录已有的文件。 +显式配置工作目录或 OverlayFS base 时,以显式配置为准。 + + +普通 host Job 默认将工作区写入直接透传到 lower。`--safe` 和 `--ask` 默认将工作区 +改动保留在 Job 存储中,退出后用 `status --review`、`apply` 或 `drop` 手动处理。 +`--stage PATH` 仅用于指定存储位置,不再是保留改动的前提。 ```bash pvisor run --stage ./run-stage -- codex -pvisor run --mount /opt/zcode:read --mount /var/lib/zcode:write -- zcode +pvisor run --safe --mount /opt/zcode:read --mount /var/lib/zcode:write -- zcode pvisor run --access '/workspace/**/.ssh:deny' -- zcode pvisor run --access '.env:ask' -- codex ``` -`--mount SOURCE[:TARGET]:ACCESS` 支持 `read`、`stage` 和 `write`;省略 target 时使用 -source。`write` 直接修改宿主机;`read` 与 `stage` 当前都成为写时复制视图的底层, -因此 `read` 目前不是强制只读边界。`--access PATH-GLOB:LEVEL` 支持 `deny`、`ask` -和 `read`;`deny` 阻止访问,`ask` 暂停命中的文件操作并询问用户,`read` 当前仅记录 -访问警告,不阻止写入。这里的访问级别是文件审计规则;`stage` 和 `write` 用于 -`--mount`,不作为 `--access` 的级别。 - -指定 `--access ...:ask` 会自动启用审计 TUI 和 safe 暂存视图,无需另加 `--audit` -或 `--tui`。弹窗可按 `1` 仅允许此文件、`2` 允许同级目录、`3` 允许相同后缀; -`d` 拒绝此次目标。对于未列入规则的代理网络目标,`--audit` 的弹窗可按 `1` 仅允许 +`--mount SOURCE:read` 向 host executor 授予原绝对路径的只读访问,要求 `--safe` 或 `--ask`; +`SOURCE:write` 直接修改宿主机。两者不支持改写 TARGET,不能与工作区、Job 存储或可写运行时路径重叠; +Linux 的私有 `/tmp` 内不能使用只读共享。显式共享不经过工作区 OverlayFS 的 ask 规则。 +`--mount SOURCE[:TARGET]:stage` 则把 SOURCE 加入工作区写时复制视图的底层,并非独立目录挂载。 +`--access PATH-GLOB:deny|ask|warn` 默认累加配置、预设和 CLI 规则:deny 拒绝、ask 询问、warn 放行并警告。 +清空配置与默认文件保护必须显式使用 `--clear-access`;之后再加入 CLI 规则。deny > ask > warn。 +原先表示警告的 `:read` 已拒绝,改用 `:warn`;真正只读请用 `--mount PATH:read`。 + +使用 `pvisor --ask -- bash` 可在命中 `ask` 文件规则或访问未列入规则的代理网络目标时询问权限; +`--ask` 同时启用 `--tui` 和 `--safe`。 +指定 `--access ...:ask` 会自动启用审计 TUI 和 safe 暂存视图,无需另加 `--ask` +或 `--tui`。弹窗用 Tab 或上下方向键切换范围、保存期限和按钮,左右方向键选择,Enter 在按钮上确认;默认选中拒绝按钮。可按 `1` 选择仅此文件、`2` 允许同级目录、`3` 允许相同后缀; +`d` 拒绝此次目标。对于未列入规则的代理网络目标,`--ask` 的弹窗可按 `1` 仅允许 此目标,或按 `2` 允许当前域名及其子域名;两种选择都限定在当前端口和传输协议, IP 地址不能使用域名范围。明确的 `deny` 规则仍然直接拒绝,不进入询问弹窗。 选择会写入当前 Job 目录的 `audit-policy.json`,之后命中相同范围时自动应用; -每次决策记录在 `audit.jsonl`。使用 `--stage PATH` 可在 Job 结束后保留这些记录。 +每次决策记录在 `audit.jsonl`。已保存决定只用于命中 ask 的访问,不能覆盖静态 deny 或外层沙箱。 +在 Permissions 面板用 `j`/`k` 选择决定,按两次 `x` 移除;之后回到更宽范围规则或重新询问。 +移除不会关闭已经打开的文件句柄;其他已运行的 TUI 在下次启动时加载更新。这些记录默认在 Job 结束后保留,`--stage PATH` 可指定位置。 代理网络审计属于协作式边界:未经过代理的直接连接不会触发此弹窗。 ```text @@ -90,9 +101,11 @@ pvisor status --review last ``` 默认 host 执行保留宿主机文件系统视图;`--filesystem sandbox` 才启用 pVisor 的 -synthetic-root/Landlock 或 Seatbelt 文件系统访问策略。`--stage ` 独立启用当前目录的 -OverlayFS stage,在显式 `--stage` 路径创建独立 -Run 和可写 stage,保留改动供人工审查,并以 `0600` 写入 `run-bundle.json`。 +synthetic-root/Landlock 或 Seatbelt 文件系统访问策略。 +`--safe` 默认暂存工作区,并给 HOME(包括在 shell 内启动的 Codex)提供独立的写时复制视图。 +没有 `--stage` 时,changeset 和 Run Bundle 默认保留在 Job 存储中,退出后可 review/apply/drop。 +显式 `--stage ` 会保留 Job 和可写 stage, +改动可供人工审查,并以 `0600` 写入 `run-bundle.json`。 `--strict` 要求每个被请求的 capability 维度都有不可绕过的 enforcement 证据, 否则在 Agent 启动前失败关闭。当前 host / container / VM 都会请求 Network 与 @@ -257,15 +270,13 @@ cooperative 的,直接 socket 仍可能绕过代理。需要不可绕过边界 ## 文件访问规则 ```bash -pvisor run --mount /opt/tool:read --access '**/.ssh:deny' --access '**/.env*:read' -- my-agent +pvisor run --safe --mount /opt/tool:read --access '**/.ssh:deny' --access '**/.env*:warn' -- my-agent ``` -`--mount SOURCE[:TARGET]:read|stage|write` 声明宿主路径;省略 TARGET 时等于 SOURCE。 -`write` 授予持久宿主写权限;`read` 和 `stage` 当前都进入写时复制视图的底层, -所以 `read` 挂载并非该路径的强制只读边界。当前一次 Run 仅接受一个与 SOURCE 不同的 -TARGET;`write` 要求 TARGET 等于 SOURCE。 -`--access PATH-GLOB:deny|read` 在视图内加规则:`deny` 隐藏并阻止匹配路径, -`read` 当前对应访问警告,并不阻止写入。访问规则不能提升到 `stage` 或 `write`。 +`--mount SOURCE:read|write` 是 host executor 的显式共享,分别授予只读或持久写入权限; +只读共享要求 `--safe` 或 `--ask`。`--mount SOURCE[:TARGET]:stage` 是工作区视图的底层组合。 +`--access PATH-GLOB:deny|ask|warn` 追加文件规则,不会替换默认保护;`--clear-access` 才会显式清空。 +`warn` 仅告警,不是只读;文件询问授权包含视图内的检查、读取、修改和删除。 规则相对于挂载根目录匹配:`*` 不跨目录,`**` 可跨目录,匹配目录时覆盖全部后代。 为防止大小写不敏感文件系统上的别名绕过,匹配不区分大小写;绝对路径、空规则及 `.`/`..` @@ -285,10 +296,10 @@ TOML 中对应: [filesystem] stage = "../stage-001" backend = "directory" -mount = [{ source = "/opt/tool", access = "read" }] +mount = [{ source = "/opt/tool", access = "stage" }] access = [ { path = "**/.ssh", level = "deny" }, - { path = "**/.env", level = "read" }, + { path = "**/.env", level = "warn" }, ] ``` @@ -397,7 +408,7 @@ pVisor Gateway、模型流量 capture store 或 Claude Resume Transport 审计 pvisor run \ --name my-agent \ --stage ../stage-001 \ - --mount /opt/tool:read \ + --mount /opt/tool:stage \ --access '**/.ssh:deny' \ --overlayfs-backend directory \ --overlaynet-allow api.openai.com:443 \ @@ -431,7 +442,7 @@ command = ["my-agent"] [filesystem] stage = "../stage-001" -mount = [{ source = "/opt/tool", access = "read" }] +mount = [{ source = "/opt/tool", access = "stage" }] access = [{ path = "**/.ssh", level = "deny" }] backend = "directory" @@ -511,8 +522,8 @@ rootfs。 `--executor` 时选择 VM executor。`--rootfs ` 使用准备好的目录, `--rootfs image=` 使用 OCI 镜像或镜像路径;三者互斥,host rootfs 在 macOS 上被拒绝。这是统一 rootfs 语法; -`--mount SOURCE[:TARGET]:read|stage|write` 向 guest 提供额外宿主路径,当前工作区是隐式底层; -省略 TARGET 时保持原路径。工作区改动进入配置的 stage,省略 `--stage` 则在 Run 退出时丢弃; +`--mount SOURCE[:TARGET]:stage` 为 guest 工作区组合额外底层,当前工作区是隐式底层; +read/write 显式宿主共享目前仅支持 host executor。工作区改动进入指定 stage 或默认 Job 存储,退出后保留; VM 根目录其他写入使用临时 upper,并在 VM 退出时丢弃。 合并后的 rootfs 是 guest `/`,`/workspace` 成为 guest cwd。在 Linux 和 @@ -595,3 +606,17 @@ staging 数据,但保留紧凑的 Run/Overlay 元数据、apply ledger 和 cap - [审查并应用 Effect](../guides/review-apply.md):过滤且可重复的 apply。 - [网络控制](../guides/network.md) 与 [捕获轨迹](../guides/capture.md):其他 Effect 维度。 + +### 共享镜像文件缓存 + +`pvisor cache serve` 在前台提供 OCI 镜像文件服务;`cache prepare IMAGE`、 +`cache list DIGEST [PATH]`、`cache stat DIGEST PATH` 和 `cache read DIGEST PATH` +通过 `PERSISTING_PVISOR_CACHE_SERVER` 访问它。默认使用用户缓存目录下的 +`persisting/pvisor/cache.sock` Unix socket。服务端可用 `--image-store DIR` +指定已有 OCI store。文件读取支持分段和 SHA-256 校验。 + +VM 镜像启动会自动探测默认 socket;服务可用时,将远程镜像挂为只读 FUSE lower, +以 1 MiB 数据块按需读取并持久缓存。默认 socket 不存在或已失效时走本地 OCI 准备。 +显式指定服务端后连接失败会报错;`PERSISTING_PVISOR_CACHE_SERVER=off` 强制本地准备。 +显式 rootfs 目录和原生 container executor 保持原有行为。 +完整协议、限制和 SSH 远程访问方式见 [共享镜像缓存协议](../../../shared-image-cache.md)。 diff --git a/scripts/ci/vm_fs_smoke.py b/scripts/ci/vm_fs_smoke.py new file mode 100644 index 00000000..25c944b8 --- /dev/null +++ b/scripts/ci/vm_fs_smoke.py @@ -0,0 +1,218 @@ +#!/usr/bin/env python3 +"""Run inside a Linux pVisor VM: python3 scripts/ci/vm_fs_smoke.py [directory ...].""" + +import errno +import fcntl +import io +import mmap +import os +import shutil +import socket +import sqlite3 +import stat +import subprocess +import sys +import tarfile +import tempfile +import traceback +from pathlib import Path + + +def expect_errno(code, operation): + try: + operation() + except OSError as error: + assert error.errno == code, (error.errno, code, str(error)) + else: + raise AssertionError(f"expected errno {code}") + + +def read_write(p): + f = p / "file" + f.write_bytes(b"hello") + with f.open("ab") as stream: + stream.write(b" world") + stream.flush() + os.fsync(stream.fileno()) + assert f.read_bytes() == b"hello world" + with f.open("r+b") as stream: + stream.seek(6) + stream.write(b"VM") + stream.truncate(8) + assert f.read_bytes() == b"hello VM" + + +def sparse_mmap(p): + with (p / "file").open("w+b", buffering=0) as stream: + stream.truncate(1024 * 1024) + assert stream.read(4096) == bytes(4096) + with mmap.mmap(stream.fileno(), 0) as mapping: + mapping[4096:4100] = b"test" + mapping.flush() + stream.seek(4096) + assert stream.read(4) == b"test" + + +def permissions(p): + f = p / "file" + f.touch() + f.chmod(0o640) + os.chown(f, 1234, 1235) + os.utime(f, ns=(1_700_000_000_123456789, 1_700_000_001_987654321)) + info = f.stat() + assert stat.S_IMODE(info.st_mode) == 0o640 + assert (info.st_uid, info.st_gid) == (1234, 1235) + assert info.st_mtime_ns == 1_700_000_001_987654321 + + +def symlinks(p): + (p / "file").write_text("data") + (p / "link").symlink_to("file") + assert os.readlink(p / "link") == "file" + assert (p / "link").read_text() == "data" + (p / "file").unlink() + assert (p / "link").is_symlink() + expect_errno(errno.ENOENT, lambda: (p / "link").read_bytes()) + (p / "loop").symlink_to("loop") + expect_errno(errno.ELOOP, lambda: (p / "loop").read_bytes()) + + +def hardlinks(p): + (p / "file").write_bytes(b"first") + os.link(p / "file", p / "alias") + assert os.path.samefile(p / "file", p / "alias"), "hard links have different inode numbers" + assert (p / "alias").stat().st_nlink == 2 + (p / "alias").write_bytes(b"second") + assert (p / "file").read_bytes() == b"second" + (p / "file").unlink() + assert (p / "alias").read_bytes() == b"second" + + +def open_unlinked(p): + f = p / "file" + with f.open("w+b", buffering=0) as stream: + stream.write(b"original") + f.unlink() + assert os.fstat(stream.fileno()).st_size == 8 + stream.seek(0) + assert stream.read() == b"original" + stream.write(b"!") + os.fsync(stream.fileno()) + os.ftruncate(stream.fileno(), 3) + os.fchmod(stream.fileno(), 0o600) + assert os.fstat(stream.fileno()).st_size == 3 + + +def atomic_replace(p): + f = p / "file" + f.write_bytes(b"old") + with f.open("rb") as old: + (p / "new").write_bytes(b"replacement") + os.replace(p / "new", f) + assert old.read() == b"old" + assert os.fstat(old.fileno()).st_size == 3 + assert f.read_bytes() == b"replacement" + + +def directories(p): + (p / "tree/child").mkdir(parents=True) + for i in range(200): + (p / "tree/child" / str(i)).touch() + assert len(list((p / "tree/child").iterdir())) == 200 + expect_errno(errno.ENOTEMPTY, lambda: (p / "tree").rmdir()) + (p / "tree").rename(p / "moved") + fd = os.open(p / "moved/child", os.O_RDONLY | os.O_DIRECTORY) + try: + assert stat.S_ISDIR(os.fstat(fd).st_mode) + os.fsync(fd) + finally: + os.close(fd) + shutil.rmtree(p / "moved") + assert not (p / "moved").exists() + + +def names_errors(p): + for name in ["空 格.txt", ".hidden", "a" * 255]: + (p / name).write_text(name) + assert (p / name).read_text() == name + expect_errno(errno.ENAMETOOLONG, lambda: (p / ("b" * 256)).touch()) + expect_errno(errno.EEXIST, lambda: os.open(p / ".hidden", os.O_CREAT | os.O_EXCL)) + expect_errno(errno.ENOTDIR, lambda: (p / ".hidden/child").stat()) + + +def locks(p): + with (p / "lock").open("w") as stream: + fcntl.flock(stream, fcntl.LOCK_EX | fcntl.LOCK_NB) + code = ( + "import fcntl,sys; f=open(sys.argv[1],'w'); fcntl.flock(f,fcntl.LOCK_EX|fcntl.LOCK_NB)" + ) + result = subprocess.run([sys.executable, "-c", code, str(p / "lock")], capture_output=True) + assert result.returncode != 0, "second process acquired exclusive lock" + fcntl.flock(stream, fcntl.LOCK_UN) + + +def archive(p): + with tarfile.open(p / "data.tar", "w") as archive: + member = tarfile.TarInfo("sub/file") + member.size = 4 + member.mode = 0o640 + archive.addfile(member, io.BytesIO(b"data")) + (p / "out").mkdir() + subprocess.run(["tar", "xf", str(p / "data.tar"), "-C", str(p / "out")], check=True) + assert (p / "out/sub/file").read_bytes() == b"data" + assert stat.S_IMODE((p / "out/sub/file").stat().st_mode) == 0o640 + + +def sqlite_transactions(p): + with sqlite3.connect(p / "db") as db: + assert db.execute("PRAGMA journal_mode=WAL").fetchone()[0] == "wal" + db.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, value TEXT)") + db.executemany("INSERT INTO items(value) VALUES (?)", [(str(i),) for i in range(1000)]) + db.commit() + db.execute("DELETE FROM items") + db.rollback() + assert db.execute("SELECT count(*) FROM items").fetchone()[0] == 1000 + assert db.execute("PRAGMA integrity_check").fetchone()[0] == "ok" + + +def fifo_socket(p): + os.mkfifo(p / "fifo") + fd = os.open(p / "fifo", os.O_RDWR | os.O_NONBLOCK) + try: + os.write(fd, b"data") + assert os.read(fd, 4) == b"data" + finally: + os.close(fd) + with socket.socket(socket.AF_UNIX) as server: + server.bind(str(p / "socket")) + assert stat.S_ISSOCK((p / "socket").stat().st_mode) + + +checks = [ + read_write, + sparse_mmap, + permissions, + symlinks, + hardlinks, + open_unlinked, + atomic_replace, + directories, + names_errors, + locks, + archive, + fifo_socket, + sqlite_transactions, +] +failures = [] +for base in sys.argv[1:] or ["/var/tmp", "."]: + for check in checks: + try: + with tempfile.TemporaryDirectory(prefix="pvisor-fs-", dir=base) as directory: + check(Path(directory).absolute()) + print(f"PASS {base}: {check.__name__}", flush=True) + except Exception as error: + traceback.print_exc() + failures.append((base, check.__name__, repr(error))) + print(f"FAIL {base}: {check.__name__}: {error!r}", flush=True) +print(f"{len(checks) * len(sys.argv[1:] or ['/var/tmp', '.'])} checks; {len(failures)} failures") +sys.exit(bool(failures)) diff --git a/vendor/krun-devices/src/virtio/fs/macos/passthrough.rs b/vendor/krun-devices/src/virtio/fs/macos/passthrough.rs index af712cbb..e5fa507b 100644 --- a/vendor/krun-devices/src/virtio/fs/macos/passthrough.rs +++ b/vendor/krun-devices/src/virtio/fs/macos/passthrough.rs @@ -49,6 +49,9 @@ struct InodeAltKey { } struct InodeData { + // Filesystems such as FSKit do not support /.vol inode paths. Pin the + // object and ask F_GETPATH for its current name, including after renames. + path_fd: Option, inode: Inode, ino: u64, dev: i32, @@ -155,6 +158,14 @@ struct HandleData { dirstream: Mutex, } +fn descriptor_path(file: &File) -> io::Result { + let mut path = [0u8; libc::PATH_MAX as usize]; + if unsafe { libc::fcntl(file.as_raw_fd(), libc::F_GETPATH, path.as_mut_ptr()) } < 0 { + return Err(linux_error(io::Error::last_os_error())); + } + CStr::from_bytes_until_nul(&path).map(CStr::to_owned).map_err(|_| einval()) +} + fn ebadf() -> io::Error { linux_error(io::Error::from_raw_os_error(libc::EBADF)) } @@ -642,6 +653,7 @@ impl Default for Config { /// directory ends up as the root of the file system process. One way to accomplish this is via a /// combination of mount namespaces and the pivot_root system call. pub struct PassthroughFs { + fd_paths: bool, inodes: RwLock>>, inode_alloc: Arc, @@ -673,9 +685,20 @@ impl PassthroughFs { return Err(linux_error(io::Error::last_os_error())); } + let mut stat = MaybeUninit::::zeroed(); + let fd_paths = unsafe { + if libc::fstat(fd, stat.as_mut_ptr()) == 0 { + let stat = stat.assume_init(); + let path = CString::new(format!("/.vol/{}/{}", stat.st_dev, stat.st_ino)).unwrap(); + libc::access(path.as_ptr(), libc::F_OK) != 0 + } else { + false + } + }; unsafe { libc::close(fd) }; Ok(PassthroughFs { + fd_paths, inodes: RwLock::new(MultikeyBTreeMap::new()), inode_alloc, @@ -711,7 +734,10 @@ impl PassthroughFs { } } - Ok(InodeHandle::Path(cstr)) + Ok(InodeHandle::Path(match &data.path_fd { + Some(fd) => descriptor_path(fd)?, + None => cstr, + })) } fn name_to_path(&self, parent: Inode, name: &CStr) -> io::Result { @@ -728,6 +754,13 @@ impl PassthroughFs { .cloned() .ok_or_else(ebadf)?; + if let Some(fd) = &data.path_fd { + let mut bytes = descriptor_path(fd)?.into_bytes(); + bytes.push(b'/'); + bytes.extend_from_slice(name.to_bytes()); + return CString::new(bytes).map_err(|_| einval()); + } + let cstr = CString::new(format!( "/.vol/{}/{}/{}", data.dev, @@ -1315,6 +1348,7 @@ impl FileSystem for PassthroughFs { dev: st.st_dev, }, Arc::new(InodeData { + path_fd: if self.fd_paths { Some(f.try_clone()?) } else { None }, inode: fuse::ROOT_ID, ino: st.st_ino, dev: st.st_dev, @@ -1408,6 +1442,11 @@ impl FileSystem for PassthroughFs { dev: st.st_dev, }, Arc::new(InodeData { + path_fd: if self.fd_paths { + let fd = unsafe { libc::open(c_path.as_ptr(), libc::O_EVTONLY | libc::O_SYMLINK | libc::O_CLOEXEC) }; + if fd < 0 { return Err(linux_error(io::Error::last_os_error())); } + Some(unsafe { File::from_raw_fd(fd) }) + } else { None }, inode, ino: st.st_ino, dev: st.st_dev, @@ -1752,8 +1791,21 @@ impl FileSystem for PassthroughFs { &self, ctx: Context, inode: Inode, - _handle: Option, + handle: Option, ) -> io::Result<(bindings::stat64, Duration)> { + if let Some(handle) = handle { + let hd = self + .handles + .read() + .unwrap() + .get(&handle) + .filter(|hd| hd.inode == inode) + .cloned() + .ok_or_else(ebadf)?; + let file = hd.file.read().unwrap(); + let st = fstat(&ctx, self.cfg.semantics, file.as_raw_fd(), false)?; + return Ok((st, self.cfg.attr_timeout)); + } self.do_getattr(&ctx, inode) } @@ -2761,3 +2813,28 @@ impl FileSystem for PassthroughFs { Ok(()) } } + +#[cfg(test)] +mod path_handle_tests { + use super::*; + + #[test] + fn descriptor_paths_follow_renames_without_volfs() { + let root = tempfile::tempdir().unwrap(); + std::fs::write(root.path().join("file"), b"data").unwrap(); + let mut filesystem = PassthroughFs::new( + Config { root_dir: root.path().to_string_lossy().into_owned(), ..Default::default() }, + Arc::new(InodeAllocator::new()), + ).unwrap(); + filesystem.fd_paths = true; // Exercise the FSKit path on any host volume. + filesystem.init(FsOptions::empty()).unwrap(); + let ctx = Context { uid: 0, gid: 0, pid: 0 }; + let entry = filesystem.lookup(ctx, fuse::ROOT_ID, c"file").unwrap(); + std::fs::rename(root.path().join("file"), root.path().join("renamed")).unwrap(); + let InodeHandle::Path(path) = filesystem.inode_to_handle(entry.inode, false).unwrap() else { + panic!("expected current descriptor path"); + }; + assert!(path.to_bytes().ends_with(b"/renamed")); + assert_eq!(filesystem.do_getattr(&ctx, entry.inode).unwrap().0.st_size, 4); + } +} diff --git a/vendor/krun-devices/src/virtio/fs/overlay.rs b/vendor/krun-devices/src/virtio/fs/overlay.rs index a0718d74..cea9cd10 100644 --- a/vendor/krun-devices/src/virtio/fs/overlay.rs +++ b/vendor/krun-devices/src/virtio/fs/overlay.rs @@ -15,6 +15,7 @@ use std::time::Duration; use persisting_overlay_core::OverlayCore; +use super::super::linux_errno::linux_error; use super::bindings; use super::filesystem::{ Context, DirEntry, Entry, Extensions, FileSystem, FsOptions, GetxattrReply, ListxattrReply, @@ -44,6 +45,7 @@ struct Layer(usize); #[derive(Debug)] struct FileHandle { + overlay_inode: u64, layer: Layer, inode: u64, handle: u64, @@ -145,6 +147,7 @@ impl OverlayFs { fn child(&self, parent: u64, name: &CStr) -> io::Result { OverlayCore::child(&self.path(parent)?, OsStr::from_bytes(name.to_bytes())) + .map_err(linux_error) } fn allocate_inode(&self, path: PathBuf) -> u64 { @@ -248,13 +251,15 @@ impl OverlayFs { } fn writable_inner(&self, ctx: Context, path: &Path) -> io::Result { - self.core.copy_up(path)?; + // OverlayCore reports host errno; passthrough already reports Linux + // errno. Convert only at the core boundary, never the whole request. + self.core.copy_up(path).map_err(linux_error)?; self.inner_inode(Layer(0), path, ctx) } fn upper_parent(&self, ctx: Context, path: &Path) -> io::Result<(u64, CString)> { - self.core.clear_whiteout(path)?; - self.core.ensure_upper_parents(path)?; + self.core.clear_whiteout(path).map_err(linux_error)?; + self.core.ensure_upper_parents(path).map_err(linux_error)?; let parent = path.parent().unwrap_or_else(|| Path::new("")); let name = path .file_name() @@ -318,7 +323,7 @@ impl FileSystem for OverlayFs { .lock() .map_err(|_| io::Error::from_raw_os_error(libc::EIO))?; let path = self.child(parent, name)?; - self.core.metadata(&path)?; + self.core.metadata(&path).map_err(linux_error)?; let inode = self.allocate_inode(path.clone()); self.entry(ctx, &path, inode) } @@ -327,12 +332,34 @@ impl FileSystem for OverlayFs { &self, ctx: Context, inode: u64, - _handle: Option, + handle: Option, ) -> io::Result<(bindings::stat64, Duration)> { let _operation = self .operation_lock .lock() .map_err(|_| io::Error::from_raw_os_error(libc::EIO))?; + // GETATTR may omit FH (e.g. stat through /proc/self/fd). An unlinked + // inode still belongs to its open file, even after its path is gone. + let handle = handle.or_else(|| { + if self.path(inode).is_ok() { + return None; + } + // ponytail: scan open handles only for detached inodes; index by + // inode if workloads with many deleted-open files make this costly. + self.handles.lock().unwrap().iter().find_map(|(id, h)| { + matches!(h, Handle::File(h) if h.overlay_inode == inode).then_some(*id) + }) + }); + if let Some(handle) = handle { + return self.with_file_handle(handle, |fs, h| { + if h.overlay_inode != inode { + return Err(io::Error::from_raw_os_error(libc::EBADF)); + } + let (mut attr, timeout) = fs.getattr(ctx, h.inode, Some(h.handle))?; + attr.st_ino = inode as _; + Ok((attr, timeout)) + }); + } let entry = self.entry(ctx, &self.path(inode)?, inode)?; Ok((entry.attr, entry.attr_timeout)) } @@ -454,7 +481,7 @@ impl FileSystem for OverlayFs { .lock() .map_err(|_| io::Error::from_raw_os_error(libc::EIO))?; let path = self.child(parent, name)?; - self.core.remove(&path, false)?; + self.core.remove(&path, false).map_err(linux_error)?; self.remove_path(&path); Ok(()) } @@ -465,7 +492,7 @@ impl FileSystem for OverlayFs { .lock() .map_err(|_| io::Error::from_raw_os_error(libc::EIO))?; let path = self.child(parent, name)?; - self.core.remove(&path, true)?; + self.core.remove(&path, true).map_err(linux_error)?; self.remove_path(&path); Ok(()) } @@ -486,10 +513,10 @@ impl FileSystem for OverlayFs { let old = self.child(olddir, oldname)?; let new = self.child(newdir, newname)?; match flags { - 0 => self.core.rename(&old, &new, false)?, - RENAME_NOREPLACE => self.core.rename(&old, &new, true)?, - RENAME_EXCHANGE => self.core.exchange(&old, &new)?, - _ => return Err(io::Error::from_raw_os_error(libc::ENOTSUP)), + 0 => self.core.rename(&old, &new, false).map_err(linux_error)?, + RENAME_NOREPLACE => self.core.rename(&old, &new, true).map_err(linux_error)?, + RENAME_EXCHANGE => self.core.exchange(&old, &new).map_err(linux_error)?, + _ => return Err(linux_error(io::Error::from_raw_os_error(libc::ENOTSUP))), } if flags == RENAME_EXCHANGE { let marker = PathBuf::from(format!(".pvisor-exchange-{}", self.inode_alloc.next())); @@ -510,7 +537,9 @@ impl FileSystem for OverlayFs { .map_err(|_| io::Error::from_raw_os_error(libc::EIO))?; let source = self.path(inode)?; let destination = self.child(newparent, newname)?; - self.core.hard_link(&source, &destination)?; + self.core + .hard_link(&source, &destination) + .map_err(linux_error)?; let new_inode = self.allocate_inode(destination.clone()); self.entry(ctx, &destination, new_inode) } @@ -529,13 +558,19 @@ impl FileSystem for OverlayFs { let path = self.path(inode)?; // Symlinks are resolved by the guest through overlay lookups, never by // the passthrough layer against an unfiltered lower directory. - if self.core.metadata(&path)?.file_type().is_symlink() { - return Err(io::Error::from_raw_os_error(libc::ELOOP)); + if self + .core + .metadata(&path) + .map_err(linux_error)? + .file_type() + .is_symlink() + { + return Err(linux_error(io::Error::from_raw_os_error(libc::ELOOP))); } let writing = flags as i32 & libc::O_ACCMODE != libc::O_RDONLY || flags as i32 & (libc::O_APPEND | libc::O_TRUNC) != 0; let layer = if writing { - self.core.copy_up(&path)?; + self.core.copy_up(&path).map_err(linux_error)?; Layer(0) } else { self.layer(&path)? @@ -544,6 +579,7 @@ impl FileSystem for OverlayFs { let (handle, options) = self.layers[layer.0].open(ctx, inner, kill_priv, flags)?; let handle = handle.ok_or_else(|| io::Error::from_raw_os_error(libc::EIO))?; let id = self.allocate_handle(Handle::File(FileHandle { + overlay_inode: inode, layer, inode: inner, handle, @@ -587,6 +623,7 @@ impl FileSystem for OverlayFs { entry.attr.st_ino = inode as _; let handle = handle.ok_or_else(|| io::Error::from_raw_os_error(libc::EIO))?; let id = self.allocate_handle(Handle::File(FileHandle { + overlay_inode: inode, layer: Layer(0), inode: inner_inode, handle, @@ -786,8 +823,8 @@ impl FileSystem for OverlayFs { .map_err(|_| io::Error::from_raw_os_error(libc::EIO))?; let path = self.path(inode)?; let mut items = Vec::new(); - for name in self.core.list_names(&path)? { - let child = OverlayCore::child(&path, &name)?; + for name in self.core.list_names(&path).map_err(linux_error)? { + let child = OverlayCore::child(&path, &name).map_err(linux_error)?; let child_inode = self.allocate_inode(child.clone()); let entry = self.entry(ctx, &child, child_inode)?; items.push(DirectoryItem { @@ -1001,6 +1038,26 @@ mod tests { gid: 0, pid: 1, }; + // Linux ENOTEMPTY is 39; macOS 66 would become EREMOTE in the guest, + // preventing dpkg from falling back to recursive directory cleanup. + std::fs::create_dir(lower.join("directory")).unwrap(); + std::fs::write(lower.join("directory/file"), b"data").unwrap(); + for copy_up in [false, true] { + if copy_up { + fs.core.copy_up(Path::new("directory")).unwrap(); + } + assert_eq!( + fs.rmdir(ctx, fuse::ROOT_ID, c"directory") + .unwrap_err() + .raw_os_error(), + Some(39), + ); + } + let directory = fs.lookup(ctx, fuse::ROOT_ID, c"directory").unwrap(); + fs.unlink(ctx, directory.inode, c"file").unwrap(); + fs.rmdir(ctx, fuse::ROOT_ID, c"directory").unwrap(); + assert!(fs.lookup(ctx, fuse::ROOT_ID, c"directory").is_err()); + let original = CString::new("original").unwrap(); fs.lookup(ctx, fuse::ROOT_ID, &original).unwrap(); fs.unlink(ctx, fuse::ROOT_ID, &original).unwrap(); @@ -1034,6 +1091,94 @@ mod tests { assert!(!lower.join("created").exists()); } + #[test] + fn getattr_keeps_open_file_identity_after_unlink_or_replacement() { + for rename in [false, true] { + let temp = tempfile::tempdir().unwrap(); + let lower = temp.path().join("lower"); + let upper = temp.path().join("upper"); + std::fs::create_dir(&lower).unwrap(); + let fs = OverlayFs::new( + Config { + lower_dirs: vec![lower.to_string_lossy().into_owned()], + upper_dir: upper.to_string_lossy().into_owned(), + work_dir: None, + preimage_dir: None, + excluded_paths: vec![], + access_policy: Default::default(), + semantics: passthrough::PermissionSemantics::LinuxComplete, + }, + Arc::new(InodeAllocator::new()), + ) + .unwrap(); + fs.init(FsOptions::empty()).unwrap(); + let ctx = Context { + uid: 0, + gid: 0, + pid: 1, + }; + let (entry, handle, _) = fs + .create( + ctx, + fuse::ROOT_ID, + c"temporary", + libc::S_IFREG as u32 | 0o600, + false, + libc::O_RDWR as u32, + 0, + Extensions::default(), + ) + .unwrap(); + std::fs::write(upper.join("temporary"), b"original").unwrap(); + assert_eq!(fs.getattr(ctx, entry.inode, handle).unwrap().0.st_size, 8); + if rename { + std::fs::write(upper.join("replacement"), b"new").unwrap(); + fs.rename( + ctx, + fuse::ROOT_ID, + c"replacement", + fuse::ROOT_ID, + c"temporary", + 0, + ) + .unwrap(); + } else { + fs.unlink(ctx, fuse::ROOT_ID, c"temporary").unwrap(); + assert!(fs.lookup(ctx, fuse::ROOT_ID, c"temporary").is_err()); + } + for fh in [handle, None] { + let (attr, _) = fs.getattr(ctx, entry.inode, fh).unwrap(); + assert_eq!(attr.st_size, 8); + assert_eq!(attr.st_ino, entry.inode); + assert_eq!(attr.st_nlink, 0); + } + if !rename { + std::fs::write(upper.join("temporary"), b"new").unwrap(); + } + let replacement = fs.lookup(ctx, fuse::ROOT_ID, c"temporary").unwrap(); + assert_ne!(replacement.inode, entry.inode); + assert_eq!(replacement.attr.st_size, 3); + assert_eq!(fs.getattr(ctx, entry.inode, None).unwrap().0.st_size, 8); + assert_eq!( + fs.getattr(ctx, replacement.inode, handle) + .unwrap_err() + .raw_os_error(), + Some(libc::EBADF) + ); + fs.release( + ctx, + entry.inode, + libc::O_RDWR as u32, + handle.unwrap(), + false, + false, + None, + ) + .unwrap(); + assert!(fs.getattr(ctx, entry.inode, handle).is_err()); + } + } + #[test] fn shares_inode_allocator_with_virtual_entries() { let temp = tempfile::tempdir().unwrap(); From 2b0e57f3e31319fff1407ddcea949f2287930e35 Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 29 Sep 2026 23:49:06 +0800 Subject: [PATCH 2/9] Reorganize pVisor modules by responsibility Move CLI-adjacent runtime and executor code into focused `executor/`, `image/`, and `runtime/` submodules; split the shared image cache into protocol, transport, client, server, and CLI files; and group replay protocol bridges under `bridge/`. Add host diagnostics shared by services and the CLI, drop the thin control re-export in favor of direct `persisting_control` exports, and document the resulting ownership boundaries. Also bound the cache server's metadata caches with an LRU, paginate directory listings so frames stay within the protocol limit, serialize image preparation onto a dedicated worker queue, and add a `--refresh` flag to `pvisor cache prepare`. Trim the vendored Zellij border glyphs to the rounded single-line subset actually used. --- Cargo.lock | 1 + crates/persisting-pvisor/Cargo.toml | 2 + crates/persisting-pvisor/src/cache.rs | 1047 ----------------- crates/persisting-pvisor/src/cli/mod.rs | 6 +- crates/persisting-pvisor/src/cli/run.rs | 28 +- crates/persisting-pvisor/src/cli/tui.rs | 2 - .../src/cli/tui/zellij/README.md | 2 +- .../src/cli/tui/zellij/audit_ui.rs | 4 - .../src/cli/tui/zellij/border_glyphs.rs | 388 +----- .../src/cli/tui/zellij/input.rs | 17 +- .../src/cli/tui/zellij/mod.rs | 33 +- .../src/cli/tui/zellij/runtime.rs | 21 +- .../src/cli/tui/zellij/status_bar.rs | 6 +- .../src/cli/tui/zellij/view.rs | 114 +- crates/persisting-pvisor/src/control.rs | 7 - crates/persisting-pvisor/src/diagnostics.rs | 23 + .../src/{ => executor}/artifact.rs | 0 .../src/{ => executor}/container.rs | 6 +- .../src/{ => executor}/delegated.rs | 0 .../src/{executor.rs => executor/mod.rs} | 11 +- .../src/{ => executor}/process.rs | 40 +- .../src/{ => executor}/sandbox.rs | 0 .../src/{ => executor/vm}/firmware.rs | 4 +- .../src/{ => executor}/vm/mod.rs | 6 + .../src/{ => executor}/vm/supported.rs | 8 +- .../src/{ => executor}/vm/unsupported.rs | 0 .../persisting-pvisor/src/image/cache/cli.rs | 103 ++ .../src/image/cache/client.rs | 140 +++ .../src/{ => image}/cache/lazy.rs | 416 ++----- .../src/image/cache/lazy/tests.rs | 378 ++++++ .../persisting-pvisor/src/image/cache/mod.rs | 31 + .../src/{ => image}/cache/progress.rs | 16 +- .../src/image/cache/protocol.rs | 119 ++ .../src/image/cache/server.rs | 457 +++++++ .../{cache => image/cache/server}/metadata.rs | 54 +- .../src/image/cache/server/tests.rs | 332 ++++++ .../src/image/cache/transport.rs | 95 ++ crates/persisting-pvisor/src/image/mod.rs | 5 + .../persisting-pvisor/src/{ => image}/oci.rs | 228 +++- crates/persisting-pvisor/src/lib.rs | 75 +- .../src/{ => runtime}/agentctl.rs | 0 .../persisting-pvisor/src/runtime/attempt.rs | 10 +- .../src/{ => runtime}/bundle.rs | 2 +- .../src/{ => runtime}/checkpoint.rs | 0 .../src/{ => runtime}/event.rs | 0 crates/persisting-pvisor/src/runtime/mod.rs | 7 +- .../src/{pvisor.rs => runtime/run.rs} | 21 +- crates/persisting-pvisor/src/runtime/zcode.rs | 8 +- crates/persisting-pvisor/src/util.rs | 10 + .../src/adapter/claude_code.rs | 6 +- .../persisting-replay/src/adapter/generic.rs | 4 +- .../{claude_bridge.rs => bridge/claude.rs} | 2 +- .../src/{ => bridge}/claude_resume.rs | 0 .../src/{codex_bridge.rs => bridge/codex.rs} | 0 crates/persisting-replay/src/bridge/mod.rs | 6 + .../opencode.rs} | 0 crates/persisting-replay/src/lib.rs | 7 +- crates/persisting-replay/src/process.rs | 8 - docs/shared-image-cache.md | 91 +- docs/src/en/development/engineering.md | 94 ++ docs/src/en/reference/cli.md | 2 +- docs/src/zh/development/engineering.md | 90 ++ docs/src/zh/reference/cli.md | 2 +- 63 files changed, 2525 insertions(+), 2070 deletions(-) delete mode 100644 crates/persisting-pvisor/src/cache.rs delete mode 100644 crates/persisting-pvisor/src/control.rs create mode 100644 crates/persisting-pvisor/src/diagnostics.rs rename crates/persisting-pvisor/src/{ => executor}/artifact.rs (100%) rename crates/persisting-pvisor/src/{ => executor}/container.rs (99%) rename crates/persisting-pvisor/src/{ => executor}/delegated.rs (100%) rename crates/persisting-pvisor/src/{executor.rs => executor/mod.rs} (94%) rename crates/persisting-pvisor/src/{ => executor}/process.rs (98%) rename crates/persisting-pvisor/src/{ => executor}/sandbox.rs (100%) rename crates/persisting-pvisor/src/{ => executor/vm}/firmware.rs (98%) rename crates/persisting-pvisor/src/{ => executor}/vm/mod.rs (77%) rename crates/persisting-pvisor/src/{ => executor}/vm/supported.rs (99%) rename crates/persisting-pvisor/src/{ => executor}/vm/unsupported.rs (100%) create mode 100644 crates/persisting-pvisor/src/image/cache/cli.rs create mode 100644 crates/persisting-pvisor/src/image/cache/client.rs rename crates/persisting-pvisor/src/{ => image}/cache/lazy.rs (62%) create mode 100644 crates/persisting-pvisor/src/image/cache/lazy/tests.rs create mode 100644 crates/persisting-pvisor/src/image/cache/mod.rs rename crates/persisting-pvisor/src/{ => image}/cache/progress.rs (94%) create mode 100644 crates/persisting-pvisor/src/image/cache/protocol.rs create mode 100644 crates/persisting-pvisor/src/image/cache/server.rs rename crates/persisting-pvisor/src/{cache => image/cache/server}/metadata.rs (53%) create mode 100644 crates/persisting-pvisor/src/image/cache/server/tests.rs create mode 100644 crates/persisting-pvisor/src/image/cache/transport.rs create mode 100644 crates/persisting-pvisor/src/image/mod.rs rename crates/persisting-pvisor/src/{ => image}/oci.rs (81%) rename crates/persisting-pvisor/src/{ => runtime}/agentctl.rs (100%) rename crates/persisting-pvisor/src/{ => runtime}/bundle.rs (99%) rename crates/persisting-pvisor/src/{ => runtime}/checkpoint.rs (100%) rename crates/persisting-pvisor/src/{ => runtime}/event.rs (100%) rename crates/persisting-pvisor/src/{pvisor.rs => runtime/run.rs} (98%) rename crates/persisting-replay/src/{claude_bridge.rs => bridge/claude.rs} (99%) rename crates/persisting-replay/src/{ => bridge}/claude_resume.rs (100%) rename crates/persisting-replay/src/{codex_bridge.rs => bridge/codex.rs} (100%) create mode 100644 crates/persisting-replay/src/bridge/mod.rs rename crates/persisting-replay/src/{opencode_bridge.rs => bridge/opencode.rs} (100%) diff --git a/Cargo.lock b/Cargo.lock index 021a9abd..b41e608c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2988,6 +2988,7 @@ dependencies = [ "libc", "libkrun", "libloading", + "lru", "persisting-control", "persisting-gateway", "persisting-overlay-core", diff --git a/crates/persisting-pvisor/Cargo.toml b/crates/persisting-pvisor/Cargo.toml index 1f1ed930..497ac67c 100644 --- a/crates/persisting-pvisor/Cargo.toml +++ b/crates/persisting-pvisor/Cargo.toml @@ -22,6 +22,7 @@ flate2.workspace = true fs2.workspace = true globset.workspace = true libc.workspace = true +lru = "0.12.5" persisting-control.workspace = true persisting-gateway = { workspace = true, default-features = false } persisting-overlaynet.workspace = true @@ -57,6 +58,7 @@ proptest.workspace = true [build-dependencies] libloading.workspace = true + [target.'cfg(target_os = "macos")'.dependencies] fuser = { workspace = true, features = ["abi-7-19", "libfuse", "macfuse-5"] } diff --git a/crates/persisting-pvisor/src/cache.rs b/crates/persisting-pvisor/src/cache.rs deleted file mode 100644 index 1e5410a6..00000000 --- a/crates/persisting-pvisor/src/cache.rs +++ /dev/null @@ -1,1047 +0,0 @@ -//! Versioned, read-only OCI file service. See docs/shared-image-cache.md. -use anyhow::{Context, bail, ensure}; -use clap::{Args, Subcommand}; -use fs2::FileExt; -use serde::{Deserialize, Serialize, de::DeserializeOwned}; -use sha2::{Digest, Sha256}; -use std::ffi::{CStr, CString, OsStr}; -use std::fs::{self, File, OpenOptions}; -use std::io::{Read, Seek, SeekFrom, Write}; -use std::net::{SocketAddr, TcpListener, TcpStream}; -use std::os::fd::{AsRawFd, FromRawFd, IntoRawFd}; -use std::os::unix::ffi::OsStrExt; -use std::os::unix::fs::{FileTypeExt, OpenOptionsExt, PermissionsExt}; -use std::os::unix::net::{UnixListener, UnixStream}; -use std::path::{Component, Path, PathBuf}; -use std::sync::{Arc, Mutex, mpsc}; -use std::time::Duration; - -use crate::oci::ImageStore; - -mod metadata; -pub(crate) mod progress; -pub use progress::ImageTotals; - -#[cfg(any(target_os = "macos", target_os = "linux"))] -mod lazy; -#[cfg(any(target_os = "macos", target_os = "linux"))] -pub(crate) use lazy::{LazyMount, prepare_image}; - -pub const SERVER_ENV: &str = "PERSISTING_PVISOR_CACHE_SERVER"; -const TOKEN_ENV: &str = "PERSISTING_PVISOR_CACHE_TOKEN"; -const MAX_FRAME: usize = 1024 * 1024; -pub const MAX_READ: u32 = 1024 * 1024; -const TIMEOUT: Duration = Duration::from_secs(300); - -#[derive(Debug, Args)] -pub struct CacheArgs { - #[command(subcommand)] - command: CacheCommand, -} - -#[derive(Debug, Subcommand)] -enum CacheCommand { - /// Serve cached OCI files (foreground; Unix socket by default). - Serve { - /// unix:///absolute/path or tcp://127.0.0.1:PORT. Defaults to CACHE_SERVER. - #[arg(long)] - listen: Option, - /// OCI cache to serve and populate. - #[arg(long, env = "PERSISTING_PVISOR_IMAGE_STORE")] - image_store: Option, - }, - /// Resolve and prepare an image on the server; print its immutable digest. - Prepare { image: String }, - /// List one directory page. Paths are relative to the image root. - List { - digest: String, - path: Option, - #[arg(long, default_value_t = 0)] - offset: usize, - }, - /// Show file attributes without following symlinks. - Stat { digest: String, path: PathBuf }, - /// Stream one regular file to stdout. Does not follow symlinks. - Read { digest: String, path: PathBuf }, -} - -/// One request per connection. All paths are Unix bytes, relative to image root. -#[derive(Debug, Serialize, Deserialize)] -#[serde(tag = "op", rename_all = "snake_case", deny_unknown_fields)] -pub enum Request { - Ping, - Prepare { - image: String, - architecture: String, - }, - List { - digest: String, - path: Vec, - offset: usize, - }, - Stat { - digest: String, - path: Vec, - }, - Read { - digest: String, - path: Vec, - offset: u64, - length: u32, - }, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct Envelope { - version: u32, - token: Option, - request: Request, -} - -#[derive(Clone, Debug, Serialize, Deserialize)] -#[serde(tag = "status", rename_all = "snake_case")] -pub enum Response { - Ready, - Prepared { - #[serde(default)] - metadata_generation: Option, - #[serde(default)] - totals: Option, - digest: String, - architecture: String, - env: std::collections::BTreeMap, - entrypoint: Vec, - cmd: Vec, - }, - Entries { - names: Vec>, - next_offset: Option, - }, - Metadata { - kind: String, - size: u64, - mode: u32, - uid: u32, - gid: u32, - inode: u64, - nlink: u64, - mtime: i64, - mtime_nsec: i64, - target: Option>, - }, - Data { - length: u32, - sha256: String, - }, - Error { - code: String, - message: String, - }, -} - -pub fn default_endpoint() -> anyhow::Result { - let base = dirs::cache_dir().context("cannot find user cache directory")?; - Ok(format!( - "unix://{}", - base.join("persisting/pvisor/cache.sock").display() - )) -} - -fn endpoint_from_env() -> anyhow::Result { - match std::env::var(SERVER_ENV) { - Ok(value) => Ok(value), - Err(std::env::VarError::NotPresent) => default_endpoint(), - Err(error) => Err(error.into()), - } -} - -fn architecture() -> &'static str { - match std::env::consts::ARCH { - "aarch64" => "arm64", - "x86_64" => "amd64", - other => other, - } -} - -enum Endpoint { - Unix(PathBuf), - Tcp(SocketAddr), -} - -fn endpoint(value: &str) -> anyhow::Result { - if let Some(path) = value.strip_prefix("unix://") { - ensure!( - Path::new(path).is_absolute(), - "Unix socket path must be absolute" - ); - return Ok(Endpoint::Unix(path.into())); - } - if let Some(address) = value.strip_prefix("tcp://") { - let address: SocketAddr = address - .parse() - .context("TCP endpoint requires an IP address and port")?; - ensure!( - address.ip().is_loopback(), - "cache TCP is loopback-only; use an SSH tunnel for remote access" - ); - return Ok(Endpoint::Tcp(address)); - } - bail!("expected unix:///absolute/path or tcp://127.0.0.1:PORT") -} - -enum Stream { - Unix(UnixStream), - Tcp(TcpStream), -} - -impl Read for Stream { - fn read(&mut self, buf: &mut [u8]) -> std::io::Result { - match self { - Self::Unix(s) => s.read(buf), - Self::Tcp(s) => s.read(buf), - } - } -} -impl Write for Stream { - fn write(&mut self, buf: &[u8]) -> std::io::Result { - match self { - Self::Unix(s) => s.write(buf), - Self::Tcp(s) => s.write(buf), - } - } - fn flush(&mut self) -> std::io::Result<()> { - match self { - Self::Unix(s) => s.flush(), - Self::Tcp(s) => s.flush(), - } - } -} -impl Stream { - fn timeouts(&self, timeout: Duration) -> std::io::Result<()> { - match self { - Self::Unix(s) => { - s.set_read_timeout(Some(timeout))?; - s.set_write_timeout(Some(timeout)) - } - Self::Tcp(s) => { - s.set_read_timeout(Some(timeout))?; - s.set_write_timeout(Some(timeout)) - } - } - } -} - -fn read_frame(stream: &mut impl Read) -> anyhow::Result { - let mut header = [0; 4]; - stream.read_exact(&mut header)?; - let length = u32::from_be_bytes(header) as usize; - ensure!( - length > 0 && length <= MAX_FRAME, - "invalid cache frame length" - ); - let mut bytes = vec![0; length]; - stream.read_exact(&mut bytes)?; - Ok(serde_json::from_slice(&bytes)?) -} -fn write_frame(stream: &mut impl Write, value: &impl Serialize) -> anyhow::Result<()> { - let bytes = serde_json::to_vec(value)?; - ensure!( - bytes.len() <= MAX_FRAME, - "cache response exceeds frame limit" - ); - stream.write_all(&(bytes.len() as u32).to_be_bytes())?; - stream.write_all(&bytes)?; - Ok(()) -} - -#[derive(Debug, thiserror::Error)] -#[error("cache connection failed: {0}")] -struct CacheConnectError(#[source] std::io::Error); - -/// Blocking client; call from the host side, outside filesystem operation locks. -pub struct CacheClient { - endpoint: String, - token: Option, -} -impl CacheClient { - pub fn from_env() -> anyhow::Result { - Self::new(endpoint_from_env()?, std::env::var(TOKEN_ENV).ok()) - } - pub fn new(address: String, token: Option) -> anyhow::Result { - if matches!(endpoint(&address)?, Endpoint::Tcp(_)) { - ensure!( - token.as_ref().is_some_and(|s| !s.is_empty()), - "TCP requires {TOKEN_ENV}" - ); - } - Ok(Self { - endpoint: address, - token, - }) - } - /// Discover the default socket, or require an explicitly configured service. - pub(crate) fn discover() -> anyhow::Result> { - let explicit = match std::env::var(SERVER_ENV) { - Ok(value) => Some(value), - Err(std::env::VarError::NotPresent) => None, - Err(error) => return Err(error.into()), - }; - if explicit.as_deref() == Some("off") { - return Ok(None); - } - let address = explicit.clone().map_or_else(default_endpoint, Ok)?; - Self::probe(address, std::env::var(TOKEN_ENV).ok(), explicit.is_some()) - } - - fn probe( - address: String, - token: Option, - explicit: bool, - ) -> anyhow::Result> { - let client = Self::new(address, token)?; - match client.request_timeout(Request::Ping, Duration::from_secs(2)) { - Ok((Response::Ready, _)) => Ok(Some(client)), - Ok(_) => bail!("cache server returned an incompatible handshake"), - Err(error) => { - let absent = error.downcast_ref::().is_some_and(|e| { - matches!( - e.0.kind(), - std::io::ErrorKind::NotFound | std::io::ErrorKind::ConnectionRefused - ) - }); - if !explicit && absent { - Ok(None) - } else { - Err(error.context("probe shared image cache")) - } - } - } - } - - /// The returned bytes are present only for `Read`, and are SHA-256 checked. - pub fn request(&self, request: Request) -> anyhow::Result<(Response, Vec)> { - self.request_timeout(request, TIMEOUT) - } - - fn request_timeout( - &self, - request: Request, - timeout: Duration, - ) -> anyhow::Result<(Response, Vec)> { - let expected = match &request { - Request::Read { length, .. } => Some(*length), - _ => None, - }; - let mut stream = match endpoint(&self.endpoint)? { - Endpoint::Unix(path) => Stream::Unix( - UnixStream::connect(path) - .map_err(CacheConnectError) - .with_context(|| { - format!( - "connect cache {}; start `pvisor cache serve`", - self.endpoint - ) - })?, - ), - Endpoint::Tcp(address) => Stream::Tcp( - TcpStream::connect_timeout(&address, Duration::from_secs(10)) - .map_err(CacheConnectError)?, - ), - }; - stream.timeouts(timeout)?; - write_frame( - &mut stream, - &Envelope { - version: 1, - token: self.token.clone(), - request, - }, - )?; - let response: Response = read_frame(&mut stream)?; - let mut body = Vec::new(); - match &response { - Response::Error { code, message } => { - let kind = match code.as_str() { - "not_found" => std::io::ErrorKind::NotFound, - "permission_denied" => std::io::ErrorKind::PermissionDenied, - _ => std::io::ErrorKind::Other, - }; - return Err(std::io::Error::new(kind, format!("cache {code}: {message}")).into()); - } - Response::Data { length, sha256 } => { - ensure!( - expected.is_some_and(|limit| *length <= limit) && *length <= MAX_READ, - "invalid cache data length" - ); - body.resize(*length as usize, 0); - stream.read_exact(&mut body)?; - ensure!(hash(&body) == *sha256, "cache data digest mismatch"); - } - _ => ensure!(expected.is_none(), "expected cache data response"), - } - Ok((response, body)) - } -} -fn hash(bytes: &[u8]) -> String { - format!("sha256:{}", crate::oci::encode_hex(&Sha256::digest(bytes))) -} - -// Every component is opened relative to its parent fd, without following links. -// This remains confined even if a directory is renamed during a request. -fn open_child(parent: &File, name: &OsStr, directory: bool) -> anyhow::Result { - let name = CString::new(name.as_bytes())?; - let flags = libc::O_RDONLY - | libc::O_CLOEXEC - | libc::O_NOFOLLOW - | libc::O_NONBLOCK - | if directory { libc::O_DIRECTORY } else { 0 }; - let fd = unsafe { libc::openat(parent.as_raw_fd(), name.as_ptr(), flags) }; - if fd < 0 { - return Err(std::io::Error::last_os_error().into()); - } - Ok(unsafe { File::from_raw_fd(fd) }) -} - -fn parent(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result<(File, Vec)> { - let digest = crate::oci::digest_hex(digest)?; - ensure!(!path.contains(&0), "NUL in cache path"); - let path = Path::new(OsStr::from_bytes(path)); - let components: Vec<_> = path.components().collect(); - ensure!( - components.iter().all(|c| matches!(c, Component::Normal(_))), - "cache path must be relative without dot or parent components" - ); - let roots = OpenOptions::new() - .read(true) - .custom_flags(libc::O_DIRECTORY | libc::O_NOFOLLOW) - .open(store.root.join("rootfs-v3/sha256"))?; - let mut directory = open_child(&roots, OsStr::new(digest), true)?; - for component in components.iter().take(components.len().saturating_sub(1)) { - directory = open_child(&directory, component.as_os_str(), true)?; - } - let name = components - .last() - .map_or_else(|| b".".to_vec(), |c| c.as_os_str().as_bytes().to_vec()); - Ok((directory, name)) -} - -fn directory_names(directory: File) -> anyhow::Result>> { - let raw = unsafe { libc::fdopendir(directory.as_raw_fd()) }; - if raw.is_null() { - return Err(std::io::Error::last_os_error().into()); - } - let _ = directory.into_raw_fd(); // fdopendir owns the descriptor on success. - struct Directory(*mut libc::DIR); - impl Drop for Directory { - fn drop(&mut self) { - unsafe { - libc::closedir(self.0); - } - } - } - let directory = Directory(raw); - let mut names = Vec::new(); - loop { - #[cfg(target_os = "macos")] - unsafe { - *libc::__error() = 0; - } - #[cfg(target_os = "linux")] - unsafe { - *libc::__errno_location() = 0; - } - let entry = unsafe { libc::readdir(directory.0) }; - if entry.is_null() { - let error = std::io::Error::last_os_error(); - if error.raw_os_error() != Some(0) { - return Err(error.into()); - } - break; - } - let name = unsafe { CStr::from_ptr((*entry).d_name.as_ptr()) }.to_bytes(); - if name != b"." && name != b".." { - names.push(name.to_vec()); - } - } - Ok(names) -} - -#[allow(clippy::unnecessary_cast)] // libc stat field widths differ by platform. -fn metadata_at(directory: &File, name: &[u8]) -> anyhow::Result { - let name = CString::new(name)?; - let mut m: libc::stat = unsafe { std::mem::zeroed() }; - if unsafe { - libc::fstatat( - directory.as_raw_fd(), - name.as_ptr(), - &mut m, - libc::AT_SYMLINK_NOFOLLOW, - ) - } != 0 - { - return Err(std::io::Error::last_os_error().into()); - } - let kind = match m.st_mode & libc::S_IFMT { - libc::S_IFREG => "file", - libc::S_IFDIR => "directory", - libc::S_IFLNK => "symlink", - _ => "special", - }; - let target = if kind == "symlink" { - let mut bytes = vec![0u8; 4096]; - let size = unsafe { - libc::readlinkat( - directory.as_raw_fd(), - name.as_ptr(), - bytes.as_mut_ptr().cast(), - bytes.len(), - ) - }; - if size < 0 { - return Err(std::io::Error::last_os_error().into()); - } - ensure!( - (size as usize) < bytes.len(), - "symlink target exceeds protocol limit" - ); - bytes.truncate(size as usize); - Some(bytes) - } else { - None - }; - Ok(Response::Metadata { - kind: kind.into(), - size: m.st_size as u64, - mode: m.st_mode as u32, - uid: m.st_uid, - gid: m.st_gid, - inode: m.st_ino as u64, - nlink: m.st_nlink as u64, - mtime: m.st_mtime as i64, - mtime_nsec: m.st_mtime_nsec as i64, - target, - }) -} - -fn handle(store: &ImageStore, request: Request) -> anyhow::Result<(Response, Vec)> { - let response = match request { - Request::Ping => Response::Ready, - Request::Prepare { - image, - architecture: requested, - } => { - let image = store.prepare_for_architecture(&image, &requested)?; - Response::Prepared { - metadata_generation: Some(metadata::generation(store, &image.digest)?), - totals: Some(progress::image_totals(store, &image.digest)?), - digest: image.digest, - architecture: requested, - env: image.env, - entrypoint: image.entrypoint, - cmd: image.cmd, - } - } - Request::List { - digest, - path, - offset, - } => { - let names = metadata::directory(store, &digest, &path)?; - ensure!(offset <= names.len(), "directory offset out of range"); - let end = offset.saturating_add(256).min(names.len()); - Response::Entries { - names: names[offset..end].to_vec(), - next_offset: (end < names.len()).then_some(end), - } - } - Request::Stat { digest, path } => metadata::stat(store, &digest, &path)?, - Request::Read { - digest, - path, - offset, - length, - } => { - ensure!( - length > 0 && length <= MAX_READ, - "read length must be 1..={MAX_READ}" - ); - let (directory, name) = parent(store, &digest, &path)?; - let mut file = open_child(&directory, OsStr::from_bytes(&name), false)?; - ensure!(file.metadata()?.is_file(), "only regular files can be read"); - file.seek(SeekFrom::Start(offset))?; - let mut body = Vec::new(); - file.take(length as u64).read_to_end(&mut body)?; - return Ok(( - Response::Data { - length: body.len() as u32, - sha256: hash(&body), - }, - body, - )); - } - }; - Ok((response, Vec::new())) -} - -fn ensure_same_user(socket: &UnixStream) -> anyhow::Result<()> { - #[cfg(target_os = "macos")] - let uid = { - let mut uid = 0; - let mut gid = 0; - if unsafe { libc::getpeereid(socket.as_raw_fd(), &mut uid, &mut gid) } != 0 { - return Err(std::io::Error::last_os_error().into()); - } - uid - }; - #[cfg(target_os = "linux")] - let uid = { - let mut credentials: libc::ucred = unsafe { std::mem::zeroed() }; - let mut length = std::mem::size_of_val(&credentials) as libc::socklen_t; - if unsafe { - libc::getsockopt( - socket.as_raw_fd(), - libc::SOL_SOCKET, - libc::SO_PEERCRED, - (&mut credentials as *mut libc::ucred).cast(), - &mut length, - ) - } != 0 - { - return Err(std::io::Error::last_os_error().into()); - } - credentials.uid - }; - #[cfg(not(any(target_os = "linux", target_os = "macos")))] - anyhow::bail!("cache peer authentication is supported only on Linux and macOS"); - #[cfg(any(target_os = "linux", target_os = "macos"))] - ensure!( - uid == unsafe { libc::geteuid() }, - "cache socket requires the same user" - ); - Ok(()) -} - -fn serve_connection( - mut stream: Stream, - store: &ImageStore, - token: Option<&str>, -) -> anyhow::Result<()> { - stream.timeouts(TIMEOUT)?; - if let Stream::Unix(socket) = &stream { - ensure_same_user(socket)?; - } - let result = (|| { - let envelope: Envelope = read_frame(&mut stream)?; - ensure!(envelope.version == 1, "unsupported cache protocol version"); - ensure!( - token.is_none() || envelope.token.as_deref() == token, - "cache authentication failed" - ); - handle(store, envelope.request) - })(); - let (response, body) = result.unwrap_or_else(|error: anyhow::Error| { - let code = match error.downcast_ref::().map(|e| e.kind()) { - Some(std::io::ErrorKind::NotFound) => "not_found", - Some(std::io::ErrorKind::PermissionDenied) => "permission_denied", - _ => "request_failed", - }; - ( - Response::Error { - code: code.into(), - message: format!("{error:#}"), - }, - Vec::new(), - ) - }); - write_frame(&mut stream, &response)?; - stream.write_all(&body)?; - Ok(()) -} - -fn serve(address: String, store: ImageStore, token: Option) -> anyhow::Result<()> { - let (send, receive) = mpsc::sync_channel::(16); - let receive = Arc::new(Mutex::new(receive)); - let store = Arc::new(store); - let token = Arc::new(token); - let dispatch = |stream| -> anyhow::Result<()> { - // A full queue closes the connection instead of allocating unbounded workers. - send.try_send(stream) - .map_err(|_| anyhow::anyhow!("cache server busy")) - }; - // Bind before starting workers so address conflicts fail without orphan workers. - enum Listener { - Unix(UnixListener, File), - Tcp(TcpListener), - } - let listener = match endpoint(&address)? { - Endpoint::Unix(path) => { - let directory = path.parent().context("socket requires parent directory")?; - fs::create_dir_all(directory)?; - let lock = OpenOptions::new() - .read(true) - .write(true) - .create(true) - .truncate(false) - .mode(0o600) - .custom_flags(libc::O_NOFOLLOW) - .open(path.with_extension("sock.lock"))?; - lock.try_lock_exclusive() - .context("cache server already running (socket lock held)")?; - match fs::symlink_metadata(&path) { - Ok(metadata) => { - ensure!( - metadata.file_type().is_socket(), - "refusing to replace non-socket path" - ); - match UnixStream::connect(&path) { - Ok(_) => bail!("cache socket already in use"), - Err(error) if error.kind() == std::io::ErrorKind::ConnectionRefused => { - fs::remove_file(&path)? - } - Err(error) => return Err(error.into()), - } - } - Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} - Err(error) => return Err(error.into()), - } - let listener = UnixListener::bind(&path)?; - fs::set_permissions(&path, fs::Permissions::from_mode(0o600))?; - Listener::Unix(listener, lock) - } - Endpoint::Tcp(address) => { - ensure!( - token.as_ref().as_ref().is_some_and(|t| !t.is_empty()), - "TCP requires {TOKEN_ENV}" - ); - Listener::Tcp(TcpListener::bind(address)?) - } - }; - for _ in 0..16 { - let receive = receive.clone(); - let store = store.clone(); - let token = token.clone(); - std::thread::spawn(move || { - loop { - let request = receive.lock().unwrap().recv(); - let Ok(stream) = request else { break }; - if let Err(error) = serve_connection(stream, &store, token.as_deref()) { - eprintln!("cache connection: {error}"); - } - } - }); - } - eprintln!("pvisor cache listening on {address}"); - match listener { - Listener::Unix(listener, _lock) => { - for stream in listener.incoming() { - let _ = dispatch(Stream::Unix(stream?)); - } - } - Listener::Tcp(listener) => { - for stream in listener.incoming() { - let _ = dispatch(Stream::Tcp(stream?)); - } - } - } - Ok(()) -} - -pub fn run(args: CacheArgs) -> anyhow::Result<()> { - if let CacheCommand::Serve { - listen, - image_store, - } = args.command - { - return serve( - listen.map_or_else(endpoint_from_env, Ok)?, - ImageStore::new(image_store)?, - std::env::var(TOKEN_ENV).ok(), - ); - } - let client = CacheClient::from_env()?; - let request = match args.command { - CacheCommand::Prepare { image } => Request::Prepare { - image, - architecture: architecture().into(), - }, - CacheCommand::List { - digest, - path, - offset, - } => Request::List { - digest, - path: path.unwrap_or_default().as_os_str().as_bytes().to_vec(), - offset, - }, - CacheCommand::Stat { digest, path } => Request::Stat { - digest, - path: path.as_os_str().as_bytes().to_vec(), - }, - CacheCommand::Read { digest, path } => { - let mut offset = 0; - let mut stdout = std::io::stdout().lock(); - loop { - let (_, body) = client.request(Request::Read { - digest: digest.clone(), - path: path.as_os_str().as_bytes().to_vec(), - offset, - length: MAX_READ, - })?; - stdout.write_all(&body)?; - offset += body.len() as u64; - if body.len() < MAX_READ as usize { - break; - } - } - return Ok(()); - } - CacheCommand::Serve { .. } => unreachable!(), - }; - let (response, _) = client.request(request)?; - println!("{}", serde_json::to_string_pretty(&response)?); - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use std::os::unix::fs::symlink; - - fn fixture() -> (tempfile::TempDir, ImageStore, String) { - let tmp = tempfile::tempdir().unwrap(); - let store = ImageStore::new(Some(tmp.path().join("store"))).unwrap(); - let digest = format!("sha256:{}", "a".repeat(64)); - let root = store.root.join("rootfs-v3/sha256").join("a".repeat(64)); - fs::create_dir(&root).unwrap(); - fs::write(root.join("hello"), b"hello world").unwrap(); - fs::create_dir(root.join("dir")).unwrap(); - symlink("hello", root.join("alias")).unwrap(); - symlink("/etc", root.join("escape")).unwrap(); - (tmp, store, digest) - } - - #[test] - fn server_metadata_reuses_directory_index_and_invalidates_rebuilt_root() { - let (_tmp, store, digest) = fixture(); - let first = metadata::directory(&store, &digest, b"").unwrap(); - let again = metadata::directory(&store, &digest, b"").unwrap(); - assert!( - Arc::ptr_eq(&first, &again), - "directory must not be scanned again" - ); - let Response::Metadata { size, .. } = metadata::stat(&store, &digest, b"hello").unwrap() - else { - panic!() - }; - assert_eq!(size, 11); - let generation = metadata::generation(&store, &digest).unwrap(); - let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); - fs::rename(&root, root.with_extension("old")).unwrap(); - fs::create_dir(&root).unwrap(); - fs::write(root.join("hello"), b"new").unwrap(); - assert_ne!(generation, metadata::generation(&store, &digest).unwrap()); - assert_eq!( - &*metadata::directory(&store, &digest, b"").unwrap(), - &[b"hello".to_vec()] - ); - let Response::Metadata { size, .. } = metadata::stat(&store, &digest, b"hello").unwrap() - else { - panic!() - }; - assert_eq!(size, 3); - } - - #[test] - fn unix_client_roundtrip_and_parallel_reads() { - let (tmp, store, digest) = fixture(); - let path = tmp.path().join("server.sock"); - let listener = UnixListener::bind(&path).unwrap(); - let server = std::thread::spawn(move || { - std::thread::scope(|scope| { - for connection in listener.incoming().take(8) { - let store = &store; - scope.spawn(move || { - serve_connection(Stream::Unix(connection.unwrap()), store, None).unwrap() - }); - } - }); - }); - std::thread::scope(|scope| { - for _ in 0..8 { - let address = format!("unix://{}", path.display()); - let digest = digest.clone(); - scope.spawn(move || { - let client = CacheClient::new(address, None).unwrap(); - let (_, body) = client - .request(Request::Read { - digest, - path: b"hello".to_vec(), - offset: 6, - length: 5, - }) - .unwrap(); - assert_eq!(body, b"world"); - }); - } - }); - server.join().unwrap(); - } - - #[test] - fn paths_metadata_pagination_and_eof() { - let (_tmp, store, digest) = fixture(); - let (response, _) = handle( - &store, - Request::List { - digest: digest.clone(), - path: vec![], - offset: 0, - }, - ) - .unwrap(); - match response { - Response::Entries { names, next_offset } => { - assert_eq!( - names, - [ - b"alias".to_vec(), - b"dir".to_vec(), - b"escape".to_vec(), - b"hello".to_vec() - ] - ); - assert!(next_offset.is_none()); - } - _ => panic!("expected directory"), - } - let (response, _) = handle( - &store, - Request::Stat { - digest: digest.clone(), - path: b"alias".to_vec(), - }, - ) - .unwrap(); - assert!( - matches!(response, Response::Metadata { target: Some(target), .. } if target == b"hello") - ); - for path in [ - b"../hello".as_slice(), - b"/etc/passwd", - b"escape/passwd", - b"alias", - b"hello\0", - ] { - assert!( - handle( - &store, - Request::Read { - digest: digest.clone(), - path: path.to_vec(), - offset: 0, - length: 10 - } - ) - .is_err() - ); - } - assert!( - handle( - &store, - Request::Read { - digest: "sha256:../../etc".into(), - path: b"passwd".to_vec(), - offset: 0, - length: 10 - } - ) - .is_err() - ); - assert!( - handle( - &store, - Request::Read { - digest: digest.clone(), - path: b"hello".to_vec(), - offset: 0, - length: MAX_READ + 1 - } - ) - .is_err() - ); - let (_, body) = handle( - &store, - Request::Read { - digest, - path: b"hello".to_vec(), - offset: 100, - length: 10, - }, - ) - .unwrap(); - assert!(body.is_empty()); - } - - #[test] - fn rejects_bad_frames_versions_and_tokens() { - assert!(read_frame::(&mut &u32::MAX.to_be_bytes()[..]).is_err()); - assert!(endpoint("tcp://0.0.0.0:9000").is_err()); - assert!(CacheClient::new("tcp://127.0.0.1:9000".into(), None).is_err()); - for (version, token) in [(2, Some("secret")), (1, Some("wrong")), (1, None)] { - let (_tmp, store, digest) = fixture(); - let (mut client, server) = UnixStream::pair().unwrap(); - let worker = std::thread::spawn(move || { - serve_connection(Stream::Unix(server), &store, Some("secret")).unwrap() - }); - write_frame( - &mut client, - &Envelope { - version, - token: token.map(str::to_owned), - request: Request::Stat { - digest, - path: b"hello".to_vec(), - }, - }, - ) - .unwrap(); - assert!(matches!( - read_frame::(&mut client).unwrap(), - Response::Error { .. } - )); - worker.join().unwrap(); - } - } - - #[test] - fn client_rejects_corrupt_content() { - let tmp = tempfile::tempdir().unwrap(); - let path = tmp.path().join("server.sock"); - let listener = UnixListener::bind(&path).unwrap(); - let worker = std::thread::spawn(move || { - let (mut socket, _) = listener.accept().unwrap(); - let _: Envelope = read_frame(&mut socket).unwrap(); - write_frame( - &mut socket, - &Response::Data { - length: 3, - sha256: hash(b"abc"), - }, - ) - .unwrap(); - socket.write_all(b"bad").unwrap(); - }); - let client = CacheClient::new(format!("unix://{}", path.display()), None).unwrap(); - let error = client - .request(Request::Read { - digest: "unused".into(), - path: b"file".to_vec(), - offset: 0, - length: 3, - }) - .unwrap_err(); - assert!(error.to_string().contains("digest mismatch")); - worker.join().unwrap(); - } -} diff --git a/crates/persisting-pvisor/src/cli/mod.rs b/crates/persisting-pvisor/src/cli/mod.rs index ff611a44..915f1b18 100644 --- a/crates/persisting-pvisor/src/cli/mod.rs +++ b/crates/persisting-pvisor/src/cli/mod.rs @@ -8,8 +8,6 @@ pub mod runtime; mod trajectory; #[cfg(unix)] mod tui; -#[cfg(unix)] -pub(crate) use tui::diagnostic; use clap::{Parser, Subcommand}; @@ -51,7 +49,7 @@ enum Command { Run(Box), /// Serve or query the shared OCI file cache. #[cfg(unix)] - Cache(crate::cache::CacheArgs), + Cache(crate::image::cache::CacheArgs), /// Apply selected staged changes from a stopped Job. Apply(runtime::ApplyArgs), /// Discard staged changes from a stopped Job. @@ -98,7 +96,7 @@ pub fn main() -> anyhow::Result<()> { } match parsed.command { #[cfg(unix)] - Command::Cache(args) => crate::cache::run(args)?, + Command::Cache(args) => crate::image::cache::run(args)?, Command::Run(args) => { let code = tokio::runtime::Runtime::new()?.block_on(run::run(*args))?; if code != 0 { diff --git a/crates/persisting-pvisor/src/cli/run.rs b/crates/persisting-pvisor/src/cli/run.rs index 91ff1272..48a0fa1a 100644 --- a/crates/persisting-pvisor/src/cli/run.rs +++ b/crates/persisting-pvisor/src/cli/run.rs @@ -98,7 +98,7 @@ use super::trajectory::{JsonlEventSink, JsonlWriter, jsonl_capture_sink}; macro_rules! run_log { ($($arg:tt)*) => {{ #[cfg(unix)] - super::tui::diagnostic(format_args!($($arg)*)); + crate::diagnostics::diagnostic(format_args!($($arg)*)); #[cfg(not(unix))] eprintln!($($arg)*); }}; @@ -959,11 +959,11 @@ async fn run_prepared_spec(args: RunArgs) -> anyhow::Result { wait.await? } }; - let output = crate::delegated::DelegatedRunOutput { + let output = crate::executor::delegated::DelegatedRunOutput { agentctl: agentctl.snapshot(), result, }; - let write_result = crate::delegated::write_result(&result_path, &output) + let write_result = crate::executor::delegated::write_result(&result_path, &output) .with_context(|| format!("write delegated RunResult to {}", result_path.display())); let cleanup_result = stage_guard .as_mut() @@ -1163,7 +1163,7 @@ async fn execute_config( ) -> anyhow::Result { normalize_filesystem_config(&mut config)?; resolve_default_vm_rootfs(&mut config)?; - let mut _image_mount: Option = None; + let mut _image_mount: Option = None; let prepared_image = if config.run.executor == RunExecutorKind::Vm && config.vm.rootfs.is_none() { let image = config @@ -1174,7 +1174,7 @@ async fn execute_config( let store = config.vm.image_store.clone(); run_log!("pVisor image: resolving {image}"); let (prepared, mount) = - tokio::task::spawn_blocking(move || crate::cache::prepare_image(&image, store)) + tokio::task::spawn_blocking(move || crate::image::cache::prepare_image(&image, store)) .await .context("OCI image preparation task failed")??; _image_mount = mount; @@ -1229,13 +1229,13 @@ async fn execute_config( all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), all(target_os = "macos", target_arch = "x86_64") )))] - if config.vm.library_dir.is_none() && crate::vm::bundled_firmware_dir().is_none() { + if config.vm.library_dir.is_none() && crate::executor::vm::bundled_firmware_dir().is_none() { run_log!( "pVisor firmware: resolving libkrunfw {}", - crate::firmware::VERSION + crate::executor::vm::firmware::VERSION ); let directory = tokio::task::spawn_blocking(|| { - crate::firmware::FirmwareStore::new()?.prepare() + crate::executor::vm::firmware::FirmwareStore::new()?.prepare() }) .await .context("libkrunfw preparation task failed")??; @@ -1275,7 +1275,7 @@ async fn execute_config( #[cfg(target_os = "linux")] let rootless_probe = (config.run.executor == RunExecutorKind::Host).then(|| { tokio::task::spawn_blocking(move || { - crate::process::rootless_runtime_available(!filesystem_isolated) + crate::executor::process::rootless_runtime_available(!filesystem_isolated) }) }); @@ -1504,8 +1504,10 @@ async fn execute_config( process.cwd = Some(workspace.display().to_string()); } if safe { - spec.metadata - .insert(crate::sandbox::REQUIRED_SANDBOX_KEY.into(), true.into()); + spec.metadata.insert( + crate::executor::sandbox::REQUIRED_SANDBOX_KEY.into(), + true.into(), + ); } spec.runtime.timeout_ms = config.run.timeout_ms; spec.runtime.resource_limits = config.run.resource_limits.clone(); @@ -1579,7 +1581,7 @@ async fn execute_config( if safe { spec.metadata - .insert(crate::sandbox::LANDLOCK_SANDBOX_KEY.into(), true.into()); + .insert(crate::executor::sandbox::LANDLOCK_SANDBOX_KEY.into(), true.into()); } { let network_boundary = if config.run.executor == RunExecutorKind::Vm @@ -3550,7 +3552,7 @@ sandbox = "required""# config.run.executor = RunExecutorKind::Vm; config.vm.rootfs = Some(temporary.path().to_path_buf()); config.vm.library_dir = Some(temporary.path().to_path_buf()); - std::fs::write(temporary.path().join(crate::vm::firmware_name()), []).unwrap(); + std::fs::write(temporary.path().join(crate::executor::vm::firmware_name()), []).unwrap(); config.overlayfs = Some(OverlayFsSettings { base: Some(temporary.path().to_path_buf()), ..OverlayFsSettings::default() diff --git a/crates/persisting-pvisor/src/cli/tui.rs b/crates/persisting-pvisor/src/cli/tui.rs index 7e298418..aad85ea7 100644 --- a/crates/persisting-pvisor/src/cli/tui.rs +++ b/crates/persisting-pvisor/src/cli/tui.rs @@ -3,5 +3,3 @@ mod zellij; pub(super) use zellij::{announce_stage, available, init_child_context, is_child, run}; - -pub(crate) use zellij::diagnostic; diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/README.md b/crates/persisting-pvisor/src/cli/tui/zellij/README.md index cb8c5ee3..1578188a 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/README.md +++ b/crates/persisting-pvisor/src/cli/tui/zellij/README.md @@ -10,5 +10,5 @@ These files are original pVisor code except for the adapted border glyphs below. [`zellij-server/src/ui/border_glyphs.rs`](https://github.com/zellij-org/zellij/blob/fc400dfef9ee79ca1412831f73d1f3c79699ea3f/zellij-server/src/ui/border_glyphs.rs) at commit `fc400dfef9ee79ca1412831f73d1f3c79699ea3f`. -Changes: imports and boundary/line-style types were localized for pVisor. +Only the rounded single-line glyphs used by pVisor are retained. The original copyright and MIT permission notice are in [LICENSE.md](LICENSE.md). diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs b/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs index fe9c9499..cf27644c 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/audit_ui.rs @@ -215,10 +215,6 @@ impl SessionPolicy { next.rules.push(rule); Some(next) } - - pub fn rule_labels(&self) -> Vec { - self.rules.iter().map(SessionRule::label).collect() - } } #[derive(Debug, Default, Clone, Copy, PartialEq, Eq, Serialize)] diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/border_glyphs.rs b/crates/persisting-pvisor/src/cli/tui/zellij/border_glyphs.rs index f74a8fe8..a2c9fb8c 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/border_glyphs.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/border_glyphs.rs @@ -1,380 +1,8 @@ -// Adapted from Zellij's zellij-server/src/ui/border_glyphs.rs at -// fc400dfef9ee79ca1412831f73d1f3c79699ea3f. See LICENSE.md in this directory. -use super::{LineStyle, boundary_type}; - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum Corner { - TopLeft, - TopRight, - BottomLeft, - BottomRight, -} - -pub fn horizontal(line_style: LineStyle) -> &'static str { - match line_style { - LineStyle::Single => "─", - LineStyle::Double => "═", - LineStyle::Heavy => "━", - LineStyle::Dashed => "┄", - LineStyle::HeavyDashed => "┅", - } -} - -pub fn vertical(line_style: LineStyle) -> &'static str { - match line_style { - LineStyle::Single => "│", - LineStyle::Double => "║", - LineStyle::Heavy => "┃", - LineStyle::Dashed => "┆", - LineStyle::HeavyDashed => "┇", - } -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Weight { - Light, - Double, - Heavy, -} - -fn weight(line_style: LineStyle) -> Weight { - match line_style { - LineStyle::Single | LineStyle::Dashed => Weight::Light, - LineStyle::Double => Weight::Double, - LineStyle::Heavy | LineStyle::HeavyDashed => Weight::Heavy, - } -} - -pub fn corner( - corner: Corner, - horizontal_style: LineStyle, - vertical_style: LineStyle, - rounded: bool, -) -> &'static str { - let horizontal_weight = weight(horizontal_style); - let mut vertical_weight = weight(vertical_style); - // there are no glyphs mixing a double arm with a heavy arm, in this case we render both arms - // in the weight of the horizontal arm - if (horizontal_weight == Weight::Double && vertical_weight == Weight::Heavy) - || (horizontal_weight == Weight::Heavy && vertical_weight == Weight::Double) - { - vertical_weight = horizontal_weight; - } - if rounded && horizontal_weight == Weight::Light && vertical_weight == Weight::Light { - return match corner { - Corner::TopLeft => boundary_type::TOP_LEFT_ROUND, - Corner::TopRight => boundary_type::TOP_RIGHT_ROUND, - Corner::BottomLeft => boundary_type::BOTTOM_LEFT_ROUND, - Corner::BottomRight => boundary_type::BOTTOM_RIGHT_ROUND, - }; - } - match (corner, horizontal_weight, vertical_weight) { - (Corner::TopLeft, Weight::Light, Weight::Light) => "┌", - (Corner::TopLeft, Weight::Double, Weight::Light) => "╒", - (Corner::TopLeft, Weight::Light, Weight::Double) => "╓", - (Corner::TopLeft, Weight::Double, Weight::Double) => "╔", - (Corner::TopLeft, Weight::Heavy, Weight::Light) => "┍", - (Corner::TopLeft, Weight::Light, Weight::Heavy) => "┎", - (Corner::TopLeft, Weight::Heavy, Weight::Heavy) => "┏", - - (Corner::TopRight, Weight::Light, Weight::Light) => "┐", - (Corner::TopRight, Weight::Double, Weight::Light) => "╕", - (Corner::TopRight, Weight::Light, Weight::Double) => "╖", - (Corner::TopRight, Weight::Double, Weight::Double) => "╗", - (Corner::TopRight, Weight::Heavy, Weight::Light) => "┑", - (Corner::TopRight, Weight::Light, Weight::Heavy) => "┒", - (Corner::TopRight, Weight::Heavy, Weight::Heavy) => "┓", - - (Corner::BottomLeft, Weight::Light, Weight::Light) => "└", - (Corner::BottomLeft, Weight::Double, Weight::Light) => "╘", - (Corner::BottomLeft, Weight::Light, Weight::Double) => "╙", - (Corner::BottomLeft, Weight::Double, Weight::Double) => "╚", - (Corner::BottomLeft, Weight::Heavy, Weight::Light) => "┕", - (Corner::BottomLeft, Weight::Light, Weight::Heavy) => "┖", - (Corner::BottomLeft, Weight::Heavy, Weight::Heavy) => "┗", - - (Corner::BottomRight, Weight::Light, Weight::Light) => "┘", - (Corner::BottomRight, Weight::Double, Weight::Light) => "╛", - (Corner::BottomRight, Weight::Light, Weight::Double) => "╜", - (Corner::BottomRight, Weight::Double, Weight::Double) => "╝", - (Corner::BottomRight, Weight::Heavy, Weight::Light) => "┙", - (Corner::BottomRight, Weight::Light, Weight::Heavy) => "┚", - (Corner::BottomRight, Weight::Heavy, Weight::Heavy) => "┛", - - (_, Weight::Double, Weight::Heavy) | (_, Weight::Heavy, Weight::Double) => { - unreachable!("double/heavy mixes are normalized above") - } - } -} - -pub fn title_separator_left(horizontal_style: LineStyle) -> &'static str { - match weight(horizontal_style) { - Weight::Light => boundary_type::VERTICAL_LEFT, - Weight::Double => "╡", - Weight::Heavy => "┥", - } -} - -pub fn title_separator_right(horizontal_style: LineStyle) -> &'static str { - match weight(horizontal_style) { - Weight::Light => boundary_type::VERTICAL_RIGHT, - Weight::Double => "╞", - Weight::Heavy => "┝", - } -} - -pub fn remap_light_glyph(glyph: &'static str, line_style: LineStyle) -> &'static str { - match line_style { - LineStyle::Single => glyph, - LineStyle::Dashed => match glyph { - boundary_type::HORIZONTAL => "┄", - boundary_type::VERTICAL => "┆", - _ => glyph, - }, - LineStyle::Double => match glyph { - boundary_type::HORIZONTAL => "═", - boundary_type::VERTICAL => "║", - boundary_type::TOP_LEFT => "╔", - boundary_type::TOP_LEFT_ROUND => "╔", - boundary_type::TOP_RIGHT => "╗", - boundary_type::TOP_RIGHT_ROUND => "╗", - boundary_type::BOTTOM_LEFT => "╚", - boundary_type::BOTTOM_LEFT_ROUND => "╚", - boundary_type::BOTTOM_RIGHT => "╝", - boundary_type::BOTTOM_RIGHT_ROUND => "╝", - boundary_type::VERTICAL_LEFT => "╣", - boundary_type::VERTICAL_RIGHT => "╠", - boundary_type::HORIZONTAL_DOWN => "╦", - boundary_type::HORIZONTAL_UP => "╩", - boundary_type::CROSS => "╬", - _ => glyph, - }, - LineStyle::Heavy | LineStyle::HeavyDashed => { - let heavy_dashed = line_style == LineStyle::HeavyDashed; - match glyph { - boundary_type::HORIZONTAL => { - if heavy_dashed { - "┅" - } else { - "━" - } - } - boundary_type::VERTICAL => { - if heavy_dashed { - "┇" - } else { - "┃" - } - } - boundary_type::TOP_LEFT => "┏", - boundary_type::TOP_LEFT_ROUND => "┏", - boundary_type::TOP_RIGHT => "┓", - boundary_type::TOP_RIGHT_ROUND => "┓", - boundary_type::BOTTOM_LEFT => "┗", - boundary_type::BOTTOM_LEFT_ROUND => "┗", - boundary_type::BOTTOM_RIGHT => "┛", - boundary_type::BOTTOM_RIGHT_ROUND => "┛", - boundary_type::VERTICAL_LEFT => "┫", - boundary_type::VERTICAL_RIGHT => "┣", - boundary_type::HORIZONTAL_DOWN => "┳", - boundary_type::HORIZONTAL_UP => "┻", - boundary_type::CROSS => "╋", - _ => glyph, - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - const ALL: [LineStyle; 5] = [ - LineStyle::Single, - LineStyle::Double, - LineStyle::Heavy, - LineStyle::Dashed, - LineStyle::HeavyDashed, - ]; - - const CORNERS: [Corner; 4] = [ - Corner::TopLeft, - Corner::TopRight, - Corner::BottomLeft, - Corner::BottomRight, - ]; - - #[test] - fn every_edge_style_has_its_own_glyph() { - let horizontals: Vec<&str> = ALL.iter().map(|s| horizontal(*s)).collect(); - let verticals: Vec<&str> = ALL.iter().map(|s| vertical(*s)).collect(); - assert_eq!(horizontals, vec!["─", "═", "━", "┄", "┅"]); - assert_eq!(verticals, vec!["│", "║", "┃", "┆", "┇"]); - } - - #[test] - fn every_corner_combination_resolves_to_a_single_width_glyph() { - for corner in CORNERS { - for horizontal_style in ALL { - for vertical_style in ALL { - for rounded in [false, true] { - let glyph = - super::corner(corner, horizontal_style, vertical_style, rounded); - assert_eq!( - glyph.chars().count(), - 1, - "{:?} {:?}/{:?} rounded={} produced {:?}", - corner, - horizontal_style, - vertical_style, - rounded, - glyph - ); - } - } - } - } - } - - #[test] - fn corners_only_round_when_both_arms_are_single_width() { - assert_eq!( - corner(Corner::TopLeft, LineStyle::Single, LineStyle::Single, true), - "╭" - ); - assert_eq!( - corner(Corner::TopLeft, LineStyle::Dashed, LineStyle::Dashed, true), - "╭" - ); - assert_eq!( - corner(Corner::TopLeft, LineStyle::Double, LineStyle::Single, true), - "╒" - ); - assert_eq!( - corner(Corner::TopLeft, LineStyle::Single, LineStyle::Heavy, true), - "┎" - ); - } - - #[test] - fn mixed_weight_corners_use_the_dedicated_glyphs() { - assert_eq!( - corner( - Corner::BottomRight, - LineStyle::Double, - LineStyle::Single, - false - ), - "╛" - ); - assert_eq!( - corner( - Corner::BottomRight, - LineStyle::Single, - LineStyle::Double, - false - ), - "╜" - ); - assert_eq!( - corner( - Corner::BottomRight, - LineStyle::Heavy, - LineStyle::Single, - false - ), - "┙" - ); - assert_eq!( - corner( - Corner::BottomRight, - LineStyle::Single, - LineStyle::Heavy, - false - ), - "┚" - ); - } - - #[test] - fn a_double_and_heavy_corner_uses_the_horizontal_arm_for_both() { - assert_eq!( - corner(Corner::TopLeft, LineStyle::Double, LineStyle::Heavy, false), - "╔" - ); - assert_eq!( - corner(Corner::TopLeft, LineStyle::Heavy, LineStyle::Double, false), - "┏" - ); - } - - #[test] - fn title_separators_follow_the_horizontal_weight() { - assert_eq!(title_separator_left(LineStyle::Single), "┤"); - assert_eq!(title_separator_right(LineStyle::Single), "├"); - assert_eq!(title_separator_left(LineStyle::Double), "╡"); - assert_eq!(title_separator_right(LineStyle::Double), "╞"); - assert_eq!(title_separator_left(LineStyle::HeavyDashed), "┥"); - assert_eq!(title_separator_right(LineStyle::HeavyDashed), "┝"); - } - - #[test] - fn remapping_covers_every_light_glyph_for_double_and_heavy() { - let light = [ - boundary_type::HORIZONTAL, - boundary_type::VERTICAL, - boundary_type::TOP_LEFT, - boundary_type::TOP_RIGHT, - boundary_type::BOTTOM_LEFT, - boundary_type::BOTTOM_RIGHT, - boundary_type::VERTICAL_LEFT, - boundary_type::VERTICAL_RIGHT, - boundary_type::HORIZONTAL_DOWN, - boundary_type::HORIZONTAL_UP, - boundary_type::CROSS, - ]; - for glyph in light { - for line_style in [LineStyle::Double, LineStyle::Heavy] { - assert_ne!( - remap_light_glyph(glyph, line_style), - glyph, - "{:?} left {} unmapped", - line_style, - glyph - ); - } - } - } - - #[test] - fn dashed_remapping_only_touches_the_straight_runs() { - assert_eq!( - remap_light_glyph(boundary_type::HORIZONTAL, LineStyle::Dashed), - "┄" - ); - assert_eq!( - remap_light_glyph(boundary_type::VERTICAL, LineStyle::Dashed), - "┆" - ); - assert_eq!( - remap_light_glyph(boundary_type::CROSS, LineStyle::Dashed), - boundary_type::CROSS - ); - assert_eq!( - remap_light_glyph(boundary_type::HORIZONTAL, LineStyle::HeavyDashed), - "┅" - ); - assert_eq!( - remap_light_glyph(boundary_type::CROSS, LineStyle::HeavyDashed), - "╋" - ); - } - - #[test] - fn a_single_style_remap_is_the_identity() { - assert_eq!( - remap_light_glyph(boundary_type::CROSS, LineStyle::Single), - boundary_type::CROSS - ); - } -} +// Rounded single-line subset of Zellij's zellij-server/src/ui/border_glyphs.rs +// at fc400dfef9ee79ca1412831f73d1f3c79699ea3f. See LICENSE.md in this directory. +pub const HORIZONTAL: &str = "─"; +pub const VERTICAL: &str = "│"; +pub const TOP_LEFT: &str = "╭"; +pub const TOP_RIGHT: &str = "╮"; +pub const BOTTOM_LEFT: &str = "╰"; +pub const BOTTOM_RIGHT: &str = "╯"; diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/input.rs b/crates/persisting-pvisor/src/cli/tui/zellij/input.rs index d7118b97..b2e44caf 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/input.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/input.rs @@ -208,15 +208,6 @@ impl UiState { self.mode == Mode::Agent } - pub fn hints(&self) -> String { - BINDINGS - .iter() - .filter(|binding| binding.mode == self.mode && !binding.hint.is_empty()) - .map(|binding| binding.hint) - .collect::>() - .join(" ") - } - /// Shortcut tiles are selected from the same bindings that handle input. pub fn ribbon_hints(&self) -> Vec<&'static str> { match self.mode { @@ -437,12 +428,12 @@ mod tests { #[test] fn displayed_hints_come_from_active_mode_bindings() { let mut state = UiState::default(); - assert_eq!(state.hints(), "Ctrl-] Menu"); + assert_eq!(state.ribbon_hints().join(" "), "Ctrl-] Menu"); state.input(PREFIX); - assert!(state.hints().contains("r Review")); - assert!(!state.hints().contains("Tab View")); + assert!(state.ribbon_hints().join(" ").contains("r Review")); + assert!(!state.ribbon_hints().join(" ").contains("Tab View")); state.input(b'r'); - assert!(state.hints().contains("Tab View")); + assert!(state.ribbon_hints().join(" ").contains("Tab View")); } #[test] diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/mod.rs b/crates/persisting-pvisor/src/cli/tui/zellij/mod.rs index 01fafe22..425d8114 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/mod.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/mod.rs @@ -3,35 +3,6 @@ //! Original source: https://github.com/zellij-org/zellij/tree/fc400dfef9ee79ca1412831f73d1f3c79699ea3f/zellij-server/src/ui //! Copyright (c) 2020 Zellij contributors. MIT license: see LICENSE.md. -#![allow(dead_code)] - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) enum LineStyle { - Single, - Double, - Heavy, - Dashed, - HeavyDashed, -} - -pub(super) mod boundary_type { - pub const TOP_RIGHT: &str = "┐"; - pub const TOP_RIGHT_ROUND: &str = "╮"; - pub const VERTICAL: &str = "│"; - pub const HORIZONTAL: &str = "─"; - pub const TOP_LEFT: &str = "┌"; - pub const TOP_LEFT_ROUND: &str = "╭"; - pub const BOTTOM_RIGHT: &str = "┘"; - pub const BOTTOM_RIGHT_ROUND: &str = "╯"; - pub const BOTTOM_LEFT: &str = "└"; - pub const BOTTOM_LEFT_ROUND: &str = "╰"; - pub const VERTICAL_LEFT: &str = "┤"; - pub const VERTICAL_RIGHT: &str = "├"; - pub const HORIZONTAL_DOWN: &str = "┬"; - pub const HORIZONTAL_UP: &str = "┴"; - pub const CROSS: &str = "┼"; -} - mod audit_ui; mod input; mod runtime; @@ -39,6 +10,4 @@ mod status_bar; mod view; pub(super) mod border_glyphs; -pub(crate) use runtime::{ - announce_stage, available, diagnostic, init_child_context, is_child, run, -}; +pub(crate) use runtime::{announce_stage, available, init_child_context, is_child, run}; diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs b/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs index 30e264ee..bc501915 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/runtime.rs @@ -24,7 +24,6 @@ const STAGE_FILE: &str = "PVISOR_UI_STAGE_FILE"; const LOG_FILE: &str = "PVISOR_UI_LOG_FILE"; const AUDIT_SOCKET: &str = "PVISOR_UI_AUDIT_SOCKET"; static CHILD_CONTEXT: OnceLock> = OnceLock::new(); -static LOG_CONTEXT: OnceLock> = OnceLock::new(); pub(crate) fn init_child_context() { let path = if std::env::var_os(CHILD_MARKER).is_some() { @@ -53,22 +52,12 @@ pub(crate) fn init_child_context() { persisting_control::audit::init(socket); } let _ = CHILD_CONTEXT.set(path); - crate::cache::progress::init_output( + crate::image::cache::progress::init_output( log_path .as_ref() .map(|path| path.with_extension("image.json")), ); - let _ = LOG_CONTEXT.set(log_path); -} - -pub(crate) fn diagnostic(args: std::fmt::Arguments<'_>) { - if let Some(Some(path)) = LOG_CONTEXT.get() - && let Ok(mut file) = OpenOptions::new().append(true).open(path) - && writeln!(file, "{args}").is_ok() - { - return; - } - eprintln!("{args}"); + crate::diagnostics::init(log_path); } pub(crate) fn announce_stage(stage: &Path) { @@ -132,7 +121,7 @@ impl Drop for ChildCleanup { #[derive(Default)] pub(super) struct Snapshot { - pub(super) image: Option, + pub(super) image: Option, pub(super) stage: Option, pub(super) record: Option, pub(super) filesystem: Option, @@ -706,8 +695,8 @@ mod tests { let directory = tempfile::tempdir().unwrap(); let log = directory.path().join("diagnostics.log"); File::create(&log).unwrap(); - LOG_CONTEXT.set(Some(log.clone())).unwrap(); - let downloads = crate::cache::progress::Downloads::new("example:latest"); + crate::diagnostics::init(Some(log.clone())); + let downloads = crate::image::cache::progress::Downloads::new("example:latest"); downloads.received(b"transfer-log-test/file\nname", 10); downloads.received(b"transfer-log-test/file\nname", 20); let mut snapshot = Snapshot::default(); diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs b/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs index a4b64b5b..86c1195c 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/status_bar.rs @@ -76,7 +76,7 @@ pub(super) fn format_bytes(bytes: u64) -> String { } fn image_summary( - image: &crate::cache::progress::ImageProgress, + image: &crate::image::cache::progress::ImageProgress, available: usize, ) -> Option { let files = image @@ -294,12 +294,12 @@ mod tests { #[test] fn image_metrics_fit_and_keep_unknown_totals_distinct_from_zero() { let mut snapshot = Snapshot { - image: Some(crate::cache::progress::ImageProgress { + image: Some(crate::image::cache::progress::ImageProgress { cached_files: 12, cached_bytes: 6 * 1024 * 1024, downloaded_files: 27, downloaded_bytes: 4 * 1024 * 1024, - totals: Some(crate::cache::ImageTotals { + totals: Some(crate::image::cache::ImageTotals { files: 1234, bytes: 80 * 1024 * 1024, }), diff --git a/crates/persisting-pvisor/src/cli/tui/zellij/view.rs b/crates/persisting-pvisor/src/cli/tui/zellij/view.rs index d18917cd..7b28bc5f 100644 --- a/crates/persisting-pvisor/src/cli/tui/zellij/view.rs +++ b/crates/persisting-pvisor/src/cli/tui/zellij/view.rs @@ -1,6 +1,6 @@ use super::input::{Panel, UiState}; use super::runtime::Snapshot; -use super::{LineStyle, border_glyphs, status_bar}; +use super::{border_glyphs, status_bar}; use anyhow::Result; use std::io::Write; use std::time::Instant; @@ -226,7 +226,7 @@ fn panel_lines(snapshot: &Snapshot, panel: Panel, started: Instant, width: usize "Transferred", image.downloaded_files, image.downloaded_bytes ), format!(" {:<13} {:>10} {:>16}", "Total", total_files, total_bytes), - " Cached: local block reads, including repeated reads".into(), + " Cached: bytes served from disk/memory cache, including repeats".into(), " Transferred: verified bytes received from the image server".into(), " Total: regular files and logical bytes in the whole image".into(), " Files are distinct paths; partial reads count; kernel cache hits excluded" @@ -394,30 +394,14 @@ fn floating_panel( ) { let (x, y, width, height) = layout.floating_rect(); let title = state.panel.title(); - let horizontal = border_glyphs::horizontal(LineStyle::Single); + let horizontal = border_glyphs::HORIZONTAL; move_to(buf, y, x); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::TopLeft, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::TOP_LEFT.as_bytes()); for _ in 0..width - 2 { buf.extend_from_slice(horizontal.as_bytes()); } - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::TopRight, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::TOP_RIGHT.as_bytes()); move_to(buf, y, x + 2); print_clipped(buf, &format!(" pVisor Review · {title} "), width - 4); @@ -446,12 +430,12 @@ fn floating_panel( let row = y + inner + 1; move_to(buf, row, x); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice(border_glyphs::vertical(LineStyle::Single).as_bytes()); + buf.extend_from_slice(border_glyphs::VERTICAL.as_bytes()); buf.extend_from_slice(b"\x1b[48;2;15;19;16m"); buf.extend_from_slice(" ".repeat((width - 2) as usize).as_bytes()); move_to(buf, row, x + width - 1); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice(border_glyphs::vertical(LineStyle::Single).as_bytes()); + buf.extend_from_slice(border_glyphs::VERTICAL.as_bytes()); move_to(buf, row, x + 2); buf.extend_from_slice(if inner == 0 { b"\x1b[48;2;15;19;16;38;2;167;230;54m" @@ -471,27 +455,11 @@ fn floating_panel( } move_to(buf, y + height - 1, x); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::BottomLeft, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::BOTTOM_LEFT.as_bytes()); for _ in 0..width - 2 { buf.extend_from_slice(horizontal.as_bytes()); } - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::BottomRight, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::BOTTOM_RIGHT.as_bytes()); if state.panel != Panel::Permissions && !lines.is_empty() { move_to(buf, y + height - 1, x + 2); print_clipped( @@ -759,28 +727,12 @@ pub(super) fn render( move_to(&mut buf, 2, 1); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::TopLeft, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); - let horizontal = border_glyphs::horizontal(LineStyle::Single); + buf.extend_from_slice(border_glyphs::TOP_LEFT.as_bytes()); + let horizontal = border_glyphs::HORIZONTAL; for _ in 0..layout.cols - 2 { buf.extend_from_slice(horizontal.as_bytes()); } - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::TopRight, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::TOP_RIGHT.as_bytes()); move_to(&mut buf, 2, 3); print_clipped( &mut buf, @@ -792,35 +744,19 @@ pub(super) fn render( let physical = row + 3; move_to(&mut buf, physical, 1); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice(border_glyphs::vertical(LineStyle::Single).as_bytes()); + buf.extend_from_slice(border_glyphs::VERTICAL.as_bytes()); draw_agent_row(&mut buf, row, layout.agent_cols, screen); move_to(&mut buf, physical, layout.cols); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice(border_glyphs::vertical(LineStyle::Single).as_bytes()); + buf.extend_from_slice(border_glyphs::VERTICAL.as_bytes()); } move_to(&mut buf, layout.rows - 1, 1); buf.extend_from_slice(ACTIVE.as_bytes()); - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::BottomLeft, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::BOTTOM_LEFT.as_bytes()); for _ in 0..layout.cols - 2 { buf.extend_from_slice(horizontal.as_bytes()); } - buf.extend_from_slice( - border_glyphs::corner( - border_glyphs::Corner::BottomRight, - LineStyle::Single, - LineStyle::Single, - true, - ) - .as_bytes(), - ); + buf.extend_from_slice(border_glyphs::BOTTOM_RIGHT.as_bytes()); if state.panel_open() { floating_panel(&mut buf, layout, state, snapshot, started); } @@ -904,7 +840,25 @@ mod tests { ); assert_eq!(after.agent_rows, rows - 4); assert_eq!(after.agent_cols, cols - 2); + let mut rendered = Vec::new(); + floating_panel( + &mut rendered, + after, + &mut state, + &Snapshot::default(), + Instant::now(), + ); + let mut parser = vt100::Parser::new(rows, cols, 0); + parser.process(&rendered); let (x, y, width, height) = after.floating_rect(); + for (row, col, glyph) in [ + (y - 1, x - 1, "╭"), + (y - 1, x + width - 2, "╮"), + (y + height - 2, x - 1, "╰"), + (y + height - 2, x + width - 2, "╯"), + ] { + assert_eq!(parser.screen().cell(row, col).unwrap().contents(), glyph); + } assert!(x >= 2 && x + width <= cols); assert!(y >= 3 && y + height < after.rows); if cols == 156 { diff --git a/crates/persisting-pvisor/src/control.rs b/crates/persisting-pvisor/src/control.rs deleted file mode 100644 index f5cc1dff..00000000 --- a/crates/persisting-pvisor/src/control.rs +++ /dev/null @@ -1,7 +0,0 @@ -//! Re-export the shared runtime control protocol. - -pub use persisting_control::{ - ControlController, ControlEffect, ControlMachine, ControlReason, ControlRequest, ControlState, - ControlTransition, NetworkGuard, NetworkHostRule, NetworkRule, PolicyControlController, - host_matches, is_public_egress_ip, normalize_host, parse_network_rule, -}; diff --git a/crates/persisting-pvisor/src/diagnostics.rs b/crates/persisting-pvisor/src/diagnostics.rs new file mode 100644 index 00000000..e7c4ce3e --- /dev/null +++ b/crates/persisting-pvisor/src/diagnostics.rs @@ -0,0 +1,23 @@ +//! Host diagnostics shared by runtime services and the CLI. +//! The frontend selects the destination; services do not depend on the TUI. + +use std::fs::OpenOptions; +use std::io::Write; +use std::path::PathBuf; +use std::sync::OnceLock; + +static LOG_CONTEXT: OnceLock> = OnceLock::new(); + +pub(crate) fn init(path: Option) { + let _ = LOG_CONTEXT.set(path); +} + +pub(crate) fn diagnostic(args: std::fmt::Arguments<'_>) { + if let Some(Some(path)) = LOG_CONTEXT.get() + && let Ok(mut file) = OpenOptions::new().append(true).open(path) + && writeln!(file, "{args}").is_ok() + { + return; + } + eprintln!("{args}"); +} diff --git a/crates/persisting-pvisor/src/artifact.rs b/crates/persisting-pvisor/src/executor/artifact.rs similarity index 100% rename from crates/persisting-pvisor/src/artifact.rs rename to crates/persisting-pvisor/src/executor/artifact.rs diff --git a/crates/persisting-pvisor/src/container.rs b/crates/persisting-pvisor/src/executor/container.rs similarity index 99% rename from crates/persisting-pvisor/src/container.rs rename to crates/persisting-pvisor/src/executor/container.rs index 34a26654..26602f72 100644 --- a/crates/persisting-pvisor/src/container.rs +++ b/crates/persisting-pvisor/src/executor/container.rs @@ -1,9 +1,9 @@ //! Native OCI runtime transport. pVisor materializes an OCI bundle and invokes //! runc/crun; no Docker or Podman daemon is required. -use crate::artifact::resolve_pvisor_binary; use crate::config::{ContainerMount, ContainerPlatform, ContainerSettings}; -use crate::delegated::{DelegatedRunFiles, RESULT_FILENAME, SPEC_FILENAME}; +use crate::executor::artifact::resolve_pvisor_binary; +use crate::executor::delegated::{DelegatedRunFiles, RESULT_FILENAME, SPEC_FILENAME}; use crate::executor::{AttemptContext, RunExecutor}; use async_trait::async_trait; use persisting_control::{ @@ -358,7 +358,7 @@ impl RunExecutor for ContainerExecutor { ); let image = executor.settings.image.clone(); let prepared = tokio::task::spawn_blocking(move || { - crate::oci::ImageStore::new(None)?.prepare(&image) + crate::image::oci::ImageStore::new(None)?.prepare(&image) }) .await??; executor.settings.rootfs = Some(prepared.rootfs); diff --git a/crates/persisting-pvisor/src/delegated.rs b/crates/persisting-pvisor/src/executor/delegated.rs similarity index 100% rename from crates/persisting-pvisor/src/delegated.rs rename to crates/persisting-pvisor/src/executor/delegated.rs diff --git a/crates/persisting-pvisor/src/executor.rs b/crates/persisting-pvisor/src/executor/mod.rs similarity index 94% rename from crates/persisting-pvisor/src/executor.rs rename to crates/persisting-pvisor/src/executor/mod.rs index 32c70a3f..1253f202 100644 --- a/crates/persisting-pvisor/src/executor.rs +++ b/crates/persisting-pvisor/src/executor/mod.rs @@ -1,4 +1,13 @@ -use crate::event::RunEventPublisher; +//! Executor contract, concrete backends, and their isolation helpers. + +pub(crate) mod artifact; +pub(crate) mod container; +pub(crate) mod delegated; +pub(crate) mod process; +pub mod sandbox; +pub(crate) mod vm; + +use crate::runtime::event::RunEventPublisher; use async_trait::async_trait; use persisting_control::{ AttemptId, ExecutorDescriptor, RunInvocation, RunResult, RunSpec, RunState, RunStatus, diff --git a/crates/persisting-pvisor/src/process.rs b/crates/persisting-pvisor/src/executor/process.rs similarity index 98% rename from crates/persisting-pvisor/src/process.rs rename to crates/persisting-pvisor/src/executor/process.rs index de1dfacf..cda93658 100644 --- a/crates/persisting-pvisor/src/process.rs +++ b/crates/persisting-pvisor/src/executor/process.rs @@ -1,14 +1,14 @@ -use crate::executor::{AttemptContext, RunExecutor}; #[cfg(any(target_os = "linux", target_os = "macos"))] -use crate::sandbox::{INTERNAL_SANDBOX_ARG, NetworkIsolation}; +use crate::executor::sandbox::{INTERNAL_SANDBOX_ARG, NetworkIsolation}; #[cfg(target_os = "macos")] -use crate::sandbox::{ +use crate::executor::sandbox::{ MACOS_SANDBOX_EXEC, SEATBELT_ATTESTATION, SeatbeltPlan, seatbelt_profile, seatbelt_profile_with_reads, }; #[cfg(target_os = "linux")] -use crate::sandbox::{ROOTLESS_ATTESTATION, SandboxPlan, landlock_runtime_available}; -use crate::sandbox::{SANDBOX_ARG0_ENV, SANDBOX_PLAN_ENV, SANDBOX_SETUP_FAILED_WARNING}; +use crate::executor::sandbox::{ROOTLESS_ATTESTATION, SandboxPlan, landlock_runtime_available}; +use crate::executor::sandbox::{SANDBOX_ARG0_ENV, SANDBOX_PLAN_ENV, SANDBOX_SETUP_FAILED_WARNING}; +use crate::executor::{AttemptContext, RunExecutor}; use async_trait::async_trait; use persisting_control::{ CapabilityDimension, CapabilityEnforcementEvidence, ExecutorDescriptor, ExecutorKind, @@ -407,12 +407,12 @@ fn stdio(mode: StdioMode) -> Stdio { #[cfg(any(target_os = "linux", target_os = "macos"))] fn network_isolation(spec: &RunSpec) -> std::io::Result { - if crate::sandbox::sandbox_required(spec) { + if crate::executor::sandbox::sandbox_required(spec) { #[cfg(target_os = "macos")] { let proxy = spec .metadata - .get(crate::sandbox::SANDBOX_PROXY_KEY) + .get(crate::executor::sandbox::SANDBOX_PROXY_KEY) .and_then(serde_json::Value::as_str) .map(str::parse::) .transpose() @@ -482,7 +482,7 @@ impl ProcessExecutor { /// Build a Linux rootless executor using `launcher` for the trusted /// namespace/Landlock setup stage. /// - /// The launcher must dispatch [`crate::sandbox::run_internal_if_requested`] + /// The launcher must dispatch [`crate::executor::sandbox::run_internal_if_requested`] /// before starting threads or an async runtime. The `pvisor` binary is the /// canonical launcher and uses this path automatically for default host Runs. #[cfg(target_os = "linux")] @@ -566,7 +566,7 @@ impl ProcessExecutor { // an OverlayFS merged root. The executable belongs to the host-process // executor and need not exist inside the projected lower filesystem. let program = resolve_host_program(&invocation.program); - if crate::sandbox::sandbox_required(spec) && !self.is_sandboxed() { + if crate::executor::sandbox::sandbox_required(spec) && !self.is_sandboxed() { return Err(std::io::Error::other( "required sandbox cannot use an unsandboxed process executor", )); @@ -622,7 +622,7 @@ impl ProcessExecutor { // A Run-owned temporary directory avoids granting the Agent the // shared /tmp or per-user Darwin temporary hierarchy. command.env("TMPDIR", scratch); - if crate::sandbox::sandbox_required(spec) { + if crate::executor::sandbox::sandbox_required(spec) { command.env("HOME", scratch); } } @@ -784,7 +784,7 @@ fn platform_launcher_command( .map(PathBuf::from) .unwrap_or(std::env::current_dir()?); let cwd = cwd.canonicalize()?; - let restrict_reads = crate::sandbox::sandbox_required(spec); + let restrict_reads = crate::executor::sandbox::sandbox_required(spec); let mut readable_paths = vec![program.clone(), launcher.canonicalize()?]; let mut writable_paths = vec![ cwd.clone(), @@ -981,7 +981,7 @@ fn rootless_plan( push_existing(&mut read_only, Path::new("/dev/pts")); let hidden_paths = spec .metadata - .get(crate::sandbox::SANDBOX_HIDDEN_PATHS_KEY) + .get(crate::executor::sandbox::SANDBOX_HIDDEN_PATHS_KEY) .and_then(serde_json::Value::as_array) .into_iter() .flatten() @@ -993,7 +993,7 @@ fn rootless_plan( // Run mounts private copy-on-write views of them at the same paths, so // programs launched later from a shell get the same protection as the // initial executable. - let safe = crate::sandbox::sandbox_required(spec); + let safe = crate::executor::sandbox::sandbox_required(spec); let mut staged_roots = Vec::new(); for path in projected_state_roots(invocation) { if safe { @@ -1088,7 +1088,7 @@ fn rootless_plan( let project_render_nodes = graphical_display && !spec .metadata - .get(crate::sandbox::SANDBOX_NO_GPU_KEY) + .get(crate::executor::sandbox::SANDBOX_NO_GPU_KEY) .and_then(serde_json::Value::as_bool) .unwrap_or(false); if project_render_nodes && let Ok(devices) = std::fs::read_dir("/dev/dri") { @@ -1822,8 +1822,10 @@ mod tests { let executor = ProcessExecutor::seatbelt_with_launcher(std::env::current_exe().unwrap()).unwrap(); for (safe, staged) in [(false, false), (false, true), (true, false), (true, true)] { - spec.metadata - .insert(crate::sandbox::REQUIRED_SANDBOX_KEY.into(), safe.into()); + spec.metadata.insert( + crate::executor::sandbox::REQUIRED_SANDBOX_KEY.into(), + safe.into(), + ); spec.metadata.insert( "pvisor.runtime.implant".into(), serde_json::json!({ @@ -1908,8 +1910,10 @@ mod tests { assert!(normal.read_write.contains(&home)); assert!(normal.staged_roots.is_empty()); - spec.metadata - .insert(crate::sandbox::REQUIRED_SANDBOX_KEY.into(), true.into()); + spec.metadata.insert( + crate::executor::sandbox::REQUIRED_SANDBOX_KEY.into(), + true.into(), + ); spec.metadata.insert( "pvisor.workspace".into(), workspace.display().to_string().into(), diff --git a/crates/persisting-pvisor/src/sandbox.rs b/crates/persisting-pvisor/src/executor/sandbox.rs similarity index 100% rename from crates/persisting-pvisor/src/sandbox.rs rename to crates/persisting-pvisor/src/executor/sandbox.rs diff --git a/crates/persisting-pvisor/src/firmware.rs b/crates/persisting-pvisor/src/executor/vm/firmware.rs similarity index 98% rename from crates/persisting-pvisor/src/firmware.rs rename to crates/persisting-pvisor/src/executor/vm/firmware.rs index e6cc7e74..ec12628c 100644 --- a/crates/persisting-pvisor/src/firmware.rs +++ b/crates/persisting-pvisor/src/executor/vm/firmware.rs @@ -48,7 +48,7 @@ impl FirmwareStore { pub fn prepare(&self) -> anyhow::Result { let platform = platform_name()?; let directory = self.root.join(VERSION).join(platform); - let firmware = directory.join(crate::vm::firmware_name()); + let firmware = directory.join(crate::executor::vm::firmware_name()); if firmware.is_file() { return Ok(directory); } @@ -81,7 +81,7 @@ impl FirmwareStore { .tempdir_in(directory)?; let payload = temporary.path().join("kernel.c"); extract_member(&archive, asset.archive_member, &payload)?; - let built = temporary.path().join(crate::vm::firmware_name()); + let built = temporary.path().join(crate::executor::vm::firmware_name()); build_platform_firmware(&payload, &built)?; let mut permissions = fs::metadata(&built)?.permissions(); #[cfg(unix)] diff --git a/crates/persisting-pvisor/src/vm/mod.rs b/crates/persisting-pvisor/src/executor/vm/mod.rs similarity index 77% rename from crates/persisting-pvisor/src/vm/mod.rs rename to crates/persisting-pvisor/src/executor/vm/mod.rs index ef883344..992def74 100644 --- a/crates/persisting-pvisor/src/vm/mod.rs +++ b/crates/persisting-pvisor/src/executor/vm/mod.rs @@ -13,3 +13,9 @@ pub(crate) use supported::{bundled_firmware_dir, firmware_name}; #[cfg(all(target_os = "macos", target_arch = "x86_64"))] pub use unsupported::{VmExecutor, run_internal_if_requested}; + +#[cfg(not(any( + all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), + all(target_os = "macos", target_arch = "x86_64") +)))] +pub(crate) mod firmware; diff --git a/crates/persisting-pvisor/src/vm/supported.rs b/crates/persisting-pvisor/src/executor/vm/supported.rs similarity index 99% rename from crates/persisting-pvisor/src/vm/supported.rs rename to crates/persisting-pvisor/src/executor/vm/supported.rs index aa51dcd8..f692395b 100644 --- a/crates/persisting-pvisor/src/vm/supported.rs +++ b/crates/persisting-pvisor/src/executor/vm/supported.rs @@ -372,7 +372,7 @@ impl RunExecutor for VmExecutor { BTreeMap::new() }; for key in [ - crate::cache::SERVER_ENV, + crate::image::cache::SERVER_ENV, "PERSISTING_PVISOR_CACHE_TOKEN", crate::AGENTCTL_ENDPOINT_ENV, crate::AGENTCTL_TOKEN_ENV, @@ -814,7 +814,11 @@ fn run_runner(spec: RunnerSpec) -> anyhow::Result<()> { read_write.extend(workspace.work.iter().cloned()); read_write.extend(workspace.preimages.iter().cloned()); } - crate::sandbox::restrict_krun_runner(read_only, read_write, spec.library_dir.clone())?; + crate::executor::sandbox::restrict_krun_runner( + read_only, + read_write, + spec.library_dir.clone(), + )?; } run_linked_krun(spec) } diff --git a/crates/persisting-pvisor/src/vm/unsupported.rs b/crates/persisting-pvisor/src/executor/vm/unsupported.rs similarity index 100% rename from crates/persisting-pvisor/src/vm/unsupported.rs rename to crates/persisting-pvisor/src/executor/vm/unsupported.rs diff --git a/crates/persisting-pvisor/src/image/cache/cli.rs b/crates/persisting-pvisor/src/image/cache/cli.rs new file mode 100644 index 00000000..15cae75a --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/cli.rs @@ -0,0 +1,103 @@ +//! `pvisor cache` command definitions and dispatch. +use super::server::serve; +use super::transport::{TOKEN_ENV, endpoint_from_env}; +use super::{CacheClient, MAX_READ, Request, architecture}; +use crate::image::oci::ImageStore; +use clap::{Args, Subcommand}; +use std::io::Write; +use std::os::unix::ffi::OsStrExt; +use std::path::PathBuf; + +#[derive(Debug, Args)] +pub struct CacheArgs { + #[command(subcommand)] + command: CacheCommand, +} + +#[derive(Debug, Subcommand)] +enum CacheCommand { + /// Serve cached OCI files (foreground; Unix socket by default). + Serve { + /// unix:///absolute/path or tcp://127.0.0.1:PORT. Defaults to CACHE_SERVER. + #[arg(long)] + listen: Option, + /// OCI cache to serve and populate. + #[arg(long, env = "PERSISTING_PVISOR_IMAGE_STORE")] + image_store: Option, + }, + /// Resolve and prepare an image on the server; print its immutable digest. + Prepare { + image: String, + /// Recheck the registry even when a fresh prepared-image record exists. + #[arg(long)] + refresh: bool, + }, + /// List one directory page. Paths are relative to the image root. + List { + digest: String, + path: Option, + #[arg(long, default_value_t = 0)] + offset: usize, + }, + /// Show file attributes without following symlinks. + Stat { digest: String, path: PathBuf }, + /// Stream one regular file to stdout. Does not follow symlinks. + Read { digest: String, path: PathBuf }, +} + +pub fn run(args: CacheArgs) -> anyhow::Result<()> { + if let CacheCommand::Serve { + listen, + image_store, + } = args.command + { + return serve( + listen.map_or_else(endpoint_from_env, Ok)?, + ImageStore::new(image_store)?, + std::env::var(TOKEN_ENV).ok(), + ); + } + let client = CacheClient::from_env()?; + let request = match args.command { + CacheCommand::Prepare { image, refresh } => Request::Prepare { + image, + architecture: architecture().into(), + refresh, + }, + CacheCommand::List { + digest, + path, + offset, + } => Request::List { + digest, + path: path.unwrap_or_default().as_os_str().as_bytes().to_vec(), + offset, + }, + CacheCommand::Stat { digest, path } => Request::Stat { + digest, + path: path.as_os_str().as_bytes().to_vec(), + }, + CacheCommand::Read { digest, path } => { + let mut offset = 0; + let mut stdout = std::io::stdout().lock(); + loop { + let (_, body) = client.request(Request::Read { + digest: digest.clone(), + path: path.as_os_str().as_bytes().to_vec(), + offset, + length: MAX_READ, + })?; + stdout.write_all(&body)?; + offset += body.len() as u64; + if body.len() < MAX_READ as usize { + break; + } + } + return Ok(()); + } + CacheCommand::Serve { .. } => unreachable!(), + }; + let (response, _) = client.request(request)?; + println!("{}", serde_json::to_string_pretty(&response)?); + Ok(()) +} diff --git a/crates/persisting-pvisor/src/image/cache/client.rs b/crates/persisting-pvisor/src/image/cache/client.rs new file mode 100644 index 00000000..47571e96 --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/client.rs @@ -0,0 +1,140 @@ +//! Blocking cache client and service discovery. +use super::protocol::{Envelope, hash, read_frame, write_frame}; +use super::transport::{ + Endpoint, Stream, TIMEOUT, TOKEN_ENV, default_endpoint, endpoint, endpoint_from_env, +}; +use super::{MAX_READ, Request, Response, SERVER_ENV}; +use anyhow::{Context, bail, ensure}; +use std::io::Read; +use std::net::TcpStream; +use std::os::unix::net::UnixStream; +use std::time::Duration; + +#[derive(Debug, thiserror::Error)] +#[error("cache connection failed: {0}")] +struct CacheConnectError(#[source] std::io::Error); + +/// Blocking client; call from the host side, outside filesystem operation locks. +pub struct CacheClient { + pub(super) endpoint: String, + token: Option, +} +impl CacheClient { + pub fn from_env() -> anyhow::Result { + Self::new(endpoint_from_env()?, std::env::var(TOKEN_ENV).ok()) + } + pub fn new(address: String, token: Option) -> anyhow::Result { + if matches!(endpoint(&address)?, Endpoint::Tcp(_)) { + ensure!( + token.as_ref().is_some_and(|s| !s.is_empty()), + "TCP requires {TOKEN_ENV}" + ); + } + Ok(Self { + endpoint: address, + token, + }) + } + /// Discover the default socket, or require an explicitly configured service. + pub(crate) fn discover() -> anyhow::Result> { + let explicit = match std::env::var(SERVER_ENV) { + Ok(value) => Some(value), + Err(std::env::VarError::NotPresent) => None, + Err(error) => return Err(error.into()), + }; + if explicit.as_deref() == Some("off") { + return Ok(None); + } + let address = explicit.clone().map_or_else(default_endpoint, Ok)?; + Self::probe(address, std::env::var(TOKEN_ENV).ok(), explicit.is_some()) + } + + pub(super) fn probe( + address: String, + token: Option, + explicit: bool, + ) -> anyhow::Result> { + let client = Self::new(address, token)?; + match client.request_timeout(Request::Ping, Duration::from_secs(2)) { + Ok((Response::Ready, _)) => Ok(Some(client)), + Ok(_) => bail!("cache server returned an incompatible handshake"), + Err(error) => { + let absent = error.downcast_ref::().is_some_and(|e| { + matches!( + e.0.kind(), + std::io::ErrorKind::NotFound | std::io::ErrorKind::ConnectionRefused + ) + }); + if !explicit && absent { + Ok(None) + } else { + Err(error.context("probe shared image cache")) + } + } + } + } + + /// The returned bytes are present only for `Read`, and are SHA-256 checked. + pub fn request(&self, request: Request) -> anyhow::Result<(Response, Vec)> { + self.request_timeout(request, TIMEOUT) + } + + fn request_timeout( + &self, + request: Request, + timeout: Duration, + ) -> anyhow::Result<(Response, Vec)> { + let expected = match &request { + Request::Read { length, .. } => Some(*length), + _ => None, + }; + let mut stream = match endpoint(&self.endpoint)? { + Endpoint::Unix(path) => Stream::Unix( + UnixStream::connect(path) + .map_err(CacheConnectError) + .with_context(|| { + format!( + "connect cache {}; start `pvisor cache serve`", + self.endpoint + ) + })?, + ), + Endpoint::Tcp(address) => Stream::Tcp( + TcpStream::connect_timeout(&address, Duration::from_secs(10)) + .map_err(CacheConnectError)?, + ), + }; + stream.timeouts(timeout)?; + write_frame( + &mut stream, + &Envelope { + version: 1, + token: self.token.clone(), + request, + }, + )?; + let response: Response = read_frame(&mut stream)?; + let mut body = Vec::new(); + match &response { + Response::Error { code, message } => { + let kind = match code.as_str() { + "not_found" => std::io::ErrorKind::NotFound, + "permission_denied" => std::io::ErrorKind::PermissionDenied, + _ => std::io::ErrorKind::Other, + }; + return Err(std::io::Error::new(kind, format!("cache {code}: {message}")).into()); + } + Response::Data { length, sha256 } => { + ensure!( + expected.is_some_and(|limit| *length <= limit) && *length <= MAX_READ, + "invalid cache data length" + ); + body.resize(*length as usize, 0); + stream.read_exact(&mut body)?; + ensure!(hash(&body) == *sha256, "cache data digest mismatch"); + } + _ => ensure!(expected.is_none(), "expected cache data response"), + } + Ok((response, body)) + } +} diff --git a/crates/persisting-pvisor/src/cache/lazy.rs b/crates/persisting-pvisor/src/image/cache/lazy.rs similarity index 62% rename from crates/persisting-pvisor/src/cache/lazy.rs rename to crates/persisting-pvisor/src/image/cache/lazy.rs index 6c8186aa..735f02c9 100644 --- a/crates/persisting-pvisor/src/cache/lazy.rs +++ b/crates/persisting-pvisor/src/image/cache/lazy.rs @@ -1,6 +1,6 @@ //! A read-only, demand-filled FUSE lower for the existing VM/OCI overlays. use super::{CacheClient, MAX_READ, Request as CacheRequest, Response, architecture, hash}; -use crate::oci::{ImageStore, PreparedImage}; +use crate::image::oci::{ImageStore, PreparedImage}; use anyhow::{Context, ensure}; use fs2::FileExt; use fuser::{ @@ -8,7 +8,8 @@ use fuser::{ ReplyDirectory, ReplyEntry, ReplyOpen, ReplyStatfs, Request, Session, }; use sha2::{Digest, Sha256}; -use std::collections::HashMap; +use std::cell::RefCell; +use std::collections::{HashMap, VecDeque}; use std::ffi::{OsStr, OsString}; use std::fs::{self, OpenOptions}; use std::io::Write; @@ -17,6 +18,7 @@ use std::os::unix::fs::OpenOptionsExt; #[cfg(target_os = "linux")] use std::os::unix::fs::PermissionsExt; use std::path::{Path, PathBuf}; +use std::sync::Arc; use std::time::{Duration, UNIX_EPOCH}; const TTL: Duration = Duration::from_secs(3600); @@ -30,7 +32,7 @@ impl Drop for LazyMount { if let Some(session) = self.session.take() && let Err(error) = session.unmount() { - crate::cli::diagnostic(format_args!( + crate::diagnostics::diagnostic(format_args!( "unmount lazy image {}: {error}", self.path.display() )); @@ -55,7 +57,7 @@ pub(crate) fn prepare_image( )); }; let downloads = super::progress::Downloads::new(image); - crate::cli::diagnostic(format_args!( + crate::diagnostics::diagnostic(format_args!( "pVisor image: lazy loading from {}", client.endpoint )); @@ -64,6 +66,7 @@ pub(crate) fn prepare_image( client.request(CacheRequest::Prepare { image: image.into(), architecture: architecture().into(), + refresh: false, }) })?; let Response::Prepared { @@ -82,7 +85,7 @@ pub(crate) fn prepare_image( platform == architecture(), "cache returned the wrong image architecture" ); - crate::oci::digest_hex(&digest)?; + crate::image::oci::digest_hex(&digest)?; let cache = dirs::cache_dir() .context("cannot find user cache directory")? .join("persisting/pvisor/blocks") @@ -91,7 +94,7 @@ pub(crate) fn prepare_image( fs::create_dir_all(&cache)?; downloads.totals(totals); if let Some(totals) = totals { - crate::cli::diagnostic(format_args!( + crate::diagnostics::diagnostic(format_args!( "pVisor image: prepared {digest}; {} files, {:.1} MiB (contents fetched on demand)", totals.files, totals.bytes as f64 / (1024.0 * 1024.0) @@ -176,10 +179,47 @@ struct Node { path: Vec, attr: FileAttr, target: Option>, + cache: PathBuf, +} + +// File identity is the outer key; 1 MiB block indices only have meaning within it. +#[derive(Default)] +struct HotBlocks { + files: HashMap>>, + order: VecDeque<(u64, u64)>, + bytes: usize, +} +impl HotBlocks { + const MAX_BYTES: usize = 64 * 1024 * 1024; + const MAX_ENTRIES: usize = 4096; + + fn get(&self, file: u64, block: u64) -> Option> { + self.files.get(&file)?.get(&block).cloned() + } + + fn insert(&mut self, file: u64, block: u64, bytes: Arc<[u8]>) { + if self.get(file, block).is_some() { + return; + } + // ponytail: bounded FIFO avoids per-read LRU maintenance; use LRU if + // eviction of frequently reused blocks becomes a measured bottleneck. + while self.bytes + bytes.len() > Self::MAX_BYTES || self.order.len() >= Self::MAX_ENTRIES { + let (file, block) = self.order.pop_front().unwrap(); + let blocks = self.files.get_mut(&file).unwrap(); + self.bytes -= blocks.remove(&block).unwrap().len(); + if blocks.is_empty() { + self.files.remove(&file); + } + } + self.bytes += bytes.len(); + self.files.entry(file).or_default().insert(block, bytes); + self.order.push_back((file, block)); + } } struct RemoteFs { downloads: super::progress::Downloads, + hot: RefCell, client: CacheClient, digest: String, cache: PathBuf, @@ -199,6 +239,7 @@ impl RemoteFs { ) -> anyhow::Result { let mut fs = Self { downloads: super::progress::Downloads::default(), + hot: RefCell::default(), client, digest, cache, @@ -268,6 +309,13 @@ impl RemoteFs { digest: self.digest.clone(), path: path.clone(), })?; + self.insert_node(path, response) + } + + fn insert_node(&mut self, path: Vec, response: Response) -> anyhow::Result { + if let Some(ino) = self.paths.get(&path) { + return self.node(*ino).cloned(); + } let Response::Metadata { kind, size, @@ -309,6 +357,7 @@ impl RemoteFs { } .context("remote timestamp overflow")?; let node = Node { + cache: self.cache.join(&hash(&path)[7..]), path: path.clone(), target, attr: FileAttr { @@ -364,6 +413,9 @@ impl RemoteFs { path.push(b'/'); } path.extend_from_slice(name); + if self.directories.contains_key(&parent.attr.ino) && !self.paths.contains_key(&path) { + return Err(std::io::Error::from(std::io::ErrorKind::NotFound).into()); + } self.lookup_path(path) } @@ -386,11 +438,42 @@ impl RemoteFs { path: node.path.clone(), offset, })?; - let Response::Entries { names, next_offset } = response else { + let Response::Entries { + names, + metadata, + next_offset, + } = response + else { anyhow::bail!("expected cache directory response"); }; + if let Some(attributes) = &metadata { + ensure!( + attributes.len() == names.len(), + "directory metadata count mismatch" + ); + } + let mut attributes = metadata.map(Vec::into_iter); for name in names { - let child = self.child(ino, OsStr::from_bytes(&name))?; + // Validate untrusted directory names before using them as paths. + ensure!( + !name.is_empty() + && name != b"." + && name != b".." + && !name.contains(&b'/') + && !name.contains(&0), + "invalid remote filename" + ); + let child = if let Some(attributes) = &mut attributes { + let mut path = node.path.clone(); + if !path.is_empty() { + path.push(b'/'); + } + path.extend_from_slice(&name); + self.insert_node(path, attributes.next().unwrap())? + } else { + // Older servers and persisted v1 pages contain names only. + self.child(ino, OsStr::from_bytes(&name))? + }; entries.push((child.attr.ino, child.attr.kind, OsString::from_vec(name))); } match next_offset { @@ -406,9 +489,12 @@ impl RemoteFs { Ok(self.directories.get(&ino).unwrap()) } - fn block(&self, node: &Node, index: u64) -> anyhow::Result> { - let directory = self.cache.join(&hash(&node.path)[7..]); - fs::create_dir_all(&directory)?; + fn block(&self, node: &Node, index: u64) -> anyhow::Result<(Arc<[u8]>, bool)> { + if let Some(bytes) = self.hot.borrow().get(node.attr.ino, index) { + return Ok((bytes, true)); + } + let directory = &node.cache; + fs::create_dir_all(directory)?; let path = directory.join(index.to_string()); let lock = OpenOptions::new() .read(true) @@ -428,8 +514,11 @@ impl RemoteFs { if bytes.len() == length + 32 && Sha256::digest(&bytes[32..]).as_slice() == &bytes[..32] => { - self.downloads.cached(&node.path, length); - return Ok(bytes[32..].to_vec()); + let body: Arc<[u8]> = bytes[32..].into(); + self.hot + .borrow_mut() + .insert(node.attr.ino, index, body.clone()); + return Ok((body, true)); } Ok(_) => { fs::remove_file(&path)?; @@ -448,11 +537,15 @@ impl RemoteFs { "remote file returned a short block before EOF" ); self.downloads.received(&node.path, body.len()); - let mut temporary = tempfile::NamedTempFile::new_in(&directory)?; + let mut temporary = tempfile::NamedTempFile::new_in(directory)?; temporary.write_all(&Sha256::digest(&body))?; temporary.write_all(&body)?; temporary.persist(&path)?; - Ok(body) + let body: Arc<[u8]> = body.into(); + self.hot + .borrow_mut() + .insert(node.attr.ino, index, body.clone()); + Ok((body, false)) } fn read_range(&self, ino: u64, offset: u64, size: u32) -> anyhow::Result> { @@ -462,13 +555,16 @@ impl RemoteFs { "read requires a regular file" ); let end = offset.saturating_add(size as u64).min(node.attr.size); - let mut result = Vec::new(); + let mut result = Vec::with_capacity(end.saturating_sub(offset) as usize); let mut cursor = offset; while cursor < end { - let bytes = self.block(node, cursor / MAX_READ as u64)?; + let (bytes, cached) = self.block(node, cursor / MAX_READ as u64)?; let begin = (cursor % MAX_READ as u64) as usize; let count = (end - cursor).min((bytes.len() - begin) as u64) as usize; result.extend_from_slice(&bytes[begin..begin + count]); + if cached { + self.downloads.cached(&node.path, count); + } cursor += count as u64; } Ok(result) @@ -488,7 +584,7 @@ fn errno(error: anyhow::Error) -> i32 { }) .unwrap_or(libc::EIO); if code != libc::ENOENT { - crate::cli::diagnostic(format_args!("lazy image I/O: {error:#}")); + crate::diagnostics::diagnostic(format_args!("lazy image I/O: {error:#}")); } code } @@ -585,286 +681,4 @@ impl Filesystem for RemoteFs { } #[cfg(test)] -mod tests { - use super::*; - use crate::cache::{Envelope, handle, read_frame, write_frame}; - use std::os::unix::net::UnixListener; - use std::sync::{ - Arc, - atomic::{AtomicBool, AtomicUsize, Ordering}, - }; - - struct Server { - stop: Arc, - worker: Option>, - reads: Arc, - } - impl Drop for Server { - fn drop(&mut self) { - self.stop.store(true, Ordering::Relaxed); - self.worker.take().unwrap().join().unwrap(); - } - } - fn fixture() -> (tempfile::TempDir, Server, CacheClient, String) { - let temp = tempfile::tempdir().unwrap(); - let store = ImageStore::new(Some(temp.path().join("store"))).unwrap(); - let digest = format!("sha256:{}", "b".repeat(64)); - let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); - fs::create_dir(&root).unwrap(); - fs::write(root.join("large"), vec![42; 3 * MAX_READ as usize]).unwrap(); - std::os::unix::fs::symlink("large", root.join("alias")).unwrap(); - let socket = temp.path().join("s"); - let listener = UnixListener::bind(&socket).unwrap(); - listener.set_nonblocking(true).unwrap(); - let stop = Arc::new(AtomicBool::new(false)); - let reads = Arc::new(AtomicUsize::new(0)); - let worker_stop = stop.clone(); - let worker_reads = reads.clone(); - let worker = std::thread::spawn(move || { - while !worker_stop.load(Ordering::Relaxed) { - let (mut socket, _) = match listener.accept() { - Ok(s) => s, - Err(e) if e.kind() == std::io::ErrorKind::WouldBlock => { - std::thread::sleep(Duration::from_millis(1)); - continue; - } - Err(e) => panic!("{e}"), - }; - socket.set_nonblocking(false).unwrap(); - let envelope: Envelope = read_frame(&mut socket).unwrap(); - if matches!(&envelope.request, CacheRequest::Read { .. }) { - worker_reads.fetch_add(1, Ordering::Relaxed); - } - let (response, bytes) = handle(&store, envelope.request).unwrap_or_else(|e| { - let code = if e - .downcast_ref::() - .is_some_and(|e| e.kind() == std::io::ErrorKind::NotFound) - { - "not_found" - } else { - "request_failed" - }; - ( - Response::Error { - code: code.into(), - message: e.to_string(), - }, - Vec::new(), - ) - }); - write_frame(&mut socket, &response).unwrap(); - socket.write_all(&bytes).unwrap(); - } - }); - let client = CacheClient::new(format!("unix://{}", socket.display()), None).unwrap(); - ( - temp, - Server { - stop, - reads, - worker: Some(worker), - }, - client, - digest, - ) - } - - #[test] - fn persistent_metadata_survives_remount_and_rejects_corruption() { - let (temp, server, client, digest) = fixture(); - let blocks = temp.path().join("blocks"); - let metadata = temp.path().join("metadata"); - let endpoint = client.endpoint.clone(); - let mut cold = RemoteFs::new( - client, - digest.clone(), - blocks.clone(), - Some(metadata.clone()), - ) - .unwrap(); - let expected = cold.child(1, OsStr::new("large")).unwrap().attr.size; - let count = cold.entries(1).unwrap().len(); - assert!(cold.child(1, OsStr::new("missing")).is_err()); - drop(server); - let client = || CacheClient::new(endpoint.clone(), None).unwrap(); - let mut warm = RemoteFs::new( - client(), - digest.clone(), - blocks.clone(), - Some(metadata.clone()), - ) - .unwrap(); - assert_eq!( - warm.child(1, OsStr::new("large")).unwrap().attr.size, - expected - ); - assert_eq!(warm.entries(1).unwrap().len(), count); - let error = warm.child(1, OsStr::new("missing")).err().unwrap(); - assert_eq!( - error.downcast_ref::().unwrap().kind(), - std::io::ErrorKind::NotFound - ); - // A different generation cannot borrow entries from the old snapshot. - assert!( - RemoteFs::new( - client(), - digest.clone(), - blocks.clone(), - Some(temp.path().join("new-generation")) - ) - .is_err() - ); - let root = CacheRequest::Stat { - digest: digest.clone(), - path: vec![], - }; - fs::write( - metadata.join(&hash(&serde_json::to_vec(&root).unwrap())[7..]), - b"corrupt", - ) - .unwrap(); - assert!(RemoteFs::new(client(), digest, blocks, Some(metadata)).is_err()); - } - - #[test] - fn reads_only_requested_blocks_and_reuses_verified_cache() { - let (temp, server, client, digest) = fixture(); - let cache = temp.path().join("client"); - let mut filesystem = RemoteFs::new(client, digest, cache, None).unwrap(); - let file = filesystem.child(1, OsStr::new("large")).unwrap(); - assert_eq!(filesystem.entries(1).unwrap().len(), 4); - assert_eq!( - server.reads.load(Ordering::Relaxed), - 0, - "metadata must not download content" - ); - let offset = MAX_READ as u64 - 4; - assert_eq!( - filesystem.read_range(file.attr.ino, offset, 16).unwrap(), - [42; 16] - ); - assert_eq!( - server.reads.load(Ordering::Relaxed), - 2, - "only two intersecting blocks are fetched" - ); - assert_eq!( - filesystem.read_range(file.attr.ino, offset, 16).unwrap(), - [42; 16] - ); - assert_eq!(server.reads.load(Ordering::Relaxed), 2); - let progress = filesystem.downloads.snapshot(); - assert_eq!(progress.downloaded_files, 1); - assert_eq!(progress.downloaded_bytes, 2 * MAX_READ as u64); - let path = filesystem.cache.join(&hash(&file.path)[7..]).join("0"); - fs::write(path, b"corrupt").unwrap(); - assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); - assert_eq!( - server.reads.load(Ordering::Relaxed), - 3, - "corrupt cache must be replaced" - ); - assert_eq!(filesystem.downloads.snapshot().downloaded_files, 1); - assert_eq!( - filesystem.downloads.snapshot().downloaded_bytes, - 3 * MAX_READ as u64 - ); - let warm_client = CacheClient::new(filesystem.client.endpoint.clone(), None).unwrap(); - let mut warm = RemoteFs::new( - warm_client, - filesystem.digest.clone(), - filesystem.cache.clone(), - None, - ) - .unwrap(); - let warm_file = warm.child(1, OsStr::new("large")).unwrap(); - assert_eq!( - warm.read_range(warm_file.attr.ino, offset, 16).unwrap(), - [42; 16] - ); - assert_eq!(warm.downloads.snapshot().downloaded_files, 0); - assert_eq!(warm.downloads.snapshot().downloaded_bytes, 0); - assert_eq!(warm.downloads.snapshot().cached_files, 1); - assert_eq!(warm.downloads.snapshot().cached_bytes, 2 * MAX_READ as u64); - warm.read_range(warm_file.attr.ino, offset, 16).unwrap(); - assert_eq!(warm.downloads.snapshot().cached_files, 1); - assert_eq!(warm.downloads.snapshot().cached_bytes, 4 * MAX_READ as u64); - drop(server); - assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); - assert!( - filesystem - .read_range(file.attr.ino, 2 * MAX_READ as u64, 1) - .is_err(), - "uncached data must fail, never become zeroes" - ); - } - - #[test] - #[ignore = "requires a working host FUSE installation"] - fn native_mount_reads_lazily_and_unmounts() { - use std::io::Read; - let (temp, server, client, digest) = fixture(); - let filesystem = RemoteFs::new(client, digest, temp.path().join("client"), None).unwrap(); - let mount = mount(filesystem, temp.path()).unwrap(); - assert_eq!( - fs::metadata(mount.path.join("large")).unwrap().len(), - 3 * MAX_READ as u64 - ); - assert_eq!(server.reads.load(Ordering::Relaxed), 0); - assert_eq!( - fs::read_link(mount.path.join("alias")).unwrap(), - Path::new("large") - ); - #[cfg(target_os = "macos")] - { - use std::os::fd::AsRawFd; - let link = OpenOptions::new() - .read(true) - .custom_flags(libc::O_SYMLINK) - .open(mount.path.join("alias")) - .unwrap(); - let mut path = [0u8; libc::PATH_MAX as usize]; - assert_eq!( - unsafe { libc::fcntl(link.as_raw_fd(), libc::F_GETPATH, path.as_mut_ptr()) }, - 0 - ); - } - let mut file = std::fs::File::open(mount.path.join("large")).unwrap(); - let mut bytes = [0u8; 16]; - file.read_exact(&mut bytes).unwrap(); - assert_eq!(bytes, [42; 16]); - assert!(server.reads.load(Ordering::Relaxed) < 3); - drop(file); - let path = mount.path.clone(); - drop(mount); - #[cfg(target_os = "macos")] - assert!(!persisting_overlayfs::is_mountpoint(&path)); - #[cfg(target_os = "linux")] - assert!(!path.exists()); - } - - #[test] - fn auto_probe_distinguishes_absence_from_explicit_failure() { - let temp = tempfile::tempdir().unwrap(); - let address = format!("unix://{}", temp.path().join("missing").display()); - assert!( - CacheClient::probe(address.clone(), None, false) - .unwrap() - .is_none() - ); - assert!(CacheClient::probe(address, None, true).is_err()); - let socket = temp.path().join("stale"); - drop(UnixListener::bind(&socket).unwrap()); - assert!( - CacheClient::probe(format!("unix://{}", socket.display()), None, false) - .unwrap() - .is_none() - ); - let (_temp, _server, client, _) = fixture(); - assert!( - CacheClient::probe(client.endpoint.clone(), None, false) - .unwrap() - .is_some() - ); - } -} +mod tests; diff --git a/crates/persisting-pvisor/src/image/cache/lazy/tests.rs b/crates/persisting-pvisor/src/image/cache/lazy/tests.rs new file mode 100644 index 00000000..7c34a681 --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/lazy/tests.rs @@ -0,0 +1,378 @@ +use super::*; +use crate::image::cache::protocol::{Envelope, read_frame, write_frame}; +use crate::image::cache::server::handle; +use std::os::unix::net::UnixListener; +use std::sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, +}; + +struct Server { + stop: Arc, + worker: Option>, + reads: Arc, + stats: Arc, + lists: Arc, +} +impl Drop for Server { + fn drop(&mut self) { + self.stop.store(true, Ordering::Relaxed); + self.worker.take().unwrap().join().unwrap(); + } +} +fn fixture() -> (tempfile::TempDir, Server, CacheClient, String) { + let temp = tempfile::tempdir().unwrap(); + let store = ImageStore::new(Some(temp.path().join("store"))).unwrap(); + let digest = format!("sha256:{}", "b".repeat(64)); + let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); + fs::create_dir(&root).unwrap(); + fs::write(root.join("large"), vec![42; 3 * MAX_READ as usize]).unwrap(); + std::os::unix::fs::symlink("large", root.join("alias")).unwrap(); + let socket = temp.path().join("s"); + let listener = UnixListener::bind(&socket).unwrap(); + listener.set_nonblocking(true).unwrap(); + let stop = Arc::new(AtomicBool::new(false)); + let reads = Arc::new(AtomicUsize::new(0)); + let stats = Arc::new(AtomicUsize::new(0)); + let lists = Arc::new(AtomicUsize::new(0)); + let worker_stats = stats.clone(); + let worker_lists = lists.clone(); + let worker_stop = stop.clone(); + let worker_reads = reads.clone(); + let worker = std::thread::spawn(move || { + while !worker_stop.load(Ordering::Relaxed) { + let (mut socket, _) = match listener.accept() { + Ok(s) => s, + Err(e) if e.kind() == std::io::ErrorKind::WouldBlock => { + std::thread::sleep(Duration::from_millis(1)); + continue; + } + Err(e) => panic!("{e}"), + }; + socket.set_nonblocking(false).unwrap(); + let envelope: Envelope = read_frame(&mut socket).unwrap(); + if matches!(&envelope.request, CacheRequest::Stat { .. }) { + worker_stats.fetch_add(1, Ordering::Relaxed); + } + if matches!(&envelope.request, CacheRequest::List { .. }) { + worker_lists.fetch_add(1, Ordering::Relaxed); + } + if matches!(&envelope.request, CacheRequest::Read { .. }) { + worker_reads.fetch_add(1, Ordering::Relaxed); + } + let (response, bytes) = handle(&store, envelope.request).unwrap_or_else(|e| { + let code = if e + .downcast_ref::() + .is_some_and(|e| e.kind() == std::io::ErrorKind::NotFound) + { + "not_found" + } else { + "request_failed" + }; + ( + Response::Error { + code: code.into(), + message: e.to_string(), + }, + Vec::new(), + ) + }); + write_frame(&mut socket, &response).unwrap(); + socket.write_all(&bytes).unwrap(); + } + }); + let client = CacheClient::new(format!("unix://{}", socket.display()), None).unwrap(); + ( + temp, + Server { + stop, + reads, + stats, + lists, + worker: Some(worker), + }, + client, + digest, + ) +} + +#[test] +fn cold_directory_batches_attributes_and_preserves_file_boundaries() { + let (temp, server, client, digest) = fixture(); + let root = temp + .path() + .join("store/rootfs-v3/sha256") + .join(&digest[7..]); + for index in 0..600 { + fs::write(root.join(format!("file-{index:03}")), [index as u8]).unwrap(); + } + let mut filesystem = RemoteFs::new( + client, + digest, + temp.path().join("blocks"), + Some(temp.path().join("metadata")), + ) + .unwrap(); + assert_eq!(filesystem.entries(1).unwrap().len(), 604); + assert_eq!(server.stats.load(Ordering::Relaxed), 1, "root only"); + assert_eq!( + server.lists.load(Ordering::Relaxed), + 3, + "three metadata pages" + ); + assert_eq!(server.reads.load(Ordering::Relaxed), 0); + for index in [0, 1, 599] { + let node = filesystem + .child(1, OsStr::new(&format!("file-{index:03}"))) + .unwrap(); + assert_eq!( + filesystem.read_range(node.attr.ino, 0, 100).unwrap(), + [index as u8] + ); + assert_eq!( + filesystem.read_range(node.attr.ino, 0, 100).unwrap(), + [index as u8] + ); + } + assert_eq!( + server.stats.load(Ordering::Relaxed), + 1, + "lookup reuses listed attributes" + ); + assert_eq!(server.reads.load(Ordering::Relaxed), 3); + assert!(filesystem.child(1, OsStr::new("absent")).is_err()); + assert_eq!( + server.stats.load(Ordering::Relaxed), + 1, + "complete listing proves absence" + ); + assert_eq!( + filesystem.downloads.snapshot().downloaded_bytes, + 3, + "small files are not padded to 1 MiB" + ); + assert_eq!(filesystem.downloads.snapshot().cached_bytes, 3); +} + +#[test] +fn hot_blocks_bound_memory_and_reuse_verified_buffers() { + let mut hot = HotBlocks::default(); + let bytes: Arc<[u8]> = vec![42; MAX_READ as usize].into(); + hot.insert(1, 0, bytes.clone()); + assert!(Arc::ptr_eq(&hot.get(1, 0).unwrap(), &bytes)); + for file in 2..=66 { + hot.insert(file, 0, bytes.clone()); + } + assert_eq!(hot.bytes, HotBlocks::MAX_BYTES); + assert_eq!(hot.files.len(), 64); + assert!(hot.get(1, 0).is_none()); + let mut tiny = HotBlocks::default(); + let byte: Arc<[u8]> = vec![1].into(); + for file in 0..=HotBlocks::MAX_ENTRIES as u64 { + tiny.insert(file, 0, byte.clone()); + } + assert_eq!(tiny.order.len(), HotBlocks::MAX_ENTRIES); + assert_eq!(tiny.files.len(), HotBlocks::MAX_ENTRIES); +} + +#[test] +fn persistent_metadata_survives_remount_and_rejects_corruption() { + let (temp, server, client, digest) = fixture(); + let blocks = temp.path().join("blocks"); + let metadata = temp.path().join("metadata"); + let endpoint = client.endpoint.clone(); + let mut cold = RemoteFs::new( + client, + digest.clone(), + blocks.clone(), + Some(metadata.clone()), + ) + .unwrap(); + let expected = cold.child(1, OsStr::new("large")).unwrap().attr.size; + let count = cold.entries(1).unwrap().len(); + assert!(cold.child(1, OsStr::new("missing")).is_err()); + drop(server); + let client = || CacheClient::new(endpoint.clone(), None).unwrap(); + let mut warm = RemoteFs::new( + client(), + digest.clone(), + blocks.clone(), + Some(metadata.clone()), + ) + .unwrap(); + assert_eq!( + warm.child(1, OsStr::new("large")).unwrap().attr.size, + expected + ); + assert_eq!(warm.entries(1).unwrap().len(), count); + let error = warm.child(1, OsStr::new("missing")).err().unwrap(); + assert_eq!( + error.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::NotFound + ); + // A different generation cannot borrow entries from the old snapshot. + assert!( + RemoteFs::new( + client(), + digest.clone(), + blocks.clone(), + Some(temp.path().join("new-generation")) + ) + .is_err() + ); + let root = CacheRequest::Stat { + digest: digest.clone(), + path: vec![], + }; + fs::write( + metadata.join(&hash(&serde_json::to_vec(&root).unwrap())[7..]), + b"corrupt", + ) + .unwrap(); + assert!(RemoteFs::new(client(), digest, blocks, Some(metadata)).is_err()); +} + +#[test] +fn reads_only_requested_blocks_and_reuses_verified_cache() { + let (temp, server, client, digest) = fixture(); + let cache = temp.path().join("client"); + let mut filesystem = RemoteFs::new(client, digest, cache, None).unwrap(); + let file = filesystem.child(1, OsStr::new("large")).unwrap(); + assert_eq!(filesystem.entries(1).unwrap().len(), 4); + assert_eq!( + server.reads.load(Ordering::Relaxed), + 0, + "metadata must not download content" + ); + let offset = MAX_READ as u64 - 4; + assert_eq!( + filesystem.read_range(file.attr.ino, offset, 16).unwrap(), + [42; 16] + ); + assert_eq!( + server.reads.load(Ordering::Relaxed), + 2, + "only two intersecting blocks are fetched" + ); + assert_eq!( + filesystem.read_range(file.attr.ino, offset, 16).unwrap(), + [42; 16] + ); + assert_eq!(server.reads.load(Ordering::Relaxed), 2); + let progress = filesystem.downloads.snapshot(); + assert_eq!(progress.downloaded_files, 1); + assert_eq!(progress.downloaded_bytes, 2 * MAX_READ as u64); + let path = filesystem.cache.join(&hash(&file.path)[7..]).join("0"); + fs::write(path, b"corrupt").unwrap(); + assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); + assert_eq!(server.reads.load(Ordering::Relaxed), 2); + *filesystem.hot.borrow_mut() = HotBlocks::default(); + assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); + assert_eq!( + server.reads.load(Ordering::Relaxed), + 3, + "corrupt cache must be replaced" + ); + assert_eq!(filesystem.downloads.snapshot().downloaded_files, 1); + assert_eq!( + filesystem.downloads.snapshot().downloaded_bytes, + 3 * MAX_READ as u64 + ); + let warm_client = CacheClient::new(filesystem.client.endpoint.clone(), None).unwrap(); + let mut warm = RemoteFs::new( + warm_client, + filesystem.digest.clone(), + filesystem.cache.clone(), + None, + ) + .unwrap(); + let warm_file = warm.child(1, OsStr::new("large")).unwrap(); + assert_eq!( + warm.read_range(warm_file.attr.ino, offset, 16).unwrap(), + [42; 16] + ); + assert_eq!(warm.downloads.snapshot().downloaded_files, 0); + assert_eq!(warm.downloads.snapshot().downloaded_bytes, 0); + assert_eq!(warm.downloads.snapshot().cached_files, 1); + assert_eq!(warm.downloads.snapshot().cached_bytes, 16); + warm.read_range(warm_file.attr.ino, offset, 16).unwrap(); + assert_eq!(warm.downloads.snapshot().cached_files, 1); + assert_eq!(warm.downloads.snapshot().cached_bytes, 32); + drop(server); + assert_eq!(filesystem.read_range(file.attr.ino, 0, 1).unwrap(), [42]); + assert!( + filesystem + .read_range(file.attr.ino, 2 * MAX_READ as u64, 1) + .is_err(), + "uncached data must fail, never become zeroes" + ); +} + +#[test] +#[ignore = "requires a working host FUSE installation"] +fn native_mount_reads_lazily_and_unmounts() { + use std::io::Read; + let (temp, server, client, digest) = fixture(); + let filesystem = RemoteFs::new(client, digest, temp.path().join("client"), None).unwrap(); + let mount = mount(filesystem, temp.path()).unwrap(); + assert_eq!( + fs::metadata(mount.path.join("large")).unwrap().len(), + 3 * MAX_READ as u64 + ); + assert_eq!(server.reads.load(Ordering::Relaxed), 0); + assert_eq!( + fs::read_link(mount.path.join("alias")).unwrap(), + Path::new("large") + ); + #[cfg(target_os = "macos")] + { + use std::os::fd::AsRawFd; + let link = OpenOptions::new() + .read(true) + .custom_flags(libc::O_SYMLINK) + .open(mount.path.join("alias")) + .unwrap(); + let mut path = [0u8; libc::PATH_MAX as usize]; + assert_eq!( + unsafe { libc::fcntl(link.as_raw_fd(), libc::F_GETPATH, path.as_mut_ptr()) }, + 0 + ); + } + let mut file = std::fs::File::open(mount.path.join("large")).unwrap(); + let mut bytes = [0u8; 16]; + file.read_exact(&mut bytes).unwrap(); + assert_eq!(bytes, [42; 16]); + assert!(server.reads.load(Ordering::Relaxed) < 3); + drop(file); + let path = mount.path.clone(); + drop(mount); + #[cfg(target_os = "macos")] + assert!(!persisting_overlayfs::is_mountpoint(&path)); + #[cfg(target_os = "linux")] + assert!(!path.exists()); +} + +#[test] +fn auto_probe_distinguishes_absence_from_explicit_failure() { + let temp = tempfile::tempdir().unwrap(); + let address = format!("unix://{}", temp.path().join("missing").display()); + assert!( + CacheClient::probe(address.clone(), None, false) + .unwrap() + .is_none() + ); + assert!(CacheClient::probe(address, None, true).is_err()); + let socket = temp.path().join("stale"); + drop(UnixListener::bind(&socket).unwrap()); + assert!( + CacheClient::probe(format!("unix://{}", socket.display()), None, false) + .unwrap() + .is_none() + ); + let (_temp, _server, client, _) = fixture(); + assert!( + CacheClient::probe(client.endpoint.clone(), None, false) + .unwrap() + .is_some() + ); +} diff --git a/crates/persisting-pvisor/src/image/cache/mod.rs b/crates/persisting-pvisor/src/image/cache/mod.rs new file mode 100644 index 00000000..ae9d2d9b --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/mod.rs @@ -0,0 +1,31 @@ +//! Shared read-only OCI cache: CLI, client, wire protocol, and server. +//! See docs/shared-image-cache.md for storage layout and lifecycle. + +mod cli; +mod client; +pub(crate) mod progress; +mod protocol; +mod server; +mod transport; + +pub use cli::{CacheArgs, run}; +pub use client::CacheClient; +pub use progress::ImageTotals; +use protocol::hash; +pub use protocol::{MAX_READ, Request, Response}; +pub use transport::default_endpoint; + +#[cfg(any(target_os = "macos", target_os = "linux"))] +mod lazy; +#[cfg(any(target_os = "macos", target_os = "linux"))] +pub(crate) use lazy::{LazyMount, prepare_image}; + +pub const SERVER_ENV: &str = "PERSISTING_PVISOR_CACHE_SERVER"; + +fn architecture() -> &'static str { + match std::env::consts::ARCH { + "aarch64" => "arm64", + "x86_64" => "amd64", + other => other, + } +} diff --git a/crates/persisting-pvisor/src/cache/progress.rs b/crates/persisting-pvisor/src/image/cache/progress.rs similarity index 94% rename from crates/persisting-pvisor/src/cache/progress.rs rename to crates/persisting-pvisor/src/image/cache/progress.rs index f30d6dd4..65086199 100644 --- a/crates/persisting-pvisor/src/cache/progress.rs +++ b/crates/persisting-pvisor/src/image/cache/progress.rs @@ -35,7 +35,7 @@ pub(super) fn loading( label: &str, work: impl FnOnce() -> anyhow::Result, ) -> anyhow::Result { - crate::cli::diagnostic(format_args!("pVisor image: {label}")); + crate::diagnostics::diagnostic(format_args!("pVisor image: {label}")); let started = Instant::now(); std::thread::scope(|scope| { let (done, wait) = mpsc::channel::<()>(); @@ -44,7 +44,7 @@ pub(super) fn loading( wait.recv_timeout(Duration::from_secs(5)), Err(mpsc::RecvTimeoutError::Timeout) ) { - crate::cli::diagnostic(format_args!( + crate::diagnostics::diagnostic(format_args!( "pVisor image: {label}; still waiting ({:.0}s elapsed)", started.elapsed().as_secs_f64() )); @@ -53,11 +53,11 @@ pub(super) fn loading( let result = work(); drop(done); match &result { - Ok(_) => crate::cli::diagnostic(format_args!( + Ok(_) => crate::diagnostics::diagnostic(format_args!( "pVisor image: {label}; done ({:.1}s)", started.elapsed().as_secs_f64() )), - Err(error) => crate::cli::diagnostic(format_args!( + Err(error) => crate::diagnostics::diagnostic(format_args!( "pVisor image: {label}; failed ({:.1}s): {error:#}", started.elapsed().as_secs_f64() )), @@ -96,7 +96,7 @@ impl Downloads { let mut snapshot = self.snapshot.borrow_mut(); snapshot.downloaded_files += u64::from(first); snapshot.downloaded_bytes += bytes as u64; - crate::cli::diagnostic(format_args!( + crate::diagnostics::diagnostic(format_args!( "pVisor image: transferred {bytes} bytes from /{} (this run: {} bytes across {} files)", String::from_utf8_lossy(path).escape_debug(), snapshot.downloaded_bytes, @@ -111,7 +111,7 @@ impl Downloads { snapshot.cached_files += u64::from(first); snapshot.cached_bytes += bytes as u64; if first { - crate::cli::diagnostic(format_args!( + crate::diagnostics::diagnostic(format_args!( "pVisor image: cached /{} ({bytes} bytes read; no download)", String::from_utf8_lossy(path).escape_debug(), )); @@ -142,10 +142,10 @@ impl Downloads { } pub(super) fn image_totals( - store: &crate::oci::ImageStore, + store: &crate::image::oci::ImageStore, digest: &str, ) -> anyhow::Result { - let hex = crate::oci::digest_hex(digest)?; + let hex = crate::image::oci::digest_hex(digest)?; let record = store .root .join("metadata/sha256") diff --git a/crates/persisting-pvisor/src/image/cache/protocol.rs b/crates/persisting-pvisor/src/image/cache/protocol.rs new file mode 100644 index 00000000..601905fc --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/protocol.rs @@ -0,0 +1,119 @@ +//! Versioned requests, responses, and length-prefixed wire frames. +use super::ImageTotals; +use anyhow::ensure; +use serde::{Deserialize, Serialize, de::DeserializeOwned}; +use sha2::{Digest, Sha256}; +use std::io::{Read, Write}; + +pub(super) const MAX_FRAME: usize = 1024 * 1024; +pub const MAX_READ: u32 = 1024 * 1024; + +/// One request per connection. All paths are Unix bytes, relative to image root. +#[derive(Debug, Serialize, Deserialize)] +#[serde(tag = "op", rename_all = "snake_case", deny_unknown_fields)] +pub enum Request { + Ping, + Prepare { + image: String, + architecture: String, + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + refresh: bool, + }, + List { + digest: String, + path: Vec, + offset: usize, + }, + Stat { + digest: String, + path: Vec, + }, + Read { + digest: String, + path: Vec, + offset: u64, + length: u32, + }, +} + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct Envelope { + pub(super) version: u32, + pub(super) token: Option, + pub(super) request: Request, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(tag = "status", rename_all = "snake_case")] +pub enum Response { + Ready, + Prepared { + #[serde(default)] + metadata_generation: Option, + #[serde(default)] + totals: Option, + digest: String, + architecture: String, + env: std::collections::BTreeMap, + entrypoint: Vec, + cmd: Vec, + }, + Entries { + names: Vec>, + /// Attributes aligned with names; absent on older servers. + #[serde(default, skip_serializing_if = "Option::is_none")] + metadata: Option>, + next_offset: Option, + }, + Metadata { + kind: String, + size: u64, + mode: u32, + uid: u32, + gid: u32, + inode: u64, + nlink: u64, + mtime: i64, + mtime_nsec: i64, + target: Option>, + }, + Data { + length: u32, + sha256: String, + }, + Error { + code: String, + message: String, + }, +} + +pub(super) fn read_frame(stream: &mut impl Read) -> anyhow::Result { + let mut header = [0; 4]; + stream.read_exact(&mut header)?; + let length = u32::from_be_bytes(header) as usize; + ensure!( + length > 0 && length <= MAX_FRAME, + "invalid cache frame length" + ); + let mut bytes = vec![0; length]; + stream.read_exact(&mut bytes)?; + Ok(serde_json::from_slice(&bytes)?) +} +pub(super) fn write_frame(stream: &mut impl Write, value: &impl Serialize) -> anyhow::Result<()> { + let bytes = serde_json::to_vec(value)?; + ensure!( + bytes.len() <= MAX_FRAME, + "cache response exceeds frame limit" + ); + stream.write_all(&(bytes.len() as u32).to_be_bytes())?; + stream.write_all(&bytes)?; + Ok(()) +} + +pub(super) fn hash(bytes: &[u8]) -> String { + format!( + "sha256:{}", + crate::image::oci::encode_hex(&Sha256::digest(bytes)) + ) +} diff --git a/crates/persisting-pvisor/src/image/cache/server.rs b/crates/persisting-pvisor/src/image/cache/server.rs new file mode 100644 index 00000000..93524aba --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/server.rs @@ -0,0 +1,457 @@ +//! Authenticated file service and bounded request/preparation workers. +use super::protocol::{Envelope, MAX_FRAME, hash, read_frame, write_frame}; +use super::transport::{Endpoint, Stream, TIMEOUT, TOKEN_ENV, endpoint}; +use super::{MAX_READ, Request, Response, progress}; +use crate::image::oci::ImageStore; +use anyhow::{Context, bail, ensure}; +use fs2::FileExt; +use std::ffi::{CStr, CString, OsStr}; +use std::fs::{self, File, OpenOptions}; +use std::io::{Read, Seek, SeekFrom, Write}; +use std::net::TcpListener; +use std::os::fd::{AsRawFd, FromRawFd, IntoRawFd}; +use std::os::unix::ffi::OsStrExt; +use std::os::unix::fs::{FileTypeExt, OpenOptionsExt, PermissionsExt}; +use std::os::unix::net::{UnixListener, UnixStream}; +use std::path::{Component, Path}; +use std::sync::{Arc, Mutex, mpsc}; +use std::time::Duration; + +mod metadata; +#[cfg(test)] +mod tests; + +// Every component is opened relative to its parent fd, without following links. +// This remains confined even if a directory is renamed during a request. +fn open_child(parent: &File, name: &OsStr, directory: bool) -> anyhow::Result { + let name = CString::new(name.as_bytes())?; + let flags = libc::O_RDONLY + | libc::O_CLOEXEC + | libc::O_NOFOLLOW + | libc::O_NONBLOCK + | if directory { libc::O_DIRECTORY } else { 0 }; + let fd = unsafe { libc::openat(parent.as_raw_fd(), name.as_ptr(), flags) }; + if fd < 0 { + return Err(std::io::Error::last_os_error().into()); + } + Ok(unsafe { File::from_raw_fd(fd) }) +} + +fn parent(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result<(File, Vec)> { + let digest = crate::image::oci::digest_hex(digest)?; + ensure!(!path.contains(&0), "NUL in cache path"); + let path = Path::new(OsStr::from_bytes(path)); + let components: Vec<_> = path.components().collect(); + ensure!( + components.iter().all(|c| matches!(c, Component::Normal(_))), + "cache path must be relative without dot or parent components" + ); + let roots = OpenOptions::new() + .read(true) + .custom_flags(libc::O_DIRECTORY | libc::O_NOFOLLOW) + .open(store.root.join("rootfs-v3/sha256"))?; + let mut directory = open_child(&roots, OsStr::new(digest), true)?; + for component in components.iter().take(components.len().saturating_sub(1)) { + directory = open_child(&directory, component.as_os_str(), true)?; + } + let name = components + .last() + .map_or_else(|| b".".to_vec(), |c| c.as_os_str().as_bytes().to_vec()); + Ok((directory, name)) +} + +fn directory_names(directory: File) -> anyhow::Result>> { + let raw = unsafe { libc::fdopendir(directory.as_raw_fd()) }; + if raw.is_null() { + return Err(std::io::Error::last_os_error().into()); + } + let _ = directory.into_raw_fd(); // fdopendir owns the descriptor on success. + struct Directory(*mut libc::DIR); + impl Drop for Directory { + fn drop(&mut self) { + unsafe { + libc::closedir(self.0); + } + } + } + let directory = Directory(raw); + let mut names = Vec::new(); + loop { + #[cfg(target_os = "macos")] + unsafe { + *libc::__error() = 0; + } + #[cfg(target_os = "linux")] + unsafe { + *libc::__errno_location() = 0; + } + let entry = unsafe { libc::readdir(directory.0) }; + if entry.is_null() { + let error = std::io::Error::last_os_error(); + if error.raw_os_error() != Some(0) { + return Err(error.into()); + } + break; + } + let name = unsafe { CStr::from_ptr((*entry).d_name.as_ptr()) }.to_bytes(); + if name != b"." && name != b".." { + names.push(name.to_vec()); + } + } + Ok(names) +} + +#[allow(clippy::unnecessary_cast)] // libc stat field widths differ by platform. +fn metadata_at(directory: &File, name: &[u8]) -> anyhow::Result { + let name = CString::new(name)?; + let mut m: libc::stat = unsafe { std::mem::zeroed() }; + if unsafe { + libc::fstatat( + directory.as_raw_fd(), + name.as_ptr(), + &mut m, + libc::AT_SYMLINK_NOFOLLOW, + ) + } != 0 + { + return Err(std::io::Error::last_os_error().into()); + } + let kind = match m.st_mode & libc::S_IFMT { + libc::S_IFREG => "file", + libc::S_IFDIR => "directory", + libc::S_IFLNK => "symlink", + _ => "special", + }; + let target = if kind == "symlink" { + let mut bytes = vec![0u8; 4096]; + let size = unsafe { + libc::readlinkat( + directory.as_raw_fd(), + name.as_ptr(), + bytes.as_mut_ptr().cast(), + bytes.len(), + ) + }; + if size < 0 { + return Err(std::io::Error::last_os_error().into()); + } + ensure!( + (size as usize) < bytes.len(), + "symlink target exceeds protocol limit" + ); + bytes.truncate(size as usize); + Some(bytes) + } else { + None + }; + Ok(Response::Metadata { + kind: kind.into(), + size: m.st_size as u64, + mode: m.st_mode as u32, + uid: m.st_uid, + gid: m.st_gid, + inode: m.st_ino as u64, + nlink: m.st_nlink as u64, + mtime: m.st_mtime as i64, + mtime_nsec: m.st_mtime_nsec as i64, + target, + }) +} + +pub(super) fn handle(store: &ImageStore, request: Request) -> anyhow::Result<(Response, Vec)> { + let response = match request { + Request::Ping => Response::Ready, + Request::Prepare { + image, + architecture: requested, + refresh, + } => { + let image = store.prepare_with_refresh(&image, &requested, refresh)?; + Response::Prepared { + metadata_generation: Some(metadata::generation(store, &image.digest)?), + totals: Some(progress::image_totals(store, &image.digest)?), + digest: image.digest, + architecture: requested, + env: image.env, + entrypoint: image.entrypoint, + cmd: image.cmd, + } + } + Request::List { + digest, + path, + offset, + } => { + let names = metadata::directory(store, &digest, &path)?; + ensure!(offset <= names.len(), "directory offset out of range"); + let mut page = Vec::new(); + let mut attributes = Vec::new(); + // Leave room for JSON field names and pagination. Symlink targets and + // non-UTF-8 names can expand substantially in JSON byte arrays. + let mut frame_bytes = 128; + for name in names.iter().skip(offset).take(256) { + let mut child = path.clone(); + if !child.is_empty() { + child.push(b'/'); + } + child.extend_from_slice(name); + let attr = metadata::stat(store, &digest, &child)?; + let bytes = serde_json::to_vec(name)?.len() + serde_json::to_vec(&attr)?.len() + 2; + if frame_bytes + bytes > MAX_FRAME { + ensure!(!page.is_empty(), "directory entry exceeds protocol limit"); + break; + } + frame_bytes += bytes; + page.push(name.clone()); + attributes.push(attr); + } + let end = offset + page.len(); + Response::Entries { + names: page, + metadata: Some(attributes), + next_offset: (end < names.len()).then_some(end), + } + } + Request::Stat { digest, path } => metadata::stat(store, &digest, &path)?, + Request::Read { + digest, + path, + offset, + length, + } => { + ensure!( + length > 0 && length <= MAX_READ, + "read length must be 1..={MAX_READ}" + ); + let (directory, name) = parent(store, &digest, &path)?; + let mut file = open_child(&directory, OsStr::from_bytes(&name), false)?; + ensure!(file.metadata()?.is_file(), "only regular files can be read"); + file.seek(SeekFrom::Start(offset))?; + let mut body = Vec::new(); + file.take(length as u64).read_to_end(&mut body)?; + return Ok(( + Response::Data { + length: body.len() as u32, + sha256: hash(&body), + }, + body, + )); + } + }; + Ok((response, Vec::new())) +} + +fn ensure_same_user(socket: &UnixStream) -> anyhow::Result<()> { + #[cfg(target_os = "macos")] + let uid = { + let mut uid = 0; + let mut gid = 0; + if unsafe { libc::getpeereid(socket.as_raw_fd(), &mut uid, &mut gid) } != 0 { + return Err(std::io::Error::last_os_error().into()); + } + uid + }; + #[cfg(target_os = "linux")] + let uid = { + let mut credentials: libc::ucred = unsafe { std::mem::zeroed() }; + let mut length = std::mem::size_of_val(&credentials) as libc::socklen_t; + if unsafe { + libc::getsockopt( + socket.as_raw_fd(), + libc::SOL_SOCKET, + libc::SO_PEERCRED, + (&mut credentials as *mut libc::ucred).cast(), + &mut length, + ) + } != 0 + { + return Err(std::io::Error::last_os_error().into()); + } + credentials.uid + }; + #[cfg(not(any(target_os = "linux", target_os = "macos")))] + anyhow::bail!("cache peer authentication is supported only on Linux and macOS"); + #[cfg(any(target_os = "linux", target_os = "macos"))] + ensure!( + uid == unsafe { libc::geteuid() }, + "cache socket requires the same user" + ); + Ok(()) +} + +fn serve_connection( + mut stream: Stream, + store: &ImageStore, + token: Option<&str>, + prepare: Option<&mpsc::SyncSender<(Stream, Request)>>, +) -> anyhow::Result<()> { + stream.timeouts(Duration::from_secs(5))?; + if let Stream::Unix(socket) = &stream { + ensure_same_user(socket)?; + } + let result = (|| { + let envelope: Envelope = read_frame(&mut stream)?; + ensure!(envelope.version == 1, "unsupported cache protocol version"); + ensure!( + token.is_none() || envelope.token.as_deref() == token, + "cache authentication failed" + ); + Ok(envelope.request) + })(); + stream.timeouts(TIMEOUT)?; + let request = match result { + Ok(request) => request, + Err(error) => return reply_result(stream, Err(error)), + }; + if matches!(request, Request::Prepare { .. }) + && let Some(queue) = prepare + { + return match queue.try_send((stream, request)) { + Ok(()) => Ok(()), + Err( + mpsc::TrySendError::Full((stream, _)) + | mpsc::TrySendError::Disconnected((stream, _)), + ) => reply_result( + stream, + Err(anyhow::anyhow!( + "image preparation queue is busy; retry later" + )), + ), + }; + } + reply_result(stream, handle(store, request)) +} + +fn reply_result( + mut stream: Stream, + result: anyhow::Result<(Response, Vec)>, +) -> anyhow::Result<()> { + let (response, body) = result.unwrap_or_else(|error: anyhow::Error| { + let code = match error.downcast_ref::().map(|e| e.kind()) { + Some(std::io::ErrorKind::NotFound) => "not_found", + Some(std::io::ErrorKind::PermissionDenied) => "permission_denied", + _ => "request_failed", + }; + ( + Response::Error { + code: code.into(), + message: format!("{error:#}"), + }, + Vec::new(), + ) + }); + write_frame(&mut stream, &response)?; + stream.write_all(&body)?; + Ok(()) +} + +pub(super) fn serve( + address: String, + store: ImageStore, + token: Option, +) -> anyhow::Result<()> { + let (send, receive) = mpsc::sync_channel::(16); + let receive = Arc::new(Mutex::new(receive)); + let store = Arc::new(store); + let token = Arc::new(token); + let dispatch = |stream| -> anyhow::Result<()> { + // A full queue closes the connection instead of allocating unbounded workers. + send.try_send(stream) + .map_err(|_| anyhow::anyhow!("cache server busy")) + }; + // Bind before starting workers so address conflicts fail without orphan workers. + enum Listener { + Unix(UnixListener, File), + Tcp(TcpListener), + } + let listener = match endpoint(&address)? { + Endpoint::Unix(path) => { + let directory = path.parent().context("socket requires parent directory")?; + fs::create_dir_all(directory)?; + let lock = OpenOptions::new() + .read(true) + .write(true) + .create(true) + .truncate(false) + .mode(0o600) + .custom_flags(libc::O_NOFOLLOW) + .open(path.with_extension("sock.lock"))?; + lock.try_lock_exclusive() + .context("cache server already running (socket lock held)")?; + match fs::symlink_metadata(&path) { + Ok(metadata) => { + ensure!( + metadata.file_type().is_socket(), + "refusing to replace non-socket path" + ); + match UnixStream::connect(&path) { + Ok(_) => bail!("cache socket already in use"), + Err(error) if error.kind() == std::io::ErrorKind::ConnectionRefused => { + fs::remove_file(&path)? + } + Err(error) => return Err(error.into()), + } + } + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Err(error) => return Err(error.into()), + } + let listener = UnixListener::bind(&path)?; + fs::set_permissions(&path, fs::Permissions::from_mode(0o600))?; + Listener::Unix(listener, lock) + } + Endpoint::Tcp(address) => { + ensure!( + token.as_ref().as_ref().is_some_and(|t| !t.is_empty()), + "TCP requires {TOKEN_ENV}" + ); + Listener::Tcp(TcpListener::bind(address)?) + } + }; + // Slow registry/extraction work must not consume the file-service workers. + let (prepare_send, prepare_receive) = mpsc::sync_channel::<(Stream, Request)>(16); + let prepare_receive = Arc::new(Mutex::new(prepare_receive)); + for _ in 0..2 { + let receive = prepare_receive.clone(); + let store = store.clone(); + std::thread::spawn(move || { + loop { + let job = receive.lock().unwrap().recv(); + let Ok((stream, request)) = job else { break }; + if let Err(error) = reply_result(stream, handle(&store, request)) { + eprintln!("cache preparation: {error}"); + } + } + }); + } + for _ in 0..16 { + let prepare_send = prepare_send.clone(); + let receive = receive.clone(); + let store = store.clone(); + let token = token.clone(); + std::thread::spawn(move || { + loop { + let request = receive.lock().unwrap().recv(); + let Ok(stream) = request else { break }; + if let Err(error) = + serve_connection(stream, &store, token.as_deref(), Some(&prepare_send)) + { + eprintln!("cache connection: {error}"); + } + } + }); + } + eprintln!("pvisor cache listening on {address}"); + match listener { + Listener::Unix(listener, _lock) => { + for stream in listener.incoming() { + let _ = dispatch(Stream::Unix(stream?)); + } + } + Listener::Tcp(listener) => { + for stream in listener.incoming() { + let _ = dispatch(Stream::Tcp(stream?)); + } + } + } + Ok(()) +} diff --git a/crates/persisting-pvisor/src/cache/metadata.rs b/crates/persisting-pvisor/src/image/cache/server/metadata.rs similarity index 53% rename from crates/persisting-pvisor/src/cache/metadata.rs rename to crates/persisting-pvisor/src/image/cache/server/metadata.rs index cb867aff..d42edd47 100644 --- a/crates/persisting-pvisor/src/cache/metadata.rs +++ b/crates/persisting-pvisor/src/image/cache/server/metadata.rs @@ -1,18 +1,19 @@ //! Metadata for immutable, fully prepared image roots. use super::*; -use std::collections::HashMap; +use lru::LruCache; +use std::num::NonZeroUsize; use std::os::unix::fs::MetadataExt; use std::sync::OnceLock; -static STATS: OnceLock>> = OnceLock::new(); +static STATS: OnceLock>> = OnceLock::new(); type Names = Arc>>; -static DIRECTORIES: OnceLock>> = OnceLock::new(); +static DIRECTORIES: OnceLock>> = OnceLock::new(); pub(super) fn generation(store: &ImageStore, digest: &str) -> anyhow::Result { let root = store .root .join("rootfs-v3/sha256") - .join(crate::oci::digest_hex(digest)?); + .join(crate::image::oci::digest_hex(digest)?); let m = fs::symlink_metadata(&root)?; ensure!(m.is_dir(), "image root is not a directory"); // Include store identity and host inode generation: extracted inode numbers @@ -35,22 +36,20 @@ fn key(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow::Result } fn cached( - cache: &OnceLock>>, + cache: &OnceLock>>, key: String, capacity: usize, load: impl FnOnce() -> anyhow::Result, ) -> anyhow::Result { - let mut cache = cache.get_or_init(Default::default).lock().unwrap(); - if let Some(value) = cache.get(&key) { - return Ok(value.clone()); + let cache = + cache.get_or_init(|| Mutex::new(LruCache::new(NonZeroUsize::new(capacity).unwrap()))); + if let Some(value) = cache.lock().unwrap().get(&key).cloned() { + return Ok(value); } - // ponytail: serialize metadata misses and clear at the entry ceiling; - // use per-key loading and an LRU only if cache contention/churn matters. + // Duplicate concurrent misses are harmless for immutable metadata. Keep + // filesystem I/O outside the lock so a slow scan cannot block cache hits. let value = load()?; - if cache.len() >= capacity { - cache.clear(); - } - cache.insert(key, value.clone()); + cache.lock().unwrap().put(key, value.clone()); Ok(value) } @@ -70,3 +69,30 @@ pub(super) fn directory(store: &ImageStore, digest: &str, path: &[u8]) -> anyhow Ok(Arc::new(names)) }) } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn cache_evicts_only_old_entries_and_loads_without_holding_lock() { + let cache = OnceLock::new(); + cached(&cache, "a".into(), 2, || Ok(1)).unwrap(); + cached(&cache, "b".into(), 2, || Ok(2)).unwrap(); + assert_eq!( + cached(&cache, "a".into(), 2, || panic!("cache miss")).unwrap(), + 1 + ); + cached(&cache, "c".into(), 2, || Ok(3)).unwrap(); + assert!(cache.get().unwrap().lock().unwrap().contains("a")); + assert!(!cache.get().unwrap().lock().unwrap().contains("b")); + cached(&cache, "d".into(), 2, || { + assert!( + cache.get().unwrap().try_lock().is_ok(), + "miss held the cache lock" + ); + Ok(4) + }) + .unwrap(); + } +} diff --git a/crates/persisting-pvisor/src/image/cache/server/tests.rs b/crates/persisting-pvisor/src/image/cache/server/tests.rs new file mode 100644 index 00000000..1eb612e2 --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/server/tests.rs @@ -0,0 +1,332 @@ +//! Service confinement, protocol, and client/server regression tests. +use super::*; +use crate::image::cache::CacheClient; +use std::os::unix::fs::symlink; + +fn fixture() -> (tempfile::TempDir, ImageStore, String) { + let tmp = tempfile::tempdir().unwrap(); + let store = ImageStore::new(Some(tmp.path().join("store"))).unwrap(); + let digest = format!("sha256:{}", "a".repeat(64)); + let root = store.root.join("rootfs-v3/sha256").join("a".repeat(64)); + fs::create_dir(&root).unwrap(); + fs::write(root.join("hello"), b"hello world").unwrap(); + fs::create_dir(root.join("dir")).unwrap(); + symlink("hello", root.join("alias")).unwrap(); + symlink("/etc", root.join("escape")).unwrap(); + (tmp, store, digest) +} + +#[test] +fn server_metadata_reuses_directory_index_and_invalidates_rebuilt_root() { + let (_tmp, store, digest) = fixture(); + let first = metadata::directory(&store, &digest, b"").unwrap(); + let again = metadata::directory(&store, &digest, b"").unwrap(); + assert!( + Arc::ptr_eq(&first, &again), + "directory must not be scanned again" + ); + let Response::Metadata { size, .. } = metadata::stat(&store, &digest, b"hello").unwrap() else { + panic!() + }; + assert_eq!(size, 11); + let generation = metadata::generation(&store, &digest).unwrap(); + let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); + fs::rename(&root, root.with_extension("old")).unwrap(); + fs::create_dir(&root).unwrap(); + fs::write(root.join("hello"), b"new").unwrap(); + assert_ne!(generation, metadata::generation(&store, &digest).unwrap()); + assert_eq!( + &*metadata::directory(&store, &digest, b"").unwrap(), + &[b"hello".to_vec()] + ); + let Response::Metadata { size, .. } = metadata::stat(&store, &digest, b"hello").unwrap() else { + panic!() + }; + assert_eq!(size, 3); +} + +#[test] +fn prepare_queue_is_bounded_and_does_not_block_ping() { + let (_tmp, store, _) = fixture(); + let (queue, pending) = mpsc::sync_channel(1); + let exchange = |request| { + let (mut client, server) = UnixStream::pair().unwrap(); + write_frame( + &mut client, + &Envelope { + version: 1, + token: None, + request, + }, + ) + .unwrap(); + serve_connection(Stream::Unix(server), &store, None, Some(&queue)).unwrap(); + client + }; + let prepare = || Request::Prepare { + image: "ubuntu:latest".into(), + architecture: "arm64".into(), + refresh: false, + }; + let _first = exchange(prepare()); + let response: Response = read_frame(&mut exchange(prepare())).unwrap(); + assert!( + matches!(response, Response::Error { .. }), + "full queue must reply rather than hang" + ); + let response: Response = read_frame(&mut exchange(Request::Ping)).unwrap(); + assert!(matches!(response, Response::Ready)); + assert!(matches!( + pending.try_recv().unwrap().1, + Request::Prepare { .. } + )); +} + +#[test] +fn unix_client_roundtrip_and_parallel_reads() { + let (tmp, store, digest) = fixture(); + let path = tmp.path().join("server.sock"); + let listener = UnixListener::bind(&path).unwrap(); + let server = std::thread::spawn(move || { + std::thread::scope(|scope| { + for connection in listener.incoming().take(8) { + let store = &store; + scope.spawn(move || { + serve_connection(Stream::Unix(connection.unwrap()), store, None, None).unwrap() + }); + } + }); + }); + std::thread::scope(|scope| { + for _ in 0..8 { + let address = format!("unix://{}", path.display()); + let digest = digest.clone(); + scope.spawn(move || { + let client = CacheClient::new(address, None).unwrap(); + let (_, body) = client + .request(Request::Read { + digest, + path: b"hello".to_vec(), + offset: 6, + length: 5, + }) + .unwrap(); + assert_eq!(body, b"world"); + }); + } + }); + server.join().unwrap(); +} + +#[test] +fn paths_metadata_pagination_and_eof() { + let (_tmp, store, digest) = fixture(); + let (response, _) = handle( + &store, + Request::List { + digest: digest.clone(), + path: vec![], + offset: 0, + }, + ) + .unwrap(); + match response { + Response::Entries { + names, + metadata, + next_offset, + } => { + assert_eq!(metadata.as_ref().unwrap().len(), names.len()); + assert_eq!( + names, + [ + b"alias".to_vec(), + b"dir".to_vec(), + b"escape".to_vec(), + b"hello".to_vec() + ] + ); + assert!(next_offset.is_none()); + } + _ => panic!("expected directory"), + } + let (response, _) = handle( + &store, + Request::Stat { + digest: digest.clone(), + path: b"alias".to_vec(), + }, + ) + .unwrap(); + assert!( + matches!(response, Response::Metadata { target: Some(target), .. } if target == b"hello") + ); + for path in [ + b"../hello".as_slice(), + b"/etc/passwd", + b"escape/passwd", + b"alias", + b"hello\0", + ] { + assert!( + handle( + &store, + Request::Read { + digest: digest.clone(), + path: path.to_vec(), + offset: 0, + length: 10 + } + ) + .is_err() + ); + } + assert!( + handle( + &store, + Request::Read { + digest: "sha256:../../etc".into(), + path: b"passwd".to_vec(), + offset: 0, + length: 10 + } + ) + .is_err() + ); + assert!( + handle( + &store, + Request::Read { + digest: digest.clone(), + path: b"hello".to_vec(), + offset: 0, + length: MAX_READ + 1 + } + ) + .is_err() + ); + let (_, body) = handle( + &store, + Request::Read { + digest, + path: b"hello".to_vec(), + offset: 100, + length: 10, + }, + ) + .unwrap(); + assert!(body.is_empty()); +} + +#[test] +fn directory_metadata_pages_fit_frames_and_old_pages_still_decode() { + use std::os::unix::ffi::OsStringExt; + let (_tmp, store, digest) = fixture(); + let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); + let target = std::ffi::OsString::from_vec(vec![b'~'; 1000]); + for index in 0..256 { + let mut name = format!("{index:03}-").into_bytes(); + name.extend(vec![b'~'; 240]); + std::os::unix::fs::symlink(&target, root.join(std::ffi::OsString::from_vec(name))).unwrap(); + } + let mut offset = 0; + let mut pages = 0; + loop { + let (response, _) = handle( + &store, + Request::List { + digest: digest.clone(), + path: vec![], + offset, + }, + ) + .unwrap(); + let mut frame = Vec::new(); + write_frame(&mut frame, &response).unwrap(); + assert!(frame.len() <= MAX_FRAME + 4); + let Response::Entries { + names, + metadata, + next_offset, + } = response + else { + panic!() + }; + assert_eq!(metadata.unwrap().len(), names.len()); + assert!(!names.is_empty()); + offset += names.len(); + pages += 1; + if let Some(next) = next_offset { + assert_eq!(next, offset); + } else { + break; + } + } + assert_eq!(offset, 260); + assert!(pages > 1); + let old: Response = + serde_json::from_str(r#"{"status":"entries","names":[[97]],"next_offset":null}"#).unwrap(); + assert!(matches!(old, Response::Entries { metadata: None, .. })); +} + +#[test] +fn rejects_bad_frames_versions_and_tokens() { + assert!(read_frame::(&mut &u32::MAX.to_be_bytes()[..]).is_err()); + assert!(endpoint("tcp://0.0.0.0:9000").is_err()); + assert!(CacheClient::new("tcp://127.0.0.1:9000".into(), None).is_err()); + for (version, token) in [(2, Some("secret")), (1, Some("wrong")), (1, None)] { + let (_tmp, store, digest) = fixture(); + let (mut client, server) = UnixStream::pair().unwrap(); + let worker = std::thread::spawn(move || { + serve_connection(Stream::Unix(server), &store, Some("secret"), None).unwrap() + }); + write_frame( + &mut client, + &Envelope { + version, + token: token.map(str::to_owned), + request: Request::Stat { + digest, + path: b"hello".to_vec(), + }, + }, + ) + .unwrap(); + assert!(matches!( + read_frame::(&mut client).unwrap(), + Response::Error { .. } + )); + worker.join().unwrap(); + } +} + +#[test] +fn client_rejects_corrupt_content() { + let tmp = tempfile::tempdir().unwrap(); + let path = tmp.path().join("server.sock"); + let listener = UnixListener::bind(&path).unwrap(); + let worker = std::thread::spawn(move || { + let (mut socket, _) = listener.accept().unwrap(); + let _: Envelope = read_frame(&mut socket).unwrap(); + write_frame( + &mut socket, + &Response::Data { + length: 3, + sha256: hash(b"abc"), + }, + ) + .unwrap(); + socket.write_all(b"bad").unwrap(); + }); + let client = CacheClient::new(format!("unix://{}", path.display()), None).unwrap(); + let error = client + .request(Request::Read { + digest: "unused".into(), + path: b"file".to_vec(), + offset: 0, + length: 3, + }) + .unwrap_err(); + assert!(error.to_string().contains("digest mismatch")); + worker.join().unwrap(); +} diff --git a/crates/persisting-pvisor/src/image/cache/transport.rs b/crates/persisting-pvisor/src/image/cache/transport.rs new file mode 100644 index 00000000..30a46357 --- /dev/null +++ b/crates/persisting-pvisor/src/image/cache/transport.rs @@ -0,0 +1,95 @@ +//! Shared Unix/TCP endpoint validation and stream I/O. +use super::SERVER_ENV; +use anyhow::{Context, bail, ensure}; +use std::io::{Read, Write}; +use std::net::{SocketAddr, TcpStream}; +use std::os::unix::net::UnixStream; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +pub(super) const TOKEN_ENV: &str = "PERSISTING_PVISOR_CACHE_TOKEN"; +pub(super) const TIMEOUT: Duration = Duration::from_secs(300); + +pub fn default_endpoint() -> anyhow::Result { + let base = dirs::cache_dir().context("cannot find user cache directory")?; + Ok(format!( + "unix://{}", + base.join("persisting/pvisor/cache.sock").display() + )) +} + +pub(super) fn endpoint_from_env() -> anyhow::Result { + match std::env::var(SERVER_ENV) { + Ok(value) => Ok(value), + Err(std::env::VarError::NotPresent) => default_endpoint(), + Err(error) => Err(error.into()), + } +} + +pub(super) enum Endpoint { + Unix(PathBuf), + Tcp(SocketAddr), +} + +pub(super) fn endpoint(value: &str) -> anyhow::Result { + if let Some(path) = value.strip_prefix("unix://") { + ensure!( + Path::new(path).is_absolute(), + "Unix socket path must be absolute" + ); + return Ok(Endpoint::Unix(path.into())); + } + if let Some(address) = value.strip_prefix("tcp://") { + let address: SocketAddr = address + .parse() + .context("TCP endpoint requires an IP address and port")?; + ensure!( + address.ip().is_loopback(), + "cache TCP is loopback-only; use an SSH tunnel for remote access" + ); + return Ok(Endpoint::Tcp(address)); + } + bail!("expected unix:///absolute/path or tcp://127.0.0.1:PORT") +} + +pub(super) enum Stream { + Unix(UnixStream), + Tcp(TcpStream), +} + +impl Read for Stream { + fn read(&mut self, buf: &mut [u8]) -> std::io::Result { + match self { + Self::Unix(s) => s.read(buf), + Self::Tcp(s) => s.read(buf), + } + } +} +impl Write for Stream { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + match self { + Self::Unix(s) => s.write(buf), + Self::Tcp(s) => s.write(buf), + } + } + fn flush(&mut self) -> std::io::Result<()> { + match self { + Self::Unix(s) => s.flush(), + Self::Tcp(s) => s.flush(), + } + } +} +impl Stream { + pub(super) fn timeouts(&self, timeout: Duration) -> std::io::Result<()> { + match self { + Self::Unix(s) => { + s.set_read_timeout(Some(timeout))?; + s.set_write_timeout(Some(timeout)) + } + Self::Tcp(s) => { + s.set_read_timeout(Some(timeout))?; + s.set_write_timeout(Some(timeout)) + } + } + } +} diff --git a/crates/persisting-pvisor/src/image/mod.rs b/crates/persisting-pvisor/src/image/mod.rs new file mode 100644 index 00000000..c76cee2b --- /dev/null +++ b/crates/persisting-pvisor/src/image/mod.rs @@ -0,0 +1,5 @@ +//! OCI preparation and the shared image file cache. + +#[cfg(unix)] +pub mod cache; +pub(crate) mod oci; diff --git a/crates/persisting-pvisor/src/oci.rs b/crates/persisting-pvisor/src/image/oci.rs similarity index 81% rename from crates/persisting-pvisor/src/oci.rs rename to crates/persisting-pvisor/src/image/oci.rs index 97c1fbc9..92f3db53 100644 --- a/crates/persisting-pvisor/src/oci.rs +++ b/crates/persisting-pvisor/src/image/oci.rs @@ -15,6 +15,7 @@ use std::collections::BTreeMap; use std::fs::{self, File, OpenOptions}; use std::io::{Read, Write}; use std::path::{Component, Path, PathBuf}; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; const MANIFEST_ACCEPT: &str = concat!( "application/vnd.oci.image.index.v1+json, ", @@ -30,8 +31,9 @@ struct ImageReference { reference: String, } -#[derive(Debug, Clone)] +#[derive(Debug, Clone, Serialize, Deserialize)] pub struct PreparedImage { + #[serde(skip)] pub rootfs: PathBuf, pub digest: String, pub env: BTreeMap, @@ -110,6 +112,14 @@ struct RootfsMetadata<'a> { manifest_digest: &'a str, } +const TAG_CACHE_TTL: Duration = Duration::from_secs(5 * 60); + +#[derive(Serialize, Deserialize)] +struct PreparedRecord { + checked_at: u64, + prepared: PreparedImage, +} + impl ImageStore { pub fn new(root: Option) -> anyhow::Result { let root = match root { @@ -121,6 +131,8 @@ impl ImageStore { fs::create_dir_all(root.join("metadata/sha256"))?; fs::create_dir_all(root.join("locks"))?; let client = Client::builder() + .connect_timeout(Duration::from_secs(10)) + .timeout(Duration::from_secs(300)) .user_agent(concat!("pvisor/", env!("CARGO_PKG_VERSION"))) .build() .context("build OCI registry client")?; @@ -135,12 +147,119 @@ impl ImageStore { &self, image: &str, architecture: &str, + ) -> anyhow::Result { + self.prepare_with_refresh(image, architecture, false) + } + + pub(crate) fn prepare_with_refresh( + &self, + image: &str, + architecture: &str, + refresh: bool, ) -> anyhow::Result { anyhow::ensure!( matches!(architecture, "arm64" | "amd64"), "unsupported image architecture {architecture}" ); let image_ref = ImageReference::parse(image)?; + self.with_prepared_cache(image_ref.clone(), architecture, refresh, || { + self.resolve_and_prepare(image, architecture, image_ref) + }) + } + + fn with_prepared_cache( + &self, + image_ref: ImageReference, + architecture: &str, + refresh: bool, + prepare: impl FnOnce() -> anyhow::Result, + ) -> anyhow::Result { + let record = self.prepared_record_path(&image_ref, architecture)?; + // Lock before contacting the registry, then recheck the record. Separate + // processes preparing the same reference share the completed result too. + let lock = OpenOptions::new() + .create(true) + .truncate(false) + .read(true) + .write(true) + .open(self.root.join("locks").join(format!( + "prepare-{}.lock", + record.file_stem().unwrap().to_string_lossy() + )))?; + lock.lock_exclusive()?; + if !refresh + && let Some(prepared) = + self.cached_prepared(&record, &image_ref.reference, SystemTime::now()) + { + return Ok(prepared); + } + let prepared = prepare()?; + let record_data = PreparedRecord { + checked_at: SystemTime::now().duration_since(UNIX_EPOCH)?.as_secs(), + prepared: prepared.clone(), + }; + let bytes = serde_json::to_vec(&record_data)?; + fs::create_dir_all(record.parent().unwrap())?; + crate::util::atomic_write(&record, &bytes, 0o600)?; + // Also make the returned platform digest immediately usable offline. + let pinned = ImageReference { + reference: prepared.digest.clone(), + ..image_ref + }; + crate::util::atomic_write( + &self.prepared_record_path(&pinned, architecture)?, + &bytes, + 0o600, + )?; + Ok(prepared) + } + + fn prepared_record_path( + &self, + image: &ImageReference, + architecture: &str, + ) -> anyhow::Result { + let key = serde_json::to_vec(&( + &image.registry, + &image.repository, + &image.reference, + architecture, + ))?; + Ok(self.root.join("metadata/prepared-v1").join(format!( + "{}.json", + crate::util::encode_hex(&Sha256::digest(&key)) + ))) + } + + fn cached_prepared( + &self, + path: &Path, + reference: &str, + now: SystemTime, + ) -> Option { + let mut record: PreparedRecord = serde_json::from_slice(&fs::read(path).ok()?).ok()?; + if !reference.starts_with("sha256:") { + let checked = UNIX_EPOCH.checked_add(Duration::from_secs(record.checked_at))?; + if now.duration_since(checked).ok()? >= TAG_CACHE_TTL { + return None; + } + } + record.prepared.rootfs = self + .root + .join("rootfs-v3/sha256") + .join(digest_hex(&record.prepared.digest).ok()?); + fs::symlink_metadata(&record.prepared.rootfs) + .ok()? + .is_dir() + .then_some(record.prepared) + } + + fn resolve_and_prepare( + &self, + image: &str, + architecture: &str, + image_ref: ImageReference, + ) -> anyhow::Result { let mut registry = RegistryClient::new(&self.client, image_ref.clone()); let (mut body, mut manifest_digest) = registry.fetch_manifest(&image_ref.reference)?; let value: serde_json::Value = serde_json::from_slice(&body) @@ -246,7 +365,7 @@ impl ImageStore { temporary.write_all(&buffer[..read])?; size += read as u64; } - let actual = format!("sha256:{}", encode_hex(&hasher.finalize())); + let actual = format!("sha256:{}", crate::util::encode_hex(&hasher.finalize())); anyhow::ensure!( actual == descriptor.digest, "OCI blob digest mismatch: expected {}, got {actual}", @@ -288,7 +407,7 @@ impl<'a> RegistryClient<'a> { let mut response = self.get(&path, Some(MANIFEST_ACCEPT))?; let mut body = Vec::new(); response.read_to_end(&mut body)?; - let digest = format!("sha256:{}", encode_hex(&Sha256::digest(&body))); + let digest = format!("sha256:{}", crate::util::encode_hex(&Sha256::digest(&body))); if reference.starts_with("sha256:") { verify_digest(reference, &body)?; } @@ -477,7 +596,7 @@ pub(crate) fn digest_hex(digest: &str) -> anyhow::Result<&str> { fn verify_digest(expected: &str, body: &[u8]) -> anyhow::Result<()> { digest_hex(expected)?; - let actual = format!("sha256:{}", encode_hex(&Sha256::digest(body))); + let actual = format!("sha256:{}", crate::util::encode_hex(&Sha256::digest(body))); anyhow::ensure!( actual == expected, "digest mismatch: expected {expected}, got {actual}" @@ -497,7 +616,7 @@ fn verify_file_digest(expected: &str, path: &Path) -> anyhow::Result<()> { } hasher.update(&buffer[..read]); } - let actual = format!("sha256:{}", encode_hex(&hasher.finalize())); + let actual = format!("sha256:{}", crate::util::encode_hex(&hasher.finalize())); anyhow::ensure!( actual == expected, "cached OCI blob {} is corrupt", @@ -506,16 +625,6 @@ fn verify_file_digest(expected: &str, path: &Path) -> anyhow::Result<()> { Ok(()) } -pub(crate) fn encode_hex(bytes: &[u8]) -> String { - const HEX: &[u8; 16] = b"0123456789abcdef"; - let mut output = String::with_capacity(bytes.len() * 2); - for byte in bytes { - output.push(HEX[(byte >> 4) as usize] as char); - output.push(HEX[(byte & 0x0f) as usize] as char); - } - output -} - fn apply_layer(blob: &Path, media_type: &str, rootfs: &Path) -> anyhow::Result<()> { let tar_file = tempfile::NamedTempFile::new()?; { @@ -786,6 +895,95 @@ fn clear_relative_directory(root: &Path, relative: &Path) -> anyhow::Result<()> mod tests { use super::*; + #[test] + fn prepared_cache_coalesces_requests_and_honors_refresh_expiry_and_architecture() { + use std::sync::{ + Barrier, + atomic::{AtomicUsize, Ordering}, + }; + let temp = tempfile::tempdir().unwrap(); + let store = ImageStore::new(Some(temp.path().into())).unwrap(); + let image = ImageReference::parse("ubuntu:latest").unwrap(); + let digest = format!("sha256:{}", "a".repeat(64)); + let root = store.root.join("rootfs-v3/sha256").join(&digest[7..]); + fs::create_dir(&root).unwrap(); + let calls = AtomicUsize::new(0); + let prepare = || { + calls.fetch_add(1, Ordering::SeqCst); + Ok(PreparedImage { + rootfs: root.clone(), + digest: digest.clone(), + env: BTreeMap::new(), + entrypoint: vec![], + cmd: vec!["bash".into()], + }) + }; + let barrier = Barrier::new(8); + std::thread::scope(|scope| { + for _ in 0..8 { + scope.spawn(|| { + barrier.wait(); + let result = store + .with_prepared_cache(image.clone(), "arm64", false, prepare) + .unwrap(); + assert_eq!(result.digest, digest); + }); + } + }); + assert_eq!(calls.load(Ordering::SeqCst), 1); + // The public entry point and a fresh store instance reuse the disk record. + let reopened = ImageStore::new(Some(temp.path().into())).unwrap(); + assert_eq!( + reopened + .prepare_for_architecture("docker.io/library/ubuntu:latest", "arm64") + .unwrap() + .cmd, + ["bash"] + ); + store + .with_prepared_cache(image.clone(), "arm64", true, prepare) + .unwrap(); + assert_eq!(calls.load(Ordering::SeqCst), 2); + let record = store.prepared_record_path(&image, "arm64").unwrap(); + let bytes = fs::read(&record).unwrap(); + let saved: PreparedRecord = serde_json::from_slice(&bytes).unwrap(); + let checked = UNIX_EPOCH + Duration::from_secs(saved.checked_at); + assert!( + store + .cached_prepared(&record, "latest", checked + TAG_CACHE_TTL) + .is_none() + ); + let pinned = ImageReference { + reference: digest.clone(), + ..image.clone() + }; + let pin_record = store.prepared_record_path(&pinned, "arm64").unwrap(); + assert!( + store + .cached_prepared(&pin_record, &digest, checked + TAG_CACHE_TTL * 100) + .is_some() + ); + assert_ne!(record, store.prepared_record_path(&image, "amd64").unwrap()); + // Failed refreshes preserve the last successful record, but return errors. + assert!( + store + .with_prepared_cache(image.clone(), "arm64", true, || anyhow::bail!("offline")) + .is_err() + ); + assert_eq!(fs::read(&record).unwrap(), bytes); + fs::write(&record, b"corrupt").unwrap(); + store + .with_prepared_cache(image, "arm64", false, prepare) + .unwrap(); + assert_eq!(calls.load(Ordering::SeqCst), 3); + fs::remove_dir(&root).unwrap(); + assert!( + store + .cached_prepared(&pin_record, &digest, checked) + .is_none() + ); + } + #[test] fn parses_docker_style_references() { assert_eq!( diff --git a/crates/persisting-pvisor/src/lib.rs b/crates/persisting-pvisor/src/lib.rs index 6796cc84..47338eac 100644 --- a/crates/persisting-pvisor/src/lib.rs +++ b/crates/persisting-pvisor/src/lib.rs @@ -6,47 +6,22 @@ #![cfg_attr(all(target_os = "macos", target_arch = "x86_64"), allow(dead_code))] -#[cfg(unix)] -pub mod cache; pub mod cli; pub mod core; mod runtime; pub mod trace; -mod agentctl; -mod artifact; -mod bundle; -mod checkpoint; mod config; -mod container; -mod control; -mod delegated; -mod event; +mod diagnostics; mod executor; -#[cfg(not(any( - all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), - all(target_os = "macos", target_arch = "x86_64") -)))] -mod firmware; -mod oci; -mod process; -mod pvisor; +mod image; + #[doc(hidden)] -pub mod sandbox; +pub use executor::sandbox; +#[cfg(unix)] +pub use image::cache; mod util; -mod vm; -pub use agentctl::{ - AGENTCTL_MAX_SESSIONS, AgentClientSnapshot, AgentCtlControl, AgentCtlServer, AgentCtlSnapshot, -}; -pub use bundle::{ - BundleArtifact, BundleRun, FilesystemSummary, NetworkSummary, RUN_BUNDLE_FILENAME, - RUN_BUNDLE_SCHEMA_VERSION, ResourceSummary, RunBundle, SafetySummary, -}; -pub use checkpoint::{ - CHECKPOINTS_DIR, CheckpointConsistency, LogicalCheckpoint, create_logical_checkpoint, - latest_logical_checkpoint, restore_logical_checkpoint, -}; pub use config::{ ContainerMount, ContainerNetwork, ContainerPlatform, ContainerSettings, FilesystemMode, GatewayDriverConfig, GatewayMode, GatewaySettings, NetworkDriverConfig, OverlayFsBackend, @@ -54,28 +29,38 @@ pub use config::{ PVisorConfig, RecordSettings, RunConfig, RunExecutorKind, RunPolicy, RunSettings, RunStdio, VmSettings, }; -pub use container::ContainerExecutor; -pub use control::{ - ControlController, ControlEffect, ControlMachine, ControlReason, ControlRequest, ControlState, - ControlTransition, NetworkGuard, NetworkHostRule, NetworkRule, PolicyControlController, - host_matches, is_public_egress_ip, normalize_host, parse_network_rule, -}; -pub use event::{ - EventAppendErrorKind, EventSink, MemoryEventSink, NoopEventSink, RunEventPublisher, -}; +pub use executor::container::ContainerExecutor; +pub use executor::process::ProcessExecutor; +pub use executor::vm::VmExecutor; +pub use executor::vm::run_internal_if_requested as run_krun_internal_if_requested; pub use executor::{AttemptContext, RunExecutor}; pub use persisting_control::{ AGENTCTL_ENDPOINT_ENV, AGENTCTL_MAX_FRAME_BYTES, AGENTCTL_TOKEN_ENV, AGENTCTL_TRANSPORT_ENV, AGENTCTL_VERSION, AGENTCTL_VERSION_ENV, AgentDirective, AgentErrorCode, AgentRequest, - AgentResponse, AgentState, + AgentResponse, AgentState, ControlController, ControlEffect, ControlMachine, ControlReason, + ControlRequest, ControlState, ControlTransition, NetworkGuard, NetworkHostRule, NetworkRule, + PolicyControlController, host_matches, is_public_egress_ip, normalize_host, parse_network_rule, }; pub use persisting_gateway::sink::CaptureEventSink as TrajectoryEventSink; -pub use process::ProcessExecutor; -pub use pvisor::{PVisor, PVisorBuilder, PVisorError, RunCancellation, RunEventStream, RunHandle}; +pub use runtime::agentctl::{ + AGENTCTL_MAX_SESSIONS, AgentClientSnapshot, AgentCtlControl, AgentCtlServer, AgentCtlSnapshot, +}; +pub use runtime::bundle::{ + BundleArtifact, BundleRun, FilesystemSummary, NetworkSummary, RUN_BUNDLE_FILENAME, + RUN_BUNDLE_SCHEMA_VERSION, ResourceSummary, RunBundle, SafetySummary, +}; +pub use runtime::checkpoint::{ + CHECKPOINTS_DIR, CheckpointConsistency, LogicalCheckpoint, create_logical_checkpoint, + latest_logical_checkpoint, restore_logical_checkpoint, +}; +pub use runtime::event::{ + EventAppendErrorKind, EventSink, MemoryEventSink, NoopEventSink, RunEventPublisher, +}; +pub use runtime::run::{ + PVisor, PVisorBuilder, PVisorError, RunCancellation, RunEventStream, RunHandle, +}; pub use runtime::{ ChangeEntry, ChangeEntryType, ChangeKind, ImplantPlan, OverlayHint, RunLineage, RuntimeCapabilities, }; pub use util::unix_now_ms; -pub use vm::VmExecutor; -pub use vm::run_internal_if_requested as run_krun_internal_if_requested; diff --git a/crates/persisting-pvisor/src/agentctl.rs b/crates/persisting-pvisor/src/runtime/agentctl.rs similarity index 100% rename from crates/persisting-pvisor/src/agentctl.rs rename to crates/persisting-pvisor/src/runtime/agentctl.rs diff --git a/crates/persisting-pvisor/src/runtime/attempt.rs b/crates/persisting-pvisor/src/runtime/attempt.rs index 77f6e544..ff8c8cbd 100644 --- a/crates/persisting-pvisor/src/runtime/attempt.rs +++ b/crates/persisting-pvisor/src/runtime/attempt.rs @@ -364,7 +364,7 @@ pub(crate) fn prepare_attempt( )?; spec.metadata.insert( - crate::sandbox::SANDBOX_PROXY_KEY.into(), + crate::executor::sandbox::SANDBOX_PROXY_KEY.into(), gateway.listen.clone().into(), ); @@ -1385,7 +1385,7 @@ fn overlay_cwd( ) -> Option { // Only Linux binds the merged view over the original path in a private root. if cfg!(target_os = "linux") - && crate::sandbox::sandbox_required(spec) + && crate::executor::sandbox::sandbox_required(spec) && !uses_krun_executor(spec) && overlay.merged_dir.is_some() && let Some(record) = record @@ -1427,8 +1427,10 @@ mod tests { #[test] fn safe_overlay_cwd_uses_original_path_only_on_linux() { let mut spec = RunSpec::process("run-1", "agent", "sh"); - spec.metadata - .insert(crate::sandbox::REQUIRED_SANDBOX_KEY.into(), true.into()); + spec.metadata.insert( + crate::executor::sandbox::REQUIRED_SANDBOX_KEY.into(), + true.into(), + ); let config = persisting_gateway::config::OverlayConfig { enabled: true, target: Some("/workspace".into()), diff --git a/crates/persisting-pvisor/src/bundle.rs b/crates/persisting-pvisor/src/runtime/bundle.rs similarity index 99% rename from crates/persisting-pvisor/src/bundle.rs rename to crates/persisting-pvisor/src/runtime/bundle.rs index c4d7eb85..54e6748e 100644 --- a/crates/persisting-pvisor/src/bundle.rs +++ b/crates/persisting-pvisor/src/runtime/bundle.rs @@ -1,9 +1,9 @@ //! Durable, versioned summary of one pVisor Run. +use crate::executor::sandbox::SANDBOX_SETUP_FAILED_WARNING; use crate::runtime::{ ChangeEntry, OverlayState, RunLineage, RunRecord, overlay_changes, overlay_status, }; -use crate::sandbox::SANDBOX_SETUP_FAILED_WARNING; use crate::util::{atomic_write, sync_directory}; use crate::{AgentCtlSnapshot, unix_now_ms}; use persisting_control::{ diff --git a/crates/persisting-pvisor/src/checkpoint.rs b/crates/persisting-pvisor/src/runtime/checkpoint.rs similarity index 100% rename from crates/persisting-pvisor/src/checkpoint.rs rename to crates/persisting-pvisor/src/runtime/checkpoint.rs diff --git a/crates/persisting-pvisor/src/event.rs b/crates/persisting-pvisor/src/runtime/event.rs similarity index 100% rename from crates/persisting-pvisor/src/event.rs rename to crates/persisting-pvisor/src/runtime/event.rs diff --git a/crates/persisting-pvisor/src/runtime/mod.rs b/crates/persisting-pvisor/src/runtime/mod.rs index 115dfae7..55bb8697 100644 --- a/crates/persisting-pvisor/src/runtime/mod.rs +++ b/crates/persisting-pvisor/src/runtime/mod.rs @@ -1,13 +1,18 @@ -//! Attempt preparation for pVisor-owned runtime drivers. +//! Run lifecycle, durable records, and pVisor-owned runtime driver coordination. //! //! pVisor assembles the optional Gateway/OverlayNet driver, network policy, and //! embedded OverlayFS before the Agent process starts. +pub(crate) mod agentctl; mod attempt; +pub(crate) mod bundle; +pub(crate) mod checkpoint; +pub(crate) mod event; mod implant; mod overlay; pub(crate) mod plan; mod registry; +pub(crate) mod run; mod supervisor; pub(crate) mod zcode; diff --git a/crates/persisting-pvisor/src/pvisor.rs b/crates/persisting-pvisor/src/runtime/run.rs similarity index 98% rename from crates/persisting-pvisor/src/pvisor.rs rename to crates/persisting-pvisor/src/runtime/run.rs index 9065f257..ca259758 100644 --- a/crates/persisting-pvisor/src/pvisor.rs +++ b/crates/persisting-pvisor/src/runtime/run.rs @@ -5,9 +5,9 @@ use crate::TrajectoryEventSink; use crate::config::{GatewayDriverConfig, NetworkDriverConfig, PVisorConfig}; -use crate::event::{EventSink, NoopEventSink, RunEventPublisher}; +use crate::executor::process::ProcessExecutor; use crate::executor::{AttemptContext, RunExecutor}; -use crate::process::ProcessExecutor; +use crate::runtime::event::{EventSink, NoopEventSink, RunEventPublisher}; use crate::runtime::{ AttemptTeardown, ImplantPlan, OverlayHint, RuntimeCapabilities, RuntimeSupervisor, RuntimeSupervisorBuilder, @@ -131,7 +131,7 @@ impl RunHandle { .begin_checkpoint(checkpoint_id.to_owned(), Some(deadline))?; loop { if let Some(captured) = checkpoint.try_capture(|| { - crate::checkpoint::create_agent_quiesced_checkpoint(record, checkpoint_id) + crate::runtime::checkpoint::create_agent_quiesced_checkpoint(record, checkpoint_id) })? { return Ok(captured); } @@ -324,14 +324,15 @@ impl PVisor { } self.runtime.apply_network_capability(&mut spec); // Runtime preparation supplies this capability from the bound listener. - spec.metadata.remove(crate::sandbox::SANDBOX_PROXY_KEY); + spec.metadata + .remove(crate::executor::sandbox::SANDBOX_PROXY_KEY); let capability_enforcement = effective_capability_enforcement( &descriptor, &spec, self.runtime.proxy_network_is_configured(), vm_network_executor && self.runtime.vm_network_is_enforcing(), ); - if crate::sandbox::sandbox_required(&spec) { + if crate::executor::sandbox::sandbox_required(&spec) { for dimension in [ CapabilityDimension::FilesystemRead, CapabilityDimension::FilesystemWrite, @@ -343,7 +344,7 @@ impl PVisor { && !matches!(spec.capabilities.network, NetworkCapability::Deny); let cooperative_rootless_chroot = cfg!(target_os = "linux") && descriptor.isolation == IsolationKind::RootlessProcess - && !crate::sandbox::landlock_required(&spec) + && !crate::executor::sandbox::landlock_required(&spec) && matches!( dimension, CapabilityDimension::FilesystemRead | CapabilityDimension::FilesystemWrite @@ -759,7 +760,7 @@ fn effective_capability_enforcement( _ => {} } } - if crate::sandbox::sandbox_required(spec) + if crate::executor::sandbox::sandbox_required(spec) && descriptor.isolation == IsolationKind::SandboxedProcess { evidence.record( @@ -1289,8 +1290,10 @@ mod tests { #[tokio::test] async fn required_sandbox_refuses_an_unsandboxed_executor_before_launch() { let mut spec = RunSpec::process("required-no-fallback", "test", "/bin/true"); - spec.metadata - .insert(crate::sandbox::REQUIRED_SANDBOX_KEY.into(), true.into()); + spec.metadata.insert( + crate::executor::sandbox::REQUIRED_SANDBOX_KEY.into(), + true.into(), + ); let error = match PVisor::new().run(spec).await { Ok(_) => panic!("required sandbox silently fell back"), Err(error) => error, diff --git a/crates/persisting-pvisor/src/runtime/zcode.rs b/crates/persisting-pvisor/src/runtime/zcode.rs index 163f1f2c..10381fef 100644 --- a/crates/persisting-pvisor/src/runtime/zcode.rs +++ b/crates/persisting-pvisor/src/runtime/zcode.rs @@ -58,7 +58,7 @@ pub(crate) fn apply_host_process_policy(spec: &mut RunSpec) -> anyhow::Result<() if let Some(helper) = &helper { let hidden_paths = spec .metadata - .entry(crate::sandbox::SANDBOX_HIDDEN_PATHS_KEY.into()) + .entry(crate::executor::sandbox::SANDBOX_HIDDEN_PATHS_KEY.into()) .or_insert_with(|| Value::Array(Vec::new())); if let Some(paths) = hidden_paths.as_array_mut() { let helper = Value::String(helper.display().to_string()); @@ -80,8 +80,10 @@ pub(crate) fn apply_host_process_policy(spec: &mut RunSpec) -> anyhow::Result<() // cannot be faithfully inferred from host mode bits (notably ACLs). Keep // Electron on its software-rendering path inside the rootless namespace. insert_switch(&mut process.args, "--disable-gpu"); - spec.metadata - .insert(crate::sandbox::SANDBOX_NO_GPU_KEY.into(), Value::Bool(true)); + spec.metadata.insert( + crate::executor::sandbox::SANDBOX_NO_GPU_KEY.into(), + Value::Bool(true), + ); eprintln!( "pVisor ZCode policy: {}pVisor process isolation enabled; software GPU rendering enabled", if helper.is_some() { diff --git a/crates/persisting-pvisor/src/util.rs b/crates/persisting-pvisor/src/util.rs index aa76c42c..b6af324d 100644 --- a/crates/persisting-pvisor/src/util.rs +++ b/crates/persisting-pvisor/src/util.rs @@ -78,6 +78,16 @@ pub(crate) fn atomic_write(path: &Path, contents: &[u8], mode: u32) -> anyhow::R result } +pub(crate) fn encode_hex(bytes: &[u8]) -> String { + const HEX: &[u8; 16] = b"0123456789abcdef"; + let mut output = String::with_capacity(bytes.len() * 2); + for byte in bytes { + output.push(HEX[(byte >> 4) as usize] as char); + output.push(HEX[(byte & 0x0f) as usize] as char); + } + output +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/persisting-replay/src/adapter/claude_code.rs b/crates/persisting-replay/src/adapter/claude_code.rs index 5974cd52..dbf55a8f 100644 --- a/crates/persisting-replay/src/adapter/claude_code.rs +++ b/crates/persisting-replay/src/adapter/claude_code.rs @@ -10,8 +10,8 @@ use super::{ MAX_TOOL_OUTPUT_BYTES, RunContext, agent_command, check_boundary, sanitized_environment, with_boundary_user_prompt_metadata, }; -use crate::claude_bridge::ClaudeBridgeHandle; -use crate::claude_resume::ResumeTransportManifest; +use crate::bridge::claude::ClaudeBridgeHandle; +use crate::bridge::claude_resume::ResumeTransportManifest; use crate::error::{ReplayError, ReplayErrorKind, ResultExt}; use crate::io::{atomic_write, atomic_write_json, canonicalize, read_regular_file, sha256}; use crate::journal::Journal; @@ -2058,7 +2058,7 @@ mod tests { wildcard_match, }; use crate::adapter::{RunContext, build_plan, run}; - use crate::claude_resume::ResumeTransportManifest; + use crate::bridge::claude_resume::ResumeTransportManifest; use crate::journal::Journal; use crate::model::{ AdapterPlan, AgentKind, FreshObservation, PlaybackRequest, ReplayMode, ToolCall, diff --git a/crates/persisting-replay/src/adapter/generic.rs b/crates/persisting-replay/src/adapter/generic.rs index f8452f8b..245eb496 100644 --- a/crates/persisting-replay/src/adapter/generic.rs +++ b/crates/persisting-replay/src/adapter/generic.rs @@ -17,7 +17,8 @@ use super::{ MAX_TOOL_OUTPUT_BYTES, RunContext, agent_command, check_boundary, prepared_outcome, with_boundary_user_prompt_metadata, }; -use crate::codex_bridge::{CodexBridgeHandle, PromptMode}; +use crate::bridge::codex::{CodexBridgeHandle, PromptMode}; +use crate::bridge::opencode as opencode_bridge; use crate::error::{ReplayError, ReplayErrorKind, ResultExt}; use crate::io::{atomic_write, atomic_write_json, canonicalize, read_regular_file, sha256}; use crate::journal::Journal; @@ -25,7 +26,6 @@ use crate::model::{ AgentKind, FreshObservation, PlaybackRequest, ReplayMode, ReplayOutcome, ReplayPlan, ToolBatch, ToolCall, }; -use crate::opencode_bridge; use crate::process::{ProcessSpec, run_process}; #[derive(Debug, Clone, Copy)] diff --git a/crates/persisting-replay/src/claude_bridge.rs b/crates/persisting-replay/src/bridge/claude.rs similarity index 99% rename from crates/persisting-replay/src/claude_bridge.rs rename to crates/persisting-replay/src/bridge/claude.rs index 79dd116a..ce26b696 100644 --- a/crates/persisting-replay/src/claude_bridge.rs +++ b/crates/persisting-replay/src/bridge/claude.rs @@ -15,7 +15,7 @@ use axum::routing::{get, post}; use serde_json::{Map, Value, json}; use tokio::sync::{Notify, oneshot}; -use crate::claude_resume::{ResumeTransportManifest, clean_resume_transport_envelope}; +use crate::bridge::claude_resume::{ResumeTransportManifest, clean_resume_transport_envelope}; use crate::error::{ReplayError, ReplayErrorKind, ResultExt}; const BRIDGE_VERSION: &str = "sandbox-replay-anthropic-openai-bridge/1"; diff --git a/crates/persisting-replay/src/claude_resume.rs b/crates/persisting-replay/src/bridge/claude_resume.rs similarity index 100% rename from crates/persisting-replay/src/claude_resume.rs rename to crates/persisting-replay/src/bridge/claude_resume.rs diff --git a/crates/persisting-replay/src/codex_bridge.rs b/crates/persisting-replay/src/bridge/codex.rs similarity index 100% rename from crates/persisting-replay/src/codex_bridge.rs rename to crates/persisting-replay/src/bridge/codex.rs diff --git a/crates/persisting-replay/src/bridge/mod.rs b/crates/persisting-replay/src/bridge/mod.rs new file mode 100644 index 00000000..fc4b3e7a --- /dev/null +++ b/crates/persisting-replay/src/bridge/mod.rs @@ -0,0 +1,6 @@ +//! Agent-specific replay protocol bridges and resume transport validation. + +pub(crate) mod claude; +pub(crate) mod claude_resume; +pub(crate) mod codex; +pub(crate) mod opencode; diff --git a/crates/persisting-replay/src/opencode_bridge.rs b/crates/persisting-replay/src/bridge/opencode.rs similarity index 100% rename from crates/persisting-replay/src/opencode_bridge.rs rename to crates/persisting-replay/src/bridge/opencode.rs diff --git a/crates/persisting-replay/src/lib.rs b/crates/persisting-replay/src/lib.rs index 664e2497..5f43b0c9 100644 --- a/crates/persisting-replay/src/lib.rs +++ b/crates/persisting-replay/src/lib.rs @@ -3,12 +3,10 @@ //! The default execution model assumes pVisor is already running inside a //! fresh sandbox. Replay therefore touches only the selected workspace and //! connects live Agents directly to their configured model endpoint. -//! Claude Code alone uses a replay-local protocol bridge to remove Resume Transport messages. +//! Agent-specific bridges handle replay and continuation protocol adjustments. mod adapter; -mod claude_bridge; -mod claude_resume; -mod codex_bridge; +mod bridge; mod comparison; mod config; mod engine; @@ -16,7 +14,6 @@ mod error; mod io; mod journal; mod model; -pub(crate) mod opencode_bridge; mod process; pub use config::{ diff --git a/crates/persisting-replay/src/process.rs b/crates/persisting-replay/src/process.rs index 302dfaa5..88cea965 100644 --- a/crates/persisting-replay/src/process.rs +++ b/crates/persisting-replay/src/process.rs @@ -12,7 +12,6 @@ use std::time::{Duration, Instant}; use crate::error::{ReplayError, ReplayErrorKind, ResultExt}; -#[allow(dead_code)] pub(crate) struct ProcessSpec { pub command: Command, pub stdin: Option>, @@ -38,13 +37,10 @@ pub(crate) struct ProcessSpec { pub log_path: PathBuf, } -#[allow(dead_code)] pub(crate) struct ProcessOutput { pub status: ExitStatus, pub stdout_tail: Vec, pub stderr_tail: Vec, - pub stdout_bytes: u64, - pub stderr_bytes: u64, pub stdout_truncated: bool, pub stderr_truncated: bool, pub timed_out: bool, @@ -58,7 +54,6 @@ struct StreamCapture { log_error: Option, } -#[allow(dead_code)] pub(crate) fn run_process(mut spec: ProcessSpec) -> Result { let log = owner_only_log(&spec.log_path)?; let log = Arc::new(Mutex::new(log)); @@ -271,8 +266,6 @@ pub(crate) fn run_process(mut spec: ProcessSpec) -> Result stderr.tail.len() as u64, stdout_tail: stdout.tail, stderr_tail: stderr.tail, - stdout_bytes: stdout.total, - stderr_bytes: stderr.total, timed_out, step_limited, background_cleanup, @@ -540,7 +533,6 @@ mod tests { let output = run_process(shell_spec("yes x | head -c 8388608", &log_path)).unwrap(); assert!(output.status.success()); - assert_eq!(output.stdout_bytes, 8 * 1024 * 1024); assert!(output.stdout_truncated); assert_eq!(output.stdout_tail.len(), 64 * 1024); assert_eq!(std::fs::metadata(log_path).unwrap().len(), 8 * 1024 * 1024); diff --git a/docs/shared-image-cache.md b/docs/shared-image-cache.md index b6b500dc..0a2984bf 100644 --- a/docs/shared-image-cache.md +++ b/docs/shared-image-cache.md @@ -6,6 +6,32 @@ on the server, then addressed by its resolved platform manifest SHA-256 digest. The existing store performs blob verification, layer application and whiteout handling. File queries do not contact the registry. +## Code layout + +The implementation lives in `crates/persisting-pvisor/src/image/cache/`: + +```text +cache/ +├── mod.rs # Public entry points and module wiring +├── cli.rs # pvisor cache subcommands +├── protocol.rs # Request/response types, framing, content hash +├── transport.rs # Unix/TCP endpoints, streams, timeouts +├── client.rs # Service discovery and validated requests +├── server.rs # Authentication, worker queues, confined file access +├── server/ +│ ├── metadata.rs # Server-side metadata and directory LRU caches +│ └── tests.rs # Protocol, confinement, and client/server tests +├── lazy.rs # FUSE mounting, block and client metadata caches +├── lazy/ +│ └── tests.rs # Lazy filesystem and cache reuse tests +└── progress.rs # Image totals and loading/download progress +``` + +`image/oci.rs` owns registry resolution, prepared-image records, blob verification, +and layer extraction. Both local loading and the cache server use `ImageStore`. +External callers keep using the `cache` module's exported API; internal protocol +and transport helpers stay private to the cache subsystem. + ## Usage ```sh @@ -14,6 +40,8 @@ pvisor cache serve # Terminal 2: use the same default socket pvisor cache prepare alpine:latest +# Force a registry refresh even within the five-minute tag cache window: +pvisor cache prepare alpine:latest --refresh # Copy the digest from the JSON result: pvisor cache list sha256:YOUR_MANIFEST_DIGEST pvisor cache stat sha256:YOUR_MANIFEST_DIGEST etc/os-release @@ -33,7 +61,7 @@ existing OCI store. It does not auto-start. ## Automatic VM lazy loading -When `pvisor --rootfs image=IMAGE -- COMMAND` prepares an OCI image for a VM, +When `pvisor run --vm --rootfs image=IMAGE -- COMMAND` prepares an OCI image, it probes the default socket with a two-second `ping` handshake. A live compatible server selects lazy loading automatically; a missing socket or refused connection (stale socket) uses the existing local OCI preparation path. Authentication, @@ -56,9 +84,14 @@ immutable; in-place edits below the root are not supported. Content is fetched i blocks into `/persisting/pvisor/blocks///`, with per-file/per-block keys. On macOS, `` is `~/Library/Caches`; on Linux it is `$XDG_CACHE_HOME`, ordinarily `~/.cache`. This block cache is -independent of `--image-store` and `PERSISTING_PVISOR_IMAGE_STORE`. Blocks are checksum-verified, published atomically and -shared across local processes using file locks. Cache hits are verified before -use; corrupted blocks are fetched again. No sparse placeholder files are exposed. +independent of `--image-store` and `PERSISTING_PVISOR_IMAGE_STORE`. Small files occupy one unpadded block; large files fetch only accessed blocks. +Each mount retains verified content in a file-keyed memory cache, capped at +64 MiB and 4096 blocks with FIFO eviction. Hot reads copy only the requested +slice and do not reopen or rehash the disk block. On a memory miss, disk blocks +are verified again; disk corruption does not change bytes already verified and +retained in memory. New blocks are checksum-verified, published atomically and +shared across local processes using file locks; corrupt disk blocks are fetched +again. No sparse placeholder files are exposed. Normal kernel readahead may fetch adjacent bytes, and copy-up may read a whole individual file. The client does not extract the full image. @@ -72,7 +105,7 @@ The server still fully prepares an uncached image before answering `prepare`. This is client-side lazy loading, not lazy OCI layer extraction on the server. The FUSE adapter and existing virtio-fs worker currently process requests synchronously: a cache miss can delay unrelated filesystem requests. No explicit -vCPU pause is used. Cache quotas/eviction, original OCI xattrs and asynchronous +vCPU pause is used. Disk cache quotas/eviction, original OCI xattrs and asynchronous virtio-fs completions are not added by this implementation. The public Rust client is `persisting_pvisor::cache::CacheClient::from_env()`. @@ -116,7 +149,8 @@ bars abbreviate them to `C / X / T`; Overview shows all three rows with exact byte counts. The Log panel records every verified block transfer with its file path, byte count and cumulative run totals. The status bar and Overview also show local cache reads: distinct file paths -and cumulative verified block bytes, including repeated reads. These are separate +and cumulative bytes served from verified disk or memory cache, including repeated +reads. Only the requested slices count, not the whole 1 MiB blocks read internally. These are separate from downloads; they exclude reads satisfied by the host or guest kernel page cache. Each file's first local cache hit appears in Log as `no download`. Directory listings only fetch metadata and do not count as content reads. Image startup and I/O diagnostics @@ -150,23 +184,36 @@ Paths and directory names are JSON arrays of Unix filename bytes, preserving non-UTF-8 filenames. Paths are relative to the image root; an empty path means the root directory. Absolute paths, parent traversal and NUL are rejected. Symlinks are returned as metadata, never followed by server path resolution. -A future filesystem client must resolve guest symlinks within the guest tree. +Guest filesystem traversal resolves symlinks within the guest tree. | `op` | Fields | Response `status` | | --- | --- | --- | | `ping` | none | `ready` (protocol v1) | -| `prepare` | `image`, `architecture` (`amd64` or `arm64`) | `prepared`: `digest`, `architecture`, `env`, `entrypoint`, `cmd`, optional `totals` (`files`, `bytes`), optional `metadata_generation` | -| `list` | `digest`, `path`, `offset` (entry index, start at 0) | `entries`: sorted `names` (up to 256), `next_offset` (null when done) | +| `prepare` | `image`, `architecture` (`amd64` or `arm64`), optional `refresh` (default false) | `prepared`: `digest`, `architecture`, `env`, `entrypoint`, `cmd`, optional `totals` (`files`, `bytes`), optional `metadata_generation` | +| `list` | `digest`, `path`, `offset` (entry index, start at 0) | `entries`: sorted `names`, optional aligned `metadata` array, `next_offset` (null when done) | | `stat` | `digest`, `path` | `metadata`: `kind`, `size`, `mode`, `uid`, `gid`, `inode`, `nlink`, `mtime`, `mtime_nsec`, `target` | | `read` | `digest`, `path`, `offset` (byte offset), `length` (1..1048576) | `data`: `length`, `sha256`, followed by raw bytes | `prepare` requests Linux images for the client's architecture, independent of -the server architecture. It resolves mutable tags on each call; the returned -platform manifest digest pins subsequent requests. Preparation can populate an -uncached image and uses existing per-image extraction locks. It may still need -the registry even when an extracted root already exists. `read`, `stat` and +the server architecture. Successful prepared-image records are persisted under +`/metadata/prepared-v1/`, including the platform digest and launch +configuration. Mutable tags reuse records for five minutes; immutable digest +records do not expire while their extracted root exists. `cache prepare IMAGE +--refresh` (protocol `refresh: true`) forces registry resolution. Failed refreshes +return an error and preserve the previous record; registry requests have a +10-second connect timeout and a 300-second total timeout. Expired tags do not silently +fall back to stale data. Missing/corrupt records or missing roots are prepared +again. A per-reference-and-architecture lock covers resolution and preparation, +so concurrent requests recheck and reuse the first successful result. Preparation +can populate an uncached image and retains the existing per-digest extraction lock. `read`, `stat` and `list` require an already prepared digest; they never implicitly pull an image. +Directory pages include the same attributes as `stat`, avoiding a separate +request per child. Pages contain at most 256 entries and shrink to fit the JSON +frame limit, including long byte-array names and symlink targets. Names-only +responses from older servers remain supported through individual `stat` +requests. Persisted pages retain their attributes across mounts. + `kind` is `file`, `directory`, `symlink` or `special`. `mode` includes Unix type and permission bits. `target` contains symlink bytes or null. Attributes reflect the extracted server filesystem; v1 does not reconstruct original tar ownership, @@ -190,13 +237,17 @@ zero-filled content. Error messages are explanatory, not machine-stable. The server shares in-memory caches for up to 4096 stat responses and 128 sorted directory indexes across requests. Directory pagination reuses the same index -instead of rescanning and sorting on every page. Each cache clears at its entry -ceiling and is rebuilt lazily after a server restart. These caches cover metadata, -not mutable tag resolution; `prepare` still contacts the registry. - -The server has 16 workers and at most 16 queued connections. Excess connections -are closed; clients may retry. Socket reads/writes have a 300-second inactivity -timeout, and TCP connects have a 10-second timeout. Long-running preparation may +instead of rescanning and sorting on every page. At capacity, LRU eviction removes +one entry instead of clearing the cache. Filesystem I/O runs outside the cache +lock; simultaneous misses may duplicate a read without blocking unrelated hits. +These in-memory caches are rebuilt lazily after a server restart. + +The server has 16 request/file workers and at most 16 queued connections. Excess +connections are closed; clients may retry. Authenticated prepare requests move +to a separate pool of two workers with a 16-request queue; when full, the server +returns an explicit busy error. Registry waits and extraction do not occupy file +workers. Incoming request reads have a five-second inactivity timeout; response +reads/writes retain a 300-second timeout, and TCP connects have a 10-second timeout. Long-running preparation may outlive a disconnected client; retrying is safe. Shutdown does not cancel individual OCI downloads gracefully. Registry download limits and cache eviction remain those of the existing image store; v1 adds neither quotas nor eviction. diff --git a/docs/src/en/development/engineering.md b/docs/src/en/development/engineering.md index 94eb7353..9cdeb17e 100644 --- a/docs/src/en/development/engineering.md +++ b/docs/src/en/development/engineering.md @@ -3,6 +3,100 @@ Run commands from the repository root. `just` lists the supported tasks; each workflow has one entry point. +## Repository and code ownership + +The Cargo workspace is organized by product responsibility. The Python `pvisor/` +package launches the packaged Rust binary; it is not a second runtime. + +| Location | Owns | +|---|---| +| `crates/persisting-pvisor/` | CLI, run orchestration, executors, image preparation and cache service | +| `crates/persisting-control/` | Shared contracts, policies, AgentCtl messages, IR and event schemas | +| `crates/persisting-gateway/` | Agent protocol forwarding, conversion, capture and projection | +| `crates/persisting-overlay-core/` | FUSE-independent overlay operations and file access enforcement | +| `crates/persisting-overlayfs/` | FUSE adapter, mounts and optional Jujutsu upper storage | +| `crates/persisting-overlaynet/` | Egress policy, HTTP proxy and VM virtio-net data plane | +| `crates/persisting-replay/` | Replay planning, native agent adapters and continuation bridges | +| `pvisor/`, `setup.py`, `scripts/packaging/` | Python launcher and wheel assembly | +| `crates/*/tests/` | Rust integration tests; unit tests stay with their owning module | +| `tests/` | Python packaging and repository workflow tests | +| `examples/`, `benchmark/` | Runnable product scenarios and performance measurements | +| `scripts/ci/` | CI-specific validation and smoke runners | +| `docs/src/en/`, `docs/src/zh/` | Paired documentation pages; `docs/site/` is generated output | +| `vendor/` | Patched third-party dependencies; keep product orchestration in `crates/` | + +The internal workspace dependencies are: + +```text +pvisor ──> control, gateway, overlaynet, overlayfs, overlay-core, replay +gateway ──> control, overlaynet +overlaynet ──> control +overlayfs ──> control, overlay-core +overlay-core ──> control +control, replay ──> no other workspace crate +``` + +### pVisor source modules + +```text +src/ +├── lib.rs # Stable embedding exports +├── bin/pvisor.rs # Binary entry point +├── cli/ # Arguments, commands, agent presets and terminal UI +├── config.rs # Runtime and executor configuration +├── core.rs, trace.rs # Operation-chain execution and trace journal +├── diagnostics.rs # Shared host logs; frontend selects the destination +├── executor/ +│ ├── mod.rs # RunExecutor and AttemptContext +│ ├── process.rs # Host process executor +│ ├── container.rs # Container executor +│ ├── sandbox.rs # Host OS isolation and internal sandbox entry point +│ ├── artifact.rs # Guest-compatible executable resolution +│ ├── delegated.rs # Delegated run spec/result hand-off +│ └── vm/ # libkrun executor and firmware acquisition +├── image/ +│ ├── oci.rs # Registry, prepared records, blobs and extraction +│ └── cache/ # Cache CLI, protocol, server, client and lazy FUSE +├── runtime/ +│ ├── run.rs # PVisor API and run lifecycle +│ ├── agentctl.rs # Per-run cooperative control server +│ ├── event.rs # Runtime event publication +│ ├── bundle.rs # Durable review summary +│ ├── checkpoint.rs # Logical checkpoint and restore +│ ├── registry.rs # Run identity, leases and local control endpoint +│ ├── attempt.rs # Per-attempt driver resources and teardown +│ ├── supervisor.rs # Capability checks and driver coordination +│ ├── plan.rs # Typed run-plan construction +│ ├── implant.rs # Runtime environment injection +│ ├── overlay.rs # Staging, review, apply/discard and recovery +│ └── zcode.rs # Process compatibility policy +└── util.rs # Small shared file/time helpers +``` + +Keep CLI parsing and presentation in `cli/`, concrete execution in `executor/`, +and run-scoped resource ownership in `runtime/`. Firmware belongs to the VM +executor, while OCI preparation belongs to `image/` and is shared by direct +loads and the cache server. Durable bundles/checkpoints live beside the run +registry rather than beside individual backends. Root exports such as +`PVisor`, `ProcessExecutor`, `cache` and the internal `sandbox` entry point retain +their existing import paths. + +In replay, `adapter/` owns native trajectory planning and agent launch choices; +`bridge/` owns the Claude, Codex and OpenCode protocol bridges and Claude resume +transport validation. Shared execution and journaling remain at the crate root. + +### Boundaries to keep improving + +This organization does not claim strict one-way layering inside pVisor: +`AttemptContext` still carries runtime-owned attachments, and runtime overlay +configuration still uses Gateway configuration types. Those need contract +changes, not just file moves. `cli/run.rs`, `runtime/overlay.rs`, and the larger +agent adapters also mix several stages; split them around actual lifecycle or +protocol boundaries when changing that behavior, rather than by line count. +Do not add another crate solely to shorten these files. Keep public exports +stable and run the affected package tests after internal moves. + + ## Contributor commands | Command | What it does | diff --git a/docs/src/en/reference/cli.md b/docs/src/en/reference/cli.md index 17e13a01..c869fd5c 100644 --- a/docs/src/en/reference/cli.md +++ b/docs/src/en/reference/cli.md @@ -692,4 +692,4 @@ lower with persistent 1 MiB block caching when a compatible server is available. A missing/stale default socket retains local OCI preparation. An explicitly configured server must work; `PERSISTING_PVISOR_CACHE_SERVER=off` forces local preparation. Explicit rootfs directories and native containers are unchanged. -See the [protocol and remote-access guide](../../../shared-image-cache.md). +See the [protocol and remote-access guide](https://github.com/DeepLink-org/Persisting/blob/main/docs/shared-image-cache.md). diff --git a/docs/src/zh/development/engineering.md b/docs/src/zh/development/engineering.md index f3c57ace..06ebdde7 100644 --- a/docs/src/zh/development/engineering.md +++ b/docs/src/zh/development/engineering.md @@ -2,6 +2,96 @@ 从仓库根目录运行命令。`just` 列出支持的任务,每种工作流保留一个入口。 +## 仓库结构与代码归属 + +Cargo workspace 按产品职责划分。Python `pvisor/` 只负责启动随包分发的 Rust +二进制,不是另一套运行时实现。 + +| 目录 | 职责 | +|---|---| +| `crates/persisting-pvisor/` | CLI、运行编排、执行器、镜像准备和缓存服务 | +| `crates/persisting-control/` | 共享契约、策略、AgentCtl 消息、IR 和事件 schema | +| `crates/persisting-gateway/` | Agent 协议转发、转换、采集与投影 | +| `crates/persisting-overlay-core/` | 不依赖 FUSE 的 OverlayFS 操作和文件访问控制 | +| `crates/persisting-overlayfs/` | FUSE 适配、挂载及可选的 Jujutsu 上层存储 | +| `crates/persisting-overlaynet/` | 出站策略、HTTP 代理和 VM virtio-net 数据通路 | +| `crates/persisting-replay/` | 回放规划、原生 Agent 适配器和续跑协议桥 | +| `pvisor/`、`setup.py`、`scripts/packaging/` | Python 启动器和 wheel 打包 | +| `crates/*/tests/` | Rust 集成测试;单元测试跟随所属模块 | +| `tests/` | Python 打包和仓库工作流测试 | +| `examples/`、`benchmark/` | 可运行的产品场景和性能测量 | +| `scripts/ci/` | CI 检查及冒烟测试入口 | +| `docs/src/en/`、`docs/src/zh/` | 成对维护的文档;`docs/site/` 是生成产物 | +| `vendor/` | 有补丁的第三方依赖;产品编排逻辑放在 `crates/` | + +workspace 内的实际依赖关系: + +```text +pvisor ──> control, gateway, overlaynet, overlayfs, overlay-core, replay +gateway ──> control, overlaynet +overlaynet ──> control +overlayfs ──> control, overlay-core +overlay-core ──> control +control, replay ──> 不依赖其他 workspace crate +``` + +### pVisor 源码模块 + +```text +src/ +├── lib.rs # 稳定的嵌入接口导出 +├── bin/pvisor.rs # 二进制入口 +├── cli/ # 参数、命令、Agent 预设和终端 UI +├── config.rs # 运行时与执行器配置 +├── core.rs, trace.rs # 操作链执行与 trace journal +├── diagnostics.rs # 共享宿主日志,前端选择输出位置 +├── executor/ +│ ├── mod.rs # RunExecutor 和 AttemptContext +│ ├── process.rs # 宿主进程执行器 +│ ├── container.rs # 容器执行器 +│ ├── sandbox.rs # 宿主 OS 隔离及内部 sandbox 入口 +│ ├── artifact.rs # 适配 guest 的可执行文件解析 +│ ├── delegated.rs # 委派执行的 spec/result 交接 +│ └── vm/ # libkrun 执行器和固件获取 +├── image/ +│ ├── oci.rs # Registry、准备记录、blob 和解包 +│ └── cache/ # 缓存 CLI、协议、服务端、客户端及懒加载 FUSE +├── runtime/ +│ ├── run.rs # PVisor API 和运行生命周期 +│ ├── agentctl.rs # 每次运行的协作控制服务 +│ ├── event.rs # 运行事件发布 +│ ├── bundle.rs # 持久化审查摘要 +│ ├── checkpoint.rs # 逻辑检查点与恢复 +│ ├── registry.rs # Run 身份、租约和本地控制端点 +│ ├── attempt.rs # 每次尝试的驱动资源与清理 +│ ├── supervisor.rs # 能力检查与驱动协调 +│ ├── plan.rs # 类型化运行计划构造 +│ ├── implant.rs # 运行环境注入 +│ ├── overlay.rs # 暂存、审查、应用/丢弃和恢复 +│ └── zcode.rs # 进程兼容策略 +└── util.rs # 少量共享文件与时间工具 +``` + +CLI 参数与展示留在 `cli/`,具体执行机制归 `executor/`,Run 资源所有权归 +`runtime/`。固件属于 VM 执行器;OCI 准备属于 `image/`,供直接加载和缓存 +服务共用。Bundle 和检查点与运行记录放在一起,不归某个执行后端。 +`PVisor`、`ProcessExecutor`、`cache` 以及内部 `sandbox` 入口等根级导出保留 +原有导入路径。 + +replay 中,`adapter/` 负责原生轨迹规划和 Agent 启动选择;`bridge/` 负责 +Claude、Codex、OpenCode 协议桥及 Claude resume transport 校验。 +共享执行和 journal 仍在 crate 根目录。 + +### 仍需逐步改善的边界 + +目录整理不代表 pVisor 内部已实现严格单向分层:`AttemptContext` 仍携带 +运行时资源附件,运行时 Overlay 配置仍使用 Gateway 的配置类型。这些需要 +修改契约,不能只靠移动文件解决。`cli/run.rs`、`runtime/overlay.rs` 和较大的 +Agent 适配器仍包含多个阶段;后续修改相关行为时,应按生命周期或协议边界 +拆分,而不是按行数切割。不要仅为缩短文件新增 crate;移动内部模块时保持 +对外导出稳定,并运行受影响包的测试。 + + ## 贡献者命令 | 命令 | 作用 | diff --git a/docs/src/zh/reference/cli.md b/docs/src/zh/reference/cli.md index aa10f133..c3dee68c 100644 --- a/docs/src/zh/reference/cli.md +++ b/docs/src/zh/reference/cli.md @@ -619,4 +619,4 @@ VM 镜像启动会自动探测默认 socket;服务可用时,将远程镜像 以 1 MiB 数据块按需读取并持久缓存。默认 socket 不存在或已失效时走本地 OCI 准备。 显式指定服务端后连接失败会报错;`PERSISTING_PVISOR_CACHE_SERVER=off` 强制本地准备。 显式 rootfs 目录和原生 container executor 保持原有行为。 -完整协议、限制和 SSH 远程访问方式见 [共享镜像缓存协议](../../../shared-image-cache.md)。 +完整协议、限制和 SSH 远程访问方式见 [共享镜像缓存协议](https://github.com/DeepLink-org/Persisting/blob/main/docs/shared-image-cache.md)。 From 47bdfde4b3e3293fa1cfba1f48da194855355201 Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 29 Sep 2026 23:49:21 +0800 Subject: [PATCH 3/9] Refactor hex encoding into shared util Remove the test-only DelegatedRunFiles::new constructor and have tests call new_with_stdio directly. Also drop the redundant PERSISTING_AGENTCTL_ prefix check in the env filter. --- .../src/executor/container.rs | 2 +- .../src/executor/delegated.rs | 18 ++++++------------ .../src/executor/vm/firmware.rs | 12 +----------- .../src/image/cache/protocol.rs | 5 +---- 4 files changed, 9 insertions(+), 28 deletions(-) diff --git a/crates/persisting-pvisor/src/executor/container.rs b/crates/persisting-pvisor/src/executor/container.rs index 26602f72..ba58a5ea 100644 --- a/crates/persisting-pvisor/src/executor/container.rs +++ b/crates/persisting-pvisor/src/executor/container.rs @@ -736,7 +736,7 @@ mod tests { let RunInvocation::Process(invocation) = &mut spec.invocation; invocation.cwd = Some(cwd.display().to_string()); invocation.inherit_env = false; - let files = DelegatedRunFiles::new(&spec).unwrap(); + let files = DelegatedRunFiles::new_with_stdio(&spec, false).unwrap(); let command = executor .build_command( &spec, diff --git a/crates/persisting-pvisor/src/executor/delegated.rs b/crates/persisting-pvisor/src/executor/delegated.rs index ddecb8e3..aea29b01 100644 --- a/crates/persisting-pvisor/src/executor/delegated.rs +++ b/crates/persisting-pvisor/src/executor/delegated.rs @@ -9,7 +9,6 @@ pub(crate) const RESULT_FILENAME: &str = "run-result.json"; #[derive(Debug, serde::Serialize, serde::Deserialize)] pub(crate) struct DelegatedRunOutput { pub(crate) result: RunResult, - #[serde(alias = "agentctl")] pub(crate) agentctl: crate::AgentCtlSnapshot, } @@ -20,12 +19,7 @@ pub(crate) struct DelegatedRunFiles { } impl DelegatedRunFiles { - #[cfg(test)] - pub(crate) fn new(spec: &RunSpec) -> anyhow::Result { - Self::new_with_stdio(spec, false) - } - - /// Create delegated files while forcing the injected pVisor to use pipes. + /// Create delegated files, optionally capturing the injected pVisor's output. /// The outer transport owns the real terminal; inheriting it in the nested /// process makes rootless OCI runs attempt tty process-group operations. pub(crate) fn new_with_stdio(spec: &RunSpec, capture: bool) -> anyhow::Result { @@ -37,9 +31,9 @@ impl DelegatedRunFiles { let mut delegated = spec.clone(); delegated.metadata.remove("pvisor.executor"); let RunInvocation::Process(process) = &mut delegated.invocation; - process.env.retain(|key, _| { - !key.starts_with("PERSISTING_AGENTCTL_") && !key.starts_with("PERSISTING_AGENTCTL_") - }); + process + .env + .retain(|key, _| !key.starts_with("PERSISTING_AGENTCTL_")); if capture { // pVisor v1 does not support captured stdin. Null stdin also // prevents the nested host executor from attempting tty control. @@ -105,7 +99,7 @@ mod tests { use super::*; #[test] - fn delegated_spec_drops_host_agentctl_and_normalizes_result_identity() { + fn delegated_spec_drops_host_agentctl() { let mut spec = RunSpec::process("run-one", "agent", "true"); let RunInvocation::Process(process) = &mut spec.invocation; process.env.insert( @@ -113,7 +107,7 @@ mod tests { "/tmp/host.sock".into(), ); process.env.insert("KEEP".into(), "yes".into()); - let files = DelegatedRunFiles::new(&spec).unwrap(); + let files = DelegatedRunFiles::new_with_stdio(&spec, false).unwrap(); let delegated: RunSpec = serde_json::from_slice(&std::fs::read(&files.spec_path).unwrap()).unwrap(); let RunInvocation::Process(process) = delegated.invocation; diff --git a/crates/persisting-pvisor/src/executor/vm/firmware.rs b/crates/persisting-pvisor/src/executor/vm/firmware.rs index ec12628c..f223200e 100644 --- a/crates/persisting-pvisor/src/executor/vm/firmware.rs +++ b/crates/persisting-pvisor/src/executor/vm/firmware.rs @@ -143,7 +143,7 @@ fn verify_archive(archive: &[u8], expected: &str) -> anyhow::Result<()> { "libkrunfw archive exceeds {} bytes", MAX_ARCHIVE_BYTES ); - let actual = encode_hex(&Sha256::digest(archive)); + let actual = crate::util::encode_hex(&Sha256::digest(archive)); anyhow::ensure!( actual == expected, "libkrunfw archive digest mismatch: expected {expected}, got {actual}" @@ -230,16 +230,6 @@ fn build_platform_firmware(source: &Path, destination: &Path) -> anyhow::Result< Ok(()) } -fn encode_hex(bytes: &[u8]) -> String { - const HEX: &[u8; 16] = b"0123456789abcdef"; - let mut output = String::with_capacity(bytes.len() * 2); - for byte in bytes { - output.push(HEX[(byte >> 4) as usize] as char); - output.push(HEX[(byte & 0x0f) as usize] as char); - } - output -} - #[cfg(test)] mod tests { use super::*; diff --git a/crates/persisting-pvisor/src/image/cache/protocol.rs b/crates/persisting-pvisor/src/image/cache/protocol.rs index 601905fc..d09313a0 100644 --- a/crates/persisting-pvisor/src/image/cache/protocol.rs +++ b/crates/persisting-pvisor/src/image/cache/protocol.rs @@ -112,8 +112,5 @@ pub(super) fn write_frame(stream: &mut impl Write, value: &impl Serialize) -> an } pub(super) fn hash(bytes: &[u8]) -> String { - format!( - "sha256:{}", - crate::image::oci::encode_hex(&Sha256::digest(bytes)) - ) + format!("sha256:{}", crate::util::encode_hex(&Sha256::digest(bytes))) } From 1d06c2d268f882b50e65a6f08b2af5968099b3c6 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 30 Sep 2026 10:33:57 +0800 Subject: [PATCH 4/9] Add develop branch to CI triggers Update CI, docs, and benchmark workflows to run on pushes and pull requests to develop in addition to main. Sync the engineering docs in both languages with the new triggers. --- .github/workflows/ci.yml | 4 ++-- .github/workflows/docs.yml | 4 ++-- .github/workflows/pvisor-benchmark.yml | 4 ++-- docs/src/en/development/engineering.md | 2 +- docs/src/zh/development/engineering.md | 2 +- 5 files changed, 8 insertions(+), 8 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 575e0a42..46a20690 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,9 +2,9 @@ name: CI on: push: - branches: [main] + branches: [main, develop] pull_request: - branches: [main] + branches: [main, develop] concurrency: group: ci-${{ github.workflow }}-${{ github.ref }} diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index e4ce6b6b..b936fde3 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -2,10 +2,10 @@ name: Documentation on: push: - branches: [main] + branches: [main, develop] paths: ['docs/**', 'scripts/*docs*.py', 'justfile', '.github/workflows/docs.yml'] pull_request: - branches: [main] + branches: [main, develop] paths: ['docs/**', 'scripts/*docs*.py', 'justfile', '.github/workflows/docs.yml'] workflow_dispatch: diff --git a/.github/workflows/pvisor-benchmark.yml b/.github/workflows/pvisor-benchmark.yml index ab36bc87..71136ef8 100644 --- a/.github/workflows/pvisor-benchmark.yml +++ b/.github/workflows/pvisor-benchmark.yml @@ -2,7 +2,7 @@ name: pVisor Benchmark on: push: - branches: [main] + branches: [main, develop] paths: - 'Cargo.*' - '.cargo/**' @@ -14,7 +14,7 @@ on: - '.github/actions/setup-build-env/**' - '.github/workflows/pvisor-benchmark.yml' pull_request: - branches: [main] + branches: [main, develop] paths: - 'Cargo.*' - '.cargo/**' diff --git a/docs/src/en/development/engineering.md b/docs/src/en/development/engineering.md index 9cdeb17e..47648f07 100644 --- a/docs/src/en/development/engineering.md +++ b/docs/src/en/development/engineering.md @@ -133,7 +133,7 @@ for example `cargo nextest run --locked -p persisting-gateway --test llm_fixture | Workflow | Trigger and responsibility | |---|---| -| CI | Push/PR to `main`: formatting, Clippy, actionlint, Python tests, benchmark harness tests, Rust tests, documented cases, and examples | +| CI | Push/PR to `main` and `develop`: formatting, Clippy, actionlint, Python tests, benchmark harness tests, Rust tests, documented cases, and examples | | Documentation | Documentation changes: build both languages and check links; only the upstream `main` branch deploys Pages | | pVisor Benchmark | Runtime/build/benchmark changes: compare candidate with the PR base or previous commit and upload reports | | Nightly Build | Daily or manual on `main`: build and verify both platform wheels, then update the nightly release | diff --git a/docs/src/zh/development/engineering.md b/docs/src/zh/development/engineering.md index 06ebdde7..fefdf8d2 100644 --- a/docs/src/zh/development/engineering.md +++ b/docs/src/zh/development/engineering.md @@ -128,7 +128,7 @@ nextest 不运行 doctest;需要时使用 `cargo test --doc -p `。 | 工作流 | 触发条件与职责 | |---|---| -| CI | 面向 `main` 的 push/PR:格式、Clippy、actionlint、Python 测试、基准工具测试、Rust 测试、文档用例与示例 | +| CI | 面向 `main` 和 `develop` 的 push/PR:格式、Clippy、actionlint、Python 测试、基准工具测试、Rust 测试、文档用例与示例 | | Documentation | 文档变更:双语构建与链接检查;仅上游仓库的 `main` 部署 Pages | | pVisor Benchmark | 运行时、构建或基准变更:与 PR 基线或前一提交比较并上传报告 | | Nightly Build | 每日或在 `main` 手动触发:构建、校验双平台 wheel,更新 nightly release | From 1f7d3e76e0275d2d3cac568871cffffe72e8da9f Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 30 Sep 2026 11:09:55 +0800 Subject: [PATCH 5/9] Deduplicate JSON writing and hashing helpers Move write_private_json into util and route delegated, VM, and run-result writes through the shared atomic replacement path. Reuse the replay sha256 helper in claude_resume. Also harden the seatbelt proxy test against Xcode's python3 launcher by resolving the real interpreter before sandboxing. --- crates/persisting-pvisor/src/cli/run.rs | 2 +- .../src/executor/delegated.rs | 31 ++----------------- .../persisting-pvisor/src/executor/sandbox.rs | 22 +++++++++++-- .../src/executor/vm/supported.rs | 7 +---- .../persisting-pvisor/src/image/cache/lazy.rs | 2 ++ crates/persisting-pvisor/src/util.rs | 30 ++++++++++++++++++ .../src/bridge/claude_resume.rs | 13 ++------ 7 files changed, 59 insertions(+), 48 deletions(-) diff --git a/crates/persisting-pvisor/src/cli/run.rs b/crates/persisting-pvisor/src/cli/run.rs index 48a0fa1a..77a5897b 100644 --- a/crates/persisting-pvisor/src/cli/run.rs +++ b/crates/persisting-pvisor/src/cli/run.rs @@ -963,7 +963,7 @@ async fn run_prepared_spec(args: RunArgs) -> anyhow::Result { agentctl: agentctl.snapshot(), result, }; - let write_result = crate::executor::delegated::write_result(&result_path, &output) + let write_result = crate::util::write_private_json(&result_path, &output) .with_context(|| format!("write delegated RunResult to {}", result_path.display())); let cleanup_result = stage_guard .as_mut() diff --git a/crates/persisting-pvisor/src/executor/delegated.rs b/crates/persisting-pvisor/src/executor/delegated.rs index aea29b01..4d8f5359 100644 --- a/crates/persisting-pvisor/src/executor/delegated.rs +++ b/crates/persisting-pvisor/src/executor/delegated.rs @@ -1,7 +1,8 @@ //! Files and result hand-off for a pVisor delegated through Docker or KVM. +use crate::util::write_private_json; use persisting_control::{AttemptId, RunInvocation, RunResult, RunSpec}; -use std::path::{Path, PathBuf}; +use std::path::PathBuf; pub(crate) const SPEC_FILENAME: &str = "run-spec.json"; pub(crate) const RESULT_FILENAME: &str = "run-result.json"; @@ -66,34 +67,6 @@ impl DelegatedRunFiles { } } -pub(crate) fn write_result(path: &Path, output: &DelegatedRunOutput) -> anyhow::Result<()> { - let parent = path - .parent() - .ok_or_else(|| anyhow::anyhow!("result path has no parent: {}", path.display()))?; - std::fs::create_dir_all(parent)?; - let temporary = parent.join(format!( - ".{}.{}.tmp", - path.file_name() - .and_then(|name| name.to_str()) - .unwrap_or("run-result"), - uuid::Uuid::new_v4().simple() - )); - write_private_json(&temporary, output)?; - std::fs::rename(temporary, path)?; - Ok(()) -} - -fn write_private_json(path: &Path, value: &impl serde::Serialize) -> anyhow::Result<()> { - let body = serde_json::to_vec_pretty(value)?; - std::fs::write(path, body)?; - #[cfg(unix)] - { - use std::os::unix::fs::PermissionsExt; - std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o600))?; - } - Ok(()) -} - #[cfg(test)] mod tests { use super::*; diff --git a/crates/persisting-pvisor/src/executor/sandbox.rs b/crates/persisting-pvisor/src/executor/sandbox.rs index bf784bf9..b7b598f2 100644 --- a/crates/persisting-pvisor/src/executor/sandbox.rs +++ b/crates/persisting-pvisor/src/executor/sandbox.rs @@ -1677,6 +1677,21 @@ mod tests { #[test] fn seatbelt_proxy_allows_only_its_tcp_endpoint() { + // /usr/bin/python3 is an Xcode launcher on some runners. Resolve its + // interpreter and runtime before entering the network test's sandbox. + let python = std::process::Command::new("/usr/bin/python3") + .args([ + "-c", + "import json, sys; print(json.dumps([sys.executable, sys.base_prefix]))", + ]) + .output() + .unwrap(); + assert!( + python.status.success(), + "{}", + String::from_utf8_lossy(&python.stderr) + ); + let [interpreter, runtime]: [PathBuf; 2] = serde_json::from_slice(&python.stdout).unwrap(); let temp = tempfile::tempdir().unwrap(); let proxy = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); let other = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); @@ -1688,7 +1703,8 @@ mod tests { Some(&[ PathBuf::from("/System"), PathBuf::from("/usr"), - PathBuf::from("/Library/Developer"), + interpreter.clone(), + runtime, PathBuf::from("/private/etc"), PathBuf::from("/dev"), ]), @@ -1704,8 +1720,10 @@ mod tests { command.arg("-D").arg(format!("{key}={}", value.display())); } let output = command + // Prove the sandboxed invocation no longer needs xcrun selection. + .env("DEVELOPER_DIR", temp.path().join("no-developer-tools")) + .arg(&interpreter) .args([ - "/usr/bin/python3", "-c", r#" import errno, socket, sys diff --git a/crates/persisting-pvisor/src/executor/vm/supported.rs b/crates/persisting-pvisor/src/executor/vm/supported.rs index f692395b..e75ab065 100644 --- a/crates/persisting-pvisor/src/executor/vm/supported.rs +++ b/crates/persisting-pvisor/src/executor/vm/supported.rs @@ -2,6 +2,7 @@ use crate::config::VmSettings; use crate::executor::{AttemptContext, RunExecutor}; +use crate::util::write_private_json; use anyhow::Context as _; use async_trait::async_trait; use persisting_control::{ @@ -1102,12 +1103,6 @@ async fn join_capture( } } -fn write_private_json(path: &Path, value: &impl Serialize) -> anyhow::Result<()> { - use std::os::unix::fs::PermissionsExt; - std::fs::write(path, serde_json::to_vec(value)?)?; - std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o600))?; - Ok(()) -} fn failed_to_start( spec: &persisting_control::RunSpec, diff --git a/crates/persisting-pvisor/src/image/cache/lazy.rs b/crates/persisting-pvisor/src/image/cache/lazy.rs index 735f02c9..84b0c271 100644 --- a/crates/persisting-pvisor/src/image/cache/lazy.rs +++ b/crates/persisting-pvisor/src/image/cache/lazy.rs @@ -336,6 +336,8 @@ impl RemoteFs { } let ino = self.next_inode; self.next_inode += 1; + // libc mode constants are u16 on macOS and u32 on Linux. + #[allow(clippy::unnecessary_cast)] let kind = match kind.as_str() { "directory" => FileType::Directory, "file" => FileType::RegularFile, diff --git a/crates/persisting-pvisor/src/util.rs b/crates/persisting-pvisor/src/util.rs index b6af324d..bf4e982c 100644 --- a/crates/persisting-pvisor/src/util.rs +++ b/crates/persisting-pvisor/src/util.rs @@ -78,6 +78,11 @@ pub(crate) fn atomic_write(path: &Path, contents: &[u8], mode: u32) -> anyhow::R result } +/// Publish owner-only JSON using the same durable replacement as Run records. +pub(crate) fn write_private_json(path: &Path, value: &impl serde::Serialize) -> anyhow::Result<()> { + atomic_write(path, &serde_json::to_vec_pretty(value)?, 0o600) +} + pub(crate) fn encode_hex(bytes: &[u8]) -> String { const HEX: &[u8; 16] = b"0123456789abcdef"; let mut output = String::with_capacity(bytes.len() * 2); @@ -107,6 +112,31 @@ mod tests { ); } + #[test] + fn private_json_preserves_previous_contents_on_failure() { + let temp = tempfile::tempdir().unwrap(); + let path = temp.path().join("nested/result.json"); + write_private_json(&path, &serde_json::json!({"state": "completed"})).unwrap(); + let original = fs::read(&path).unwrap(); + assert_eq!( + fs::metadata(&path).unwrap().permissions().mode() & 0o777, + 0o600 + ); + let invalid = std::collections::BTreeMap::from([(vec![1, 2], "invalid JSON key")]); + assert!(write_private_json(&path, &invalid).is_err()); + assert_eq!(fs::read(&path).unwrap(), original); + + let directory = temp.path().join("existing-directory"); + fs::create_dir(&directory).unwrap(); + assert!(write_private_json(&directory, &true).is_err()); + assert!(directory.is_dir()); + assert_eq!( + fs::read_dir(temp.path()).unwrap().count(), + 2, + "temporary file leaked" + ); + } + #[test] fn durable_directory_creation_handles_nested_paths() { let temp = tempfile::tempdir().unwrap(); diff --git a/crates/persisting-replay/src/bridge/claude_resume.rs b/crates/persisting-replay/src/bridge/claude_resume.rs index 5f6f69f7..de8e54fa 100644 --- a/crates/persisting-replay/src/bridge/claude_resume.rs +++ b/crates/persisting-replay/src/bridge/claude_resume.rs @@ -9,10 +9,10 @@ use std::collections::BTreeSet; +use crate::io::sha256; use anyhow::{Context, Result, ensure}; use serde::{Deserialize, Serialize}; use serde_json::{Map, Value, json}; -use sha2::{Digest, Sha256}; pub const TRANSPORT_SCHEMA_VERSION: &str = "sandbox-playback.claude-resume-transport/v1"; pub const PROFILE_ID: &str = "claude-code/2.1.220/native-resume-v1"; @@ -206,7 +206,7 @@ pub fn canonical_messages_sha256(messages: &[Value]) -> Result { messages.iter().all(Value::is_object), "messages must be a list of JSON objects" ); - Ok(sha256_hex( + Ok(sha256( canonical_json(&Value::Array(messages.to_vec())).as_bytes(), )) } @@ -223,7 +223,7 @@ pub fn canonical_observation_sha256(tool_result_block: &Value) -> Result .and_then(Value::as_bool) .unwrap_or(false), }); - Ok(sha256_hex(canonical_json(&value).as_bytes())) + Ok(sha256(canonical_json(&value).as_bytes())) } /// Validate and remove exactly one Claude Code resume envelope. @@ -607,13 +607,6 @@ fn valid_sha256(value: &str) -> bool { .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) } -fn sha256_hex(value: &[u8]) -> String { - Sha256::digest(value) - .iter() - .map(|byte| format!("{byte:02x}")) - .collect() -} - /// Render JSON with recursively sorted object keys, compact separators and /// UTF-8 text, matching Python's `json.dumps(sort_keys=True,separators=(",",":"))`. fn canonical_json(value: &Value) -> String { From a9fb085efe5ecab58e87ade82df7e3eb78c63697 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 30 Sep 2026 11:26:41 +0800 Subject: [PATCH 6/9] Allow TCP binds for the proxy-only sandbox Permit `network-bind` on local TCP so `connect()` can implicitly bind an ephemeral port, while listen and outbound peers stay restricted. Update the sandbox test to cover explicit and implicit local binds. --- .../persisting-pvisor/src/executor/sandbox.rs | 26 ++++++++++++------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/crates/persisting-pvisor/src/executor/sandbox.rs b/crates/persisting-pvisor/src/executor/sandbox.rs index b7b598f2..e518466c 100644 --- a/crates/persisting-pvisor/src/executor/sandbox.rs +++ b/crates/persisting-pvisor/src/executor/sandbox.rs @@ -722,14 +722,18 @@ pub(crate) fn seatbelt_profile_with_reads( profile.push_str(")\n"); } if let NetworkIsolation::ProxyOnly(endpoint) = network { - // connect() may implicitly bind an ephemeral local port. Restrict - // peers and deny inbound connections instead of denying that bind. - profile = profile.replace("(deny network-bind (local ip))", ""); // Deny all IP except the allocated loopback TCP proxy port; unrelated localhost // services must not become alternate egress paths. profile = profile.replace("(allow network-outbound (remote ip \"localhost:*\"))", ""); match endpoint { Some(endpoint) if endpoint.ip().is_loopback() && endpoint.port() != 0 => { + // A missing bind rule inherits the network-inbound denial. + // TCP connect may implicitly bind: permit that operation, + // while listen and outbound peers remain restricted. + profile = profile.replace( + "(deny network-bind (local ip))", + "(allow network-bind (local tcp))", + ); profile = profile.replace( "(remote ip \"localhost:*\")", &format!("(remote tcp \"localhost:{}\")", endpoint.port()), @@ -1731,13 +1735,17 @@ allowed, denied = map(int, sys.argv[1:]) for kind, port in [(socket.SOCK_STREAM, denied), (socket.SOCK_DGRAM, allowed)]: with socket.socket(socket.AF_INET, kind) as s: assert s.connect_ex(('127.0.0.1', port)) in (errno.EACCES, errno.EPERM) -try: - with socket.socket() as s: - s.bind(('127.0.0.1', 0)) - s.listen(1) -except PermissionError: pass -else: raise AssertionError('listener escaped') +with socket.socket() as s: + s.bind(('127.0.0.1', 0)) + try: s.listen(1) + except PermissionError: pass + else: raise AssertionError('listener escaped') with socket.create_connection(('127.0.0.1', allowed), timeout=1): pass +# Exercise the local bind explicitly: connect() can also perform it implicitly. +for source in ['0.0.0.0', '127.0.0.1']: + with socket.create_connection(('127.0.0.1', allowed), timeout=1, + source_address=(source, 0)): pass + "#, ]) .arg(endpoint.port().to_string()) From 992bcf2d0968a9a6780a7644a4a1c9efcd5637bc Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 30 Sep 2026 11:47:03 +0800 Subject: [PATCH 7/9] Fix Seatbelt network policy for Unix IPC Allow network-bind and inbound Unix sockets directly while granting outbound Unix connections positively by path. Negated Unix-path denies could also reject permitted TCP connections on macOS, causing intermittent failures across ephemeral ports. Expand the test to cover local and denied socket roots and repeat it to surface flakiness. --- .../persisting-pvisor/src/executor/sandbox.rs | 124 ++++++++++-------- 1 file changed, 69 insertions(+), 55 deletions(-) diff --git a/crates/persisting-pvisor/src/executor/sandbox.rs b/crates/persisting-pvisor/src/executor/sandbox.rs index e518466c..204bc36b 100644 --- a/crates/persisting-pvisor/src/executor/sandbox.rs +++ b/crates/persisting-pvisor/src/executor/sandbox.rs @@ -652,9 +652,8 @@ pub(crate) fn seatbelt_profile_with_reads( // Deny by default for a network-isolated Run. The allowlist below is // intentionally small and mirrors the system services required by // shells, language runtimes, PTYs, and read-only preferences. Socket - // operations are admitted only so the filtered denies below can retain - // Run-local Unix IPC while rejecting non-loopback IP and ambient host - // Unix sockets. + // binds/listeners are allowed for Unix IPC; outbound connections require + // an explicit loopback or Unix-path grant below. let mut profile = String::from( "(version 1)\n\ (deny default)\n\ @@ -682,7 +681,7 @@ pub(crate) fn seatbelt_profile_with_reads( (global-name \"com.apple.PowerManagement.control\"))\n\ (allow file-ioctl (regex #\"^/dev/ttys[0-9]+$\"))\n\ (allow system-socket (socket-domain AF_UNIX))\n\ - (allow network*)\n\ + (allow network-bind network-inbound (local unix-socket))\n\ (deny network-bind (local ip))\n\ (deny network-inbound (local ip))\n\ (deny network-outbound\n\ @@ -764,20 +763,18 @@ pub(crate) fn seatbelt_profile_with_reads( } else { profile.push_str("(allow file-write*)\n"); } - profile.push_str("(deny network-outbound\n (require-all\n (remote unix-socket)\n"); + // Keep Unix-path grants positive and separate from IP rules. A negated + // Unix-path deny can also reject permitted TCP connections on macOS. for index in 0..allowed_unix_sockets.len() { profile.push_str(&format!( - " (require-not (remote unix-socket\n\ - (literal (param \"PVISOR_UNIX_SOCKET_{index}\"))))\n" + "(allow network-outbound (remote unix-socket (literal (param \"PVISOR_UNIX_SOCKET_{index}\"))))\n" )); } for index in 0..local_socket_roots.len() { profile.push_str(&format!( - " (require-not (remote unix-socket\n\ - (subpath (param \"PVISOR_SOCKET_ROOT_{index}\"))))\n" + "(allow network-outbound (remote unix-socket (subpath (param \"PVISOR_SOCKET_ROOT_{index}\"))))\n" )); } - profile.push_str(" )\n)\n"); return Ok((profile, parameters)); } @@ -1696,42 +1693,54 @@ mod tests { String::from_utf8_lossy(&python.stderr) ); let [interpreter, runtime]: [PathBuf; 2] = serde_json::from_slice(&python.stdout).unwrap(); - let temp = tempfile::tempdir().unwrap(); - let proxy = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); - let other = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); - let endpoint = proxy.local_addr().unwrap(); - let socket_path = temp.path().join("agentctl.sock"); - let _unix = std::os::unix::net::UnixListener::bind(&socket_path).unwrap(); - let (profile, params) = seatbelt_profile_with_reads( - &[temp.path().to_owned()], - Some(&[ - PathBuf::from("/System"), - PathBuf::from("/usr"), - interpreter.clone(), - runtime, - PathBuf::from("/private/etc"), - PathBuf::from("/dev"), - ]), - &[socket_path], - &[temp.path().to_owned()], - NetworkIsolation::ProxyOnly(Some(endpoint)), - true, - ) - .unwrap(); - let mut command = std::process::Command::new(MACOS_SANDBOX_EXEC); - command.current_dir(temp.path()).arg("-p").arg(&profile); - for (key, value) in params { - command.arg("-D").arg(format!("{key}={}", value.display())); - } - let output = command - // Prove the sandboxed invocation no longer needs xcrun selection. - .env("DEVELOPER_DIR", temp.path().join("no-developer-tools")) - .arg(&interpreter) - .args([ - "-c", - r#" + // The old mixed deny policy failed intermittently across ephemeral ports. + for _ in 0..16 { + let temp = tempfile::tempdir().unwrap(); + let proxy = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let other = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let endpoint = proxy.local_addr().unwrap(); + let socket_path = temp.path().join("agentctl.sock"); + let _unix = std::os::unix::net::UnixListener::bind(&socket_path).unwrap(); + let local_root = temp.path().join("ipc"); + std::fs::create_dir(&local_root).unwrap(); + let local_socket = local_root.join("local.sock"); + let denied_socket = temp.path().join("denied.sock"); + let _local = std::os::unix::net::UnixListener::bind(&local_socket).unwrap(); + let _denied = std::os::unix::net::UnixListener::bind(&denied_socket).unwrap(); + let (profile, params) = seatbelt_profile_with_reads( + &[temp.path().to_owned()], + Some(&[ + PathBuf::from("/System"), + PathBuf::from("/usr"), + interpreter.clone(), + runtime.clone(), + PathBuf::from("/private/etc"), + PathBuf::from("/dev"), + ]), + std::slice::from_ref(&socket_path), + &[local_root], + NetworkIsolation::ProxyOnly(Some(endpoint)), + true, + ) + .unwrap(); + let mut command = std::process::Command::new(MACOS_SANDBOX_EXEC); + command.current_dir(temp.path()).arg("-p").arg(&profile); + for (key, value) in params { + command.arg("-D").arg(format!("{key}={}", value.display())); + } + let output = command + // Prove the sandboxed invocation no longer needs xcrun selection. + .env("DEVELOPER_DIR", temp.path().join("no-developer-tools")) + .arg(&interpreter) + .args([ + "-c", + r#" import errno, socket, sys -allowed, denied = map(int, sys.argv[1:]) +allowed, denied = map(int, sys.argv[1:3]) +for path in sys.argv[3:5]: + with socket.socket(socket.AF_UNIX) as s: s.connect(path) +with socket.socket(socket.AF_UNIX) as s: + assert s.connect_ex(sys.argv[5]) in (errno.EACCES, errno.EPERM) for kind, port in [(socket.SOCK_STREAM, denied), (socket.SOCK_DGRAM, allowed)]: with socket.socket(socket.AF_INET, kind) as s: assert s.connect_ex(('127.0.0.1', port)) in (errno.EACCES, errno.EPERM) @@ -1747,16 +1756,21 @@ for source in ['0.0.0.0', '127.0.0.1']: source_address=(source, 0)): pass "#, - ]) - .arg(endpoint.port().to_string()) - .arg(other.local_addr().unwrap().port().to_string()) - .output() - .unwrap(); - assert!( - output.status.success(), - "{}", - String::from_utf8_lossy(&output.stderr) - ); + ]) + .arg(endpoint.port().to_string()) + .arg(other.local_addr().unwrap().port().to_string()) + .arg(&socket_path) + .arg(&local_socket) + .arg(&denied_socket) + .output() + .unwrap(); + assert!( + output.status.success(), + "proxy={endpoint}, interpreter={}, profile={profile}\n{}", + interpreter.display(), + String::from_utf8_lossy(&output.stderr) + ); + } } #[test] From 194a3c1a7e093817ec37866973546b281adbcb36 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 30 Sep 2026 12:08:20 +0800 Subject: [PATCH 8/9] Align rebased macOS tests with independent filesystem policies --- crates/persisting-pvisor/src/cli/run.rs | 20 +++++++++++++------ .../src/executor/vm/supported.rs | 1 - crates/persisting-pvisor/tests/macos_safe.rs | 19 ++++++++++++++++-- 3 files changed, 31 insertions(+), 9 deletions(-) diff --git a/crates/persisting-pvisor/src/cli/run.rs b/crates/persisting-pvisor/src/cli/run.rs index 77a5897b..59d7099d 100644 --- a/crates/persisting-pvisor/src/cli/run.rs +++ b/crates/persisting-pvisor/src/cli/run.rs @@ -794,7 +794,7 @@ pub async fn run(mut args: RunArgs) -> anyhow::Result { } // Host runs keep the best-effort lifecycle/evidence profile by default; // filesystem restrictions, staging, and network isolation remain opt-in. - // `--safe`/`--audit` request the Agent-aware preset on top of that. + // `--safe`/`--ask` request the Agent-aware preset on top of that. let run_id = format!("run-{}", uuid::Uuid::new_v4()); let mut config = load_run_config(&args, personal_config_root().as_deref(), true)?; apply_run_options(&mut config, args.clone())?; @@ -1229,7 +1229,9 @@ async fn execute_config( all(target_os = "linux", target_env = "musl", target_arch = "x86_64"), all(target_os = "macos", target_arch = "x86_64") )))] - if config.vm.library_dir.is_none() && crate::executor::vm::bundled_firmware_dir().is_none() { + if config.vm.library_dir.is_none() + && crate::executor::vm::bundled_firmware_dir().is_none() + { run_log!( "pVisor firmware: resolving libkrunfw {}", crate::executor::vm::firmware::VERSION @@ -1372,7 +1374,7 @@ async fn execute_config( let executor: Arc = match config.run.executor { #[cfg(target_os = "linux")] RunExecutorKind::Host if safe || filesystem_isolated || network_namespace_required => { - // The --safe/--audit preset demands its boundary; the independent + // The --safe/--ask preset demands its boundary; the independent // --filesystem/--overlaynet policies stay best-effort and fall // back to the host process with a warning. if rootless_available { @@ -1580,8 +1582,10 @@ async fn execute_config( ); if safe { - spec.metadata - .insert(crate::executor::sandbox::LANDLOCK_SANDBOX_KEY.into(), true.into()); + spec.metadata.insert( + crate::executor::sandbox::LANDLOCK_SANDBOX_KEY.into(), + true.into(), + ); } { let network_boundary = if config.run.executor == RunExecutorKind::Vm @@ -3552,7 +3556,11 @@ sandbox = "required""# config.run.executor = RunExecutorKind::Vm; config.vm.rootfs = Some(temporary.path().to_path_buf()); config.vm.library_dir = Some(temporary.path().to_path_buf()); - std::fs::write(temporary.path().join(crate::executor::vm::firmware_name()), []).unwrap(); + std::fs::write( + temporary.path().join(crate::executor::vm::firmware_name()), + [], + ) + .unwrap(); config.overlayfs = Some(OverlayFsSettings { base: Some(temporary.path().to_path_buf()), ..OverlayFsSettings::default() diff --git a/crates/persisting-pvisor/src/executor/vm/supported.rs b/crates/persisting-pvisor/src/executor/vm/supported.rs index e75ab065..2f35ff1f 100644 --- a/crates/persisting-pvisor/src/executor/vm/supported.rs +++ b/crates/persisting-pvisor/src/executor/vm/supported.rs @@ -1103,7 +1103,6 @@ async fn join_capture( } } - fn failed_to_start( spec: &persisting_control::RunSpec, attempt_id: &persisting_control::AttemptId, diff --git a/crates/persisting-pvisor/tests/macos_safe.rs b/crates/persisting-pvisor/tests/macos_safe.rs index a08928f1..75fabc7b 100644 --- a/crates/persisting-pvisor/tests/macos_safe.rs +++ b/crates/persisting-pvisor/tests/macos_safe.rs @@ -112,7 +112,14 @@ fn safe_profile_stages_reviews_and_applies_on_macos() { command .env("PERSISTING_RUN_HOME", &run_home) .current_dir(&workspace) - .args(["run", "--stdio", "capture", "--stage"]) + .args([ + "run", + "--filesystem", + "sandbox", + "--stdio", + "capture", + "--stage", + ]) .arg(run_home.join("stage")) .args([ "--", @@ -264,6 +271,8 @@ fn deny_all_blocks_ip_and_host_unix_sockets_on_macos() { .args([ "run", "--overlaynet-deny-all", + "--filesystem", + "sandbox", "--stdio", "capture", "--stage", @@ -510,7 +519,13 @@ fn ask_preserves_default_rules_read_only_shares_and_job_changes() { .env("PVISOR_UI_STAGE_FILE", temp.path().join("stage")) .env("PVISOR_UI_LOG_FILE", &log) .env("PVISOR_UI_AUDIT_SOCKET", &socket) - .args(["--ask", "--no-config", "--access", "custom:ask", "--mount"]) + .args([ + "--ask", + "--no-agent-defaults", + "--access", + "custom:ask", + "--mount", + ]) .arg(format!("{}:read", reference.display())) .args([ "--", From 5a3bf84258c9bab30802b1946bbd37165d7c4ef3 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 30 Sep 2026 12:38:22 +0800 Subject: [PATCH 9/9] Fix Linux sandbox tests after isolation policy rebase --- crates/persisting-pvisor/src/executor/process.rs | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/crates/persisting-pvisor/src/executor/process.rs b/crates/persisting-pvisor/src/executor/process.rs index cda93658..53570c8c 100644 --- a/crates/persisting-pvisor/src/executor/process.rs +++ b/crates/persisting-pvisor/src/executor/process.rs @@ -1904,6 +1904,7 @@ mod tests { temporary.path().join("root"), temporary.path().join("attestation"), NetworkIsolation::Ambient, + true, ) .unwrap(); assert!(normal.read_write.contains(&PathBuf::from("/"))); @@ -1930,13 +1931,14 @@ mod tests { temporary.path().join("root"), temporary.path().join("attestation"), NetworkIsolation::Ambient, + true, ) .unwrap(); assert!(!safe.read_write.contains(&PathBuf::from("/"))); assert_eq!(safe.staged_roots, vec![home]); assert_eq!(safe.staged_workspace, Some(workspace)); assert_eq!(safe.staged_workspace_source, Some(merged)); - assert!(safe.landlock); + assert!(safe.filesystem_isolated); } #[cfg(target_os = "linux")] @@ -1989,7 +1991,7 @@ mod tests { ProcessExecutor::rootless_with_launcher(std::env::current_exe().unwrap()).unwrap(); let RunInvocation::Process(invocation) = &spec.invocation; let command = executor.spawn_command(&spec, invocation).unwrap(); - let encoded = command + let plan_path = command .command .as_std() .get_envs() @@ -1997,6 +1999,7 @@ mod tests { (key == SANDBOX_PLAN_ENV).then(|| value.unwrap().to_string_lossy().into_owned()) }) .unwrap(); + let encoded = std::fs::read_to_string(plan_path).unwrap(); let plan: SandboxPlan = serde_json::from_str(&encoded).unwrap(); assert!(!plan.filesystem_isolated); }