diff --git a/Cargo.lock b/Cargo.lock index aec661e0d..2a5f13ba3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7254,6 +7254,7 @@ dependencies = [ "async-trait", "base64 0.22.1", "blake3", + "bytes", "chrono", "criterion", "datafusion", @@ -7266,9 +7267,11 @@ dependencies = [ "lance-datafusion", "lance-file", "lance-index", + "lance-io", "lance-linalg", "lance-table", "libc", + "object_store", "opendal 0.57.0", "persisting-agentctl", "persisting-events", diff --git a/Cargo.toml b/Cargo.toml index 4a6e19ae5..480791773 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -60,6 +60,7 @@ lance-core = "11.0.0" lance-file = "11.0.0" lance-datafusion = "11.0.0" lance-index = { version = "11.0.0", features = ["tokenizer-jieba"] } +lance-io = "11.0.0" lance-linalg = "11.0.0" lance-table = "11.0.0" libc = "0.2" @@ -75,6 +76,7 @@ opendal = { version = "0.57.0", default-features = true, features = [ "services-s3", "services-tos", ] } +object_store = "0.13.2" persisting-agentctl = { path = "crates/persisting-agentctl" } persisting-events = { path = "crates/persisting-events" } persisting-gateway = { path = "crates/persisting-gateway" } diff --git a/crates/persisting-pchronicle-cli/src/gateway_partition.rs b/crates/persisting-pchronicle-cli/src/gateway_partition.rs index 47c31dbf2..bdf2b329a 100644 --- a/crates/persisting-pchronicle-cli/src/gateway_partition.rs +++ b/crates/persisting-pchronicle-cli/src/gateway_partition.rs @@ -1,7 +1,7 @@ //! Safe, bounded physical partitioning below one logical Gateway Dataset. use std::collections::HashMap; -use std::sync::Mutex; +use std::sync::{Mutex, MutexGuard}; use anyhow::{Context, Result}; use chrono::{DateTime, Datelike, Timelike, Utc}; @@ -10,6 +10,12 @@ const MAX_TEMPLATE_BYTES: usize = 256; const MAX_TEMPLATE_SEGMENTS: usize = 16; const MAX_USER_SEGMENT_BYTES: usize = 80; +fn lock_recover(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) +} + #[derive(Debug, Clone, PartialEq, Eq)] enum Segment { Literal(String), @@ -149,7 +155,7 @@ impl GatewayPartitionRouter { let Some(split) = &self.split else { return self.dataset_uri.clone(); }; - let mut routes = self.routes.lock().unwrap(); + let mut routes = lock_recover(&self.routes); routes .entry(route_key.to_string()) .or_insert_with(|| { diff --git a/crates/persisting-pchronicle-cli/src/lib.rs b/crates/persisting-pchronicle-cli/src/lib.rs index df2949ad1..2c2104dd9 100644 --- a/crates/persisting-pchronicle-cli/src/lib.rs +++ b/crates/persisting-pchronicle-cli/src/lib.rs @@ -121,7 +121,9 @@ impl Cli { /// invocations return None and retain their existing execution path. pub fn run_catalog_worker_before_runtime(cli: &Cli) -> Option> { match &cli.command { - Command::Serve(args) if args.catalog_query_worker => Some(server::catalog_worker::run()), + Command::Serve(args) if args.catalog_query_worker => { + Some(server::catalog_worker::run(cli.log_level)) + } _ => None, } } @@ -172,6 +174,18 @@ pub enum LogLevel { Debug, } +impl LogLevel { + /// The `--log-level` value that parses back to this variant. + pub(crate) fn as_arg(self) -> &'static str { + match self { + Self::Error => "error", + Self::Warn => "warn", + Self::Info => "info", + Self::Debug => "debug", + } + } +} + struct DiagnosticWriter<'a> { level: LogLevel, inner: &'a mut dyn Write, @@ -3589,6 +3603,7 @@ pub(crate) async fn find_expression_predicate_for_dataset( for predicate in text_predicates { let mut matches = Vec::::new(); let mut searched = false; + let mut jobs = Vec::new(); for dataset in snapshot.datasets() { if dataset_filter.is_some_and(|filter| dataset.mount.name != filter) { continue; @@ -3596,44 +3611,55 @@ pub(crate) async fn find_expression_predicate_for_dataset( for source in &dataset.sources { if source.status != CatalogSourceStatus::Ready || source.kind != CatalogSourceKind::Store - || source_filter.is_some_and(|filter| filter != source.file) + || source_filter.is_some_and(|filter| { + filter != source.file && !source.file.starts_with(&format!("{filter}/")) + }) { continue; } - let Some(paths) = + if let Some(paths) = snapshot.storyline_table_paths(&dataset.mount.name, &source.file)? - else { - continue; - }; - match search_storyline_step_matches_fts_in_columns( - &paths, - &predicate.query, - predicate.field.columns(), - ) - .await { - Ok(step_matches) => { - available = true; - searched = true; - let source_predicate = predicate - .field - .source_predicate() - .map(|value| format!(" AND ({value})")) - .unwrap_or_default(); - matches.extend(step_matches.into_iter().map(|(document_id, step_id)| { - format!( - "(_file_ = {} AND document_id = {} AND step_id = {}{})", - sql_string(&source.file), - sql_string(&document_id), - step_id, - source_predicate, - ) - })); - } - Err(error) => errors.push(format!( - "FTS unavailable for {} / {}: {error:#}", - dataset.mount.name, source.file - )), + jobs.push((dataset.mount.name.clone(), source.file.clone(), paths)); + } + } + } + let query = predicate.query.clone(); + let columns = predicate.field.columns(); + let results = stream::iter(jobs) + .map(|(dataset, file, paths)| { + let query = query.clone(); + async move { + let result = + search_storyline_step_matches_fts_in_columns(&paths, &query, columns).await; + (dataset, file, result) + } + }) + .buffer_unordered(4) + .collect::>() + .await; + for (dataset, file, result) in results { + match result { + Ok(step_matches) => { + available = true; + searched = true; + let source_predicate = predicate + .field + .source_predicate() + .map(|value| format!(" AND ({value})")) + .unwrap_or_default(); + matches.extend(step_matches.into_iter().map(|(document_id, step_id)| { + format!( + "(_file_ = {} AND document_id = {} AND step_id = {}{})", + sql_string(&file), + sql_string(&document_id), + step_id, + source_predicate, + ) + })); + } + Err(error) => { + errors.push(format!("FTS unavailable for {dataset} / {file}: {error:#}")) } } } diff --git a/crates/persisting-pchronicle-cli/src/server/catalog.rs b/crates/persisting-pchronicle-cli/src/server/catalog.rs index 15aee69ab..dcf09da3a 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog.rs @@ -157,9 +157,20 @@ impl CatalogAcl { } pub(crate) fn public_mounts(&self) -> Vec { - self.public_for_all() + self.libraries_for_public() .into_iter() - .filter_map(|library| DatasetMount::new(&library.name, &library.uri).ok()) + .filter_map(|library| { + DatasetMount::new(&library.name, &library.uri) + .ok() + .map(|mount| { + mount.with_backend(persisting_pchronicle::storage::StoreConfig { + endpoint: library.endpoint.clone(), + region: library.region.clone(), + access_key: library.access_key.clone(), + secret_key: library.secret_key.clone(), + }) + }) + }) .collect() } @@ -171,6 +182,32 @@ impl CatalogAcl { .collect() } + pub(crate) fn libraries_for_public(&self) -> Vec { + self.public_datasets + .iter() + .filter_map(|name| self.libraries.get(name)) + .cloned() + .collect() + } + + fn visible_for_headers( + &self, + headers: &axum::http::HeaderMap, + ) -> Result, ApiError> { + match credentials_from_headers(headers) { + Some((access, secret)) => self + .authenticate(&access, &secret) + .map(|user| self.list_for(user)) + .ok_or_else(catalog_unauthorized), + None if !headers.contains_key(ACCESS_KEY_HEADER) + && !headers.contains_key(SECRET_KEY_HEADER) => + { + Ok(self.public_for_all()) + } + None => Err(catalog_unauthorized()), + } + } + fn credentials_for_public(&self, dataset: &str) -> Option<(&str, &str)> { self.users_by_access_key .iter() @@ -714,7 +751,11 @@ pub(crate) fn apply_library_env(library: &CatalogLibrary) { } } } - if let Some(region) = library.region.as_deref() { + if let Some(region) = library + .region + .as_deref() + .or_else(|| library.uri.starts_with("s3://").then_some("us-west-2")) + { unsafe { std::env::set_var("AWS_REGION", region); std::env::set_var("AWS_DEFAULT_REGION", region); @@ -813,7 +854,8 @@ fn parent_handles_path(path: &str) -> bool { .strip_prefix("/api/v1") .or_else(|| path.strip_prefix("/api")) .unwrap_or(path); - rest == "/health" + rest.starts_with("/requests/") + || rest == "/health" || rest == "/ui" || rest == "/catalog/datasets" || rest.starts_with("/catalog/datasets/") @@ -825,17 +867,7 @@ pub(super) async fn list_datasets( ) -> Result>, ApiError> { let axum::Extension(snapshot) = snapshot.ok_or_else(|| ApiError::not_found("catalog is not enabled"))?; - let acl = &snapshot.acl; - let has_credential_headers = - headers.contains_key(ACCESS_KEY_HEADER) || headers.contains_key(SECRET_KEY_HEADER); - let libraries = match credentials_from_headers(&headers) { - Some((access, secret)) => acl - .authenticate(&access, &secret) - .map(|user| acl.list_for(user)) - .ok_or_else(catalog_unauthorized)?, - None if !has_credential_headers => acl.public_for_all(), - None => return Err(catalog_unauthorized()), - }; + let libraries = snapshot.acl.visible_for_headers(&headers)?; Ok(axum::Json(libraries)) } @@ -877,18 +909,30 @@ pub(super) async fn catalog_data_plane_layer( if parent_handles_path(&path) { return next.run(request).await; } - // Anonymous browsing is limited to wildcard-granted datasets. - if credentials_from_headers(request.headers()).is_none() - && path.ends_with("/query/tables") + // UI metadata is ACL-derived, not an all-dataset storage query. In + // particular, authenticated callers may have incompatible S3 backends. + let api_path = path + .strip_prefix("/api/v1") + .or_else(|| path.strip_prefix("/api")) + .unwrap_or(&path); + let ui_tables = api_path == "/query/tables" && url::form_urlencoded::parse(request.uri().query().unwrap_or("").as_bytes()) - .any(|(key, value)| key == "ui" && (value == "true" || value == "1")) - { - if let Some(library) = acl.libraries.values().find(|library| { - acl.public_for_all() + .any(|(key, value)| key == "ui" && (value == "true" || value == "1")); + let root_tree = api_path == "/explorer/tree" + && !url::form_urlencoded::parse(request.uri().query().unwrap_or("").as_bytes()) + .any(|(key, value)| (key == "dataset" || key == "prefix") && !value.is_empty()); + if request.method() == axum::http::Method::GET && (ui_tables || root_tree) { + let libraries = match acl.visible_for_headers(request.headers()) { + Ok(libraries) => libraries, + Err(error) => return error.into_response(), + }; + if root_tree { + let mounts = libraries .iter() - .any(|public| public.name == library.name) - }) { - apply_library_env(library); + .filter_map(|library| DatasetMount::new(&library.name, &library.uri).ok()) + .collect::>(); + return axum::Json(super::explorer::catalog_tree_from_mount_specs(&mounts)) + .into_response(); } let catalog = super::QueryCatalog { snapshot_id: String::new(), @@ -896,7 +940,16 @@ pub(super) async fn catalog_data_plane_layer( database: String::new(), storage_path: String::new(), path_column: "_file_", - datasets: Vec::new(), + datasets: libraries + .into_iter() + .map(|library| super::QueryDatasetSummary { + browse: None, + name: library.name, + uri: library.uri, + ready_sources: 0, + error_sources: 0, + }) + .collect(), tables: super::query_table_summaries(), }; return axum::Json(catalog).into_response(); @@ -929,13 +982,6 @@ pub(super) async fn catalog_data_plane_layer( .unwrap_or_else(|error| error.into_response()); } if !dataset.is_empty() { - if let Some(library) = acl - .libraries - .values() - .find(|library| library.name == dataset) - { - apply_library_env(library); - } if let Some((access_key, secret_key)) = acl.credentials_for_public(dataset) { let headers = request.headers_mut(); if let (Ok(access_key), Ok(secret_key)) = (access_key.parse(), secret_key.parse()) { @@ -1328,6 +1374,8 @@ dataset = "prod" #[test] fn parent_keeps_health_and_catalog_ticket_routes() { assert!(parent_handles_path("/api/health")); + assert!(parent_handles_path("/api/requests/example")); + assert!(parent_handles_path("/api/v1/requests/example")); assert!(parent_handles_path("/api/v1/catalog/datasets")); assert!(parent_handles_path("/api/v1/catalog/datasets/prod")); assert!(!parent_handles_path("/api/catalog")); @@ -1485,6 +1533,69 @@ uri = "{}" builder.body(axum::body::Body::empty()).unwrap() } + #[tokio::test] + async fn ui_metadata_handles_mixed_backends_without_mounting_workers() { + use tower::ServiceExt; + let acl = + CatalogAcl::parse(&SAMPLE.replacen("127.0.0.1:9000", "127.0.0.1:9001", 1)).unwrap(); + assert!( + super::super::catalog_worker::validate_backends( + &acl.libraries.values().cloned().collect::>() + ) + .is_err() + ); + let mut state = super::super::app_state(super::super::ChronicleServerConfig::front_only()); + state.catalog_acl = Some(std::sync::Arc::new(CatalogState::new(acl, None))); + let warehouse = super::super::PreparedWarehouse { state }; + let app = warehouse.router(); + for path in [ + "/api/query/tables?ui=true", + "/api/v1/query/tables?ui=1", + "/api/explorer/tree?dataset=&prefix=", + "/api/v1/explorer/tree", + ] { + for (access, secret, expected) in [ + (Some("USER_AK"), Some("USER_SK"), 2), + (Some("BOB_AK"), Some("BOB_SK"), 1), + (None, None, 0), + ] { + let (status, body) = catalog_body( + app.clone() + .oneshot(catalog_request(path, access, secret)) + .await + .unwrap(), + ) + .await; + assert_eq!(status, axum::http::StatusCode::OK, "{path}: {body}"); + let json: serde_json::Value = serde_json::from_str(&body).unwrap(); + let entries = json[if path.contains("query/tables") { + "datasets" + } else { + "children" + }] + .as_array() + .unwrap(); + assert_eq!(entries.len(), expected, "{body}"); + if expected == 1 { + assert_eq!(entries[0]["name"], "evals"); + } + assert!(!body.contains("BACKEND_AK") && !body.contains("BACKEND_SK")); + } + for secret in [Some("wrong"), None] { + let (status, _) = catalog_body( + app.clone() + .oneshot(catalog_request(path, Some("USER_AK"), secret)) + .await + .unwrap(), + ) + .await; + assert_eq!(status, axum::http::StatusCode::UNAUTHORIZED); + } + } + assert!(warehouse.state.browse.get().is_none()); + assert!(warehouse.state.catalog.read().await.is_none()); + } + #[tokio::test] async fn catalog_list_allows_anonymous_public_datasets_and_rejects_invalid_credentials() { use tower::ServiceExt; diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index 52343a515..ff4f96fab 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -1,27 +1,30 @@ //! Exec workers have an immutable authenticated scope. No storage client or //! runtime is inherited from the listening process. use std::{ - collections::HashMap, io::{Read, Write}, path::PathBuf, process::Stdio, - sync::Arc, + sync::{Arc, Once}, time::Duration, }; use anyhow::{Context, Result}; use serde::{Deserialize, Serialize, de::DeserializeOwned}; use tokio::io::{AsyncReadExt, AsyncWriteExt}; -use tokio::sync::{Mutex, Semaphore}; +use tokio::sync::{Mutex, Notify, OwnedMutexGuard, Semaphore}; +use tokio::time::Instant; use tower::ServiceExt; use super::{ catalog::{CatalogLibrary, apply_library_env}, - problem::ApiError, + problem::{ApiError, ExecutionStage}, }; const MAX_WORKERS: usize = 8; const MAX_REQUESTS: usize = 32; +const MAX_WORKERS_PER_SCOPE: usize = 4; +const WORKER_IDLE_TIMEOUT: Duration = Duration::from_secs(120); +const REAP_INTERVAL: Duration = Duration::from_secs(30); const FRAME_LIMIT: usize = 40 * 1024 * 1024; const REQUEST_TIMEOUT: Duration = Duration::from_secs(60); @@ -53,23 +56,142 @@ impl WorkerResponse { } } +#[derive(Serialize, Deserialize)] +#[serde(tag = "type", content = "value")] +enum WorkerEvent { + Progress(super::request_progress::Snapshot), + Response(WorkerResponse), +} + #[derive(Serialize, Deserialize)] struct Bootstrap { mounts: Vec, } -type Slot = Arc>>; +struct SlotState { + worker: Option, + idle_since: Instant, +} + +struct Slot { + scope: String, + state: Arc>, +} + +#[derive(Default)] +struct PoolState { + slots: Mutex>, + available: Arc, +} + +// A reserved slot is never visible as idle, including while its child starts. +// Drop wakes all scopes: a released slot can satisfy a different scope by +// eviction even if the first waiter has reached its per-scope limit. +struct Lease { + guard: Option>, + available: Arc, +} + +impl Drop for Lease { + fn drop(&mut self) { + if let Some(mut guard) = self.guard.take() { + guard.idle_since = Instant::now(); + drop(guard); + } + self.available.notify_waiters(); + } +} + +impl PoolState { + async fn try_lease(&self, scope: &str) -> Result, ApiError> { + let mut slots = self.slots.lock().await; + for slot in slots.iter().filter(|slot| slot.scope == scope) { + if let Ok(guard) = slot.state.clone().try_lock_owned() { + return Ok(Some(Lease { + guard: Some(guard), + available: self.available.clone(), + })); + } + } + if slots.iter().filter(|slot| slot.scope == scope).count() >= MAX_WORKERS_PER_SCOPE { + return Ok(None); + } + if slots.len() >= MAX_WORKERS { + // Never enqueue on a busy slot. Reclaim only an idle child from + // another scope, and wait for its exit before reusing its capacity. + let idle = slots + .iter() + .enumerate() + .filter_map(|(index, slot)| { + slot.state + .clone() + .try_lock_owned() + .ok() + .map(|guard| (index, guard)) + }) + .min_by_key(|(_, guard)| guard.idle_since); + let Some((index, mut guard)) = idle else { + return Ok(None); + }; + if let Some(worker) = guard.worker.as_mut() { + worker + .child + .kill() + .await + .map_err(|error| worker_error(error.into()))?; + } + slots.remove(index); + } + let state = Arc::new(Mutex::new(SlotState { + worker: None, + idle_since: Instant::now(), + })); + let guard = state + .clone() + .try_lock_owned() + .expect("new worker slot is idle"); + slots.push(Slot { + scope: scope.to_owned(), + state, + }); + Ok(Some(Lease { + guard: Some(guard), + available: self.available.clone(), + })) + } + + async fn reap_idle(&self) { + let mut slots = self.slots.lock().await; + for index in (0..slots.len()).rev() { + let Ok(mut guard) = slots[index].state.clone().try_lock_owned() else { + continue; + }; + if guard.idle_since.elapsed() < WORKER_IDLE_TIMEOUT { + continue; + } + if let Some(worker) = guard.worker.as_mut() + && worker.child.kill().await.is_err() + { + continue; + } + slots.remove(index); + } + self.available.notify_waiters(); + } +} pub(super) struct WorkerPool { - slots: Mutex>, + state: Arc, requests: Semaphore, + reaper: Once, } impl Default for WorkerPool { fn default() -> Self { Self { - slots: Mutex::new(HashMap::new()), + state: Arc::new(PoolState::default()), requests: Semaphore::new(MAX_REQUESTS), + reaper: Once::new(), } } } @@ -78,33 +200,33 @@ impl WorkerPool { pub(super) fn admit(&self) -> Result, ApiError> { self.requests .try_acquire() - .map_err(|_| ApiError::unavailable()) + .map_err(|_| ApiError::unavailable().with_stage(ExecutionStage::Admission)) } - async fn slot(&self, scope: &str) -> Result { - let mut slots = self.slots.lock().await; - if let Some(slot) = slots.get(scope) { - return Ok(slot.clone()); - } - if slots.len() >= MAX_WORKERS { - // Only evict a worker with no in-flight or queued request. Wait for - // its exit before spawning a replacement, keeping the process cap. - let idle = slots - .iter() - .find(|(_, slot)| Arc::strong_count(slot) == 1) - .map(|(key, _)| key.clone()); - let Some(idle) = idle else { - return Err(ApiError::unavailable()); - }; - if let Some(slot) = slots.remove(&idle) - && let Some(mut worker) = slot.lock().await.take() - { - let _ = worker.child.kill().await; + async fn lease(&self, scope: &str) -> Result { + self.reaper.call_once(|| { + let state = Arc::downgrade(&self.state); + tokio::spawn(async move { + loop { + tokio::time::sleep(REAP_INTERVAL).await; + let Some(state) = state.upgrade() else { + break; + }; + state.reap_idle().await; + } + }); + }); + loop { + // Register before checking capacity so completion cannot be missed + // between a failed checkout and going to sleep. + let ready = self.state.available.notified(); + tokio::pin!(ready); + ready.as_mut().enable(); + if let Some(lease) = self.state.try_lease(scope).await? { + return Ok(lease); } + ready.await; } - let slot = Arc::new(Mutex::new(None)); - slots.insert(scope.to_owned(), slot.clone()); - Ok(slot) } pub(super) async fn execute( @@ -114,27 +236,30 @@ impl WorkerPool { request: WorkerRequest, ) -> Result { tokio::time::timeout(REQUEST_TIMEOUT, async { - let slot = self.slot(&scope).await?; - let mut guard = slot.lock().await; + super::request_progress::phase("worker_queue"); + let mut lease = self.lease(&scope).await?; + let guard = lease.guard.as_mut().expect("reserved worker slot"); // Ownership stays in this future during IPC: cancellation, timeout // or a partial frame drops/kills it rather than reusing dirty pipes. - let mut worker = match guard.take() { + super::request_progress::phase("worker_start"); + let mut worker = match guard.worker.take() { Some(worker) => worker, None => Worker::start(&scope, mounts).await.map_err(worker_error)?, }; + super::request_progress::phase("worker_execution"); let response = worker.exchange(&request).await.map_err(worker_error)?; let response = response.into_response().map_err(worker_error)?; - *guard = Some(worker); + guard.worker = Some(worker); Ok(response) }) .await - .map_err(|_| ApiError::unavailable())? + .map_err(|_| ApiError::unavailable().with_stage(ExecutionStage::Query))? } } fn worker_error(error: anyhow::Error) -> ApiError { // Protocol/OS diagnostics only; never log bootstrap payloads or child stderr. - ApiError::internal("", "catalog_worker", error) + ApiError::internal("", "catalog_worker", error).with_stage(ExecutionStage::Worker) } struct Worker { @@ -148,9 +273,12 @@ fn command( exe: PathBuf, home: &std::path::Path, cache: &std::path::Path, + blocks: &std::path::Path, ) -> tokio::process::Command { let mut command = tokio::process::Command::new(exe); command + .arg("--log-level") + .arg(super::request_log::initialized_log_level().as_arg()) .arg("serve") .arg("--catalog-query-worker") .env_clear() @@ -160,6 +288,10 @@ fn command( .env("XDG_CONFIG_HOME", home) .env("XDG_CACHE_HOME", home) .env("PCHRONICLE_CACHE_DIR", cache) + // `env_clear` plus a throwaway HOME makes the Lance block cache resolve + // under a directory that dies with the worker, so every worker refetched + // the same index pages from the object store. Name it explicitly. + .env("PCHRONICLE_LANCE_CACHE_DIR", blocks) .env("AWS_EC2_METADATA_DISABLED", "true") .env("RAYON_NUM_THREADS", "2") .env("AWS_CONFIG_FILE", home.join("no-aws-config")) @@ -181,6 +313,10 @@ fn command( "SSL_CERT_FILE", "SSL_CERT_DIR", "PCHRONICLE_QUERY_MEMORY_LIMIT", + "PCHRONICLE_LANCE_CACHE_CAPACITY_BYTES", + // Workers issue the reads, so admission tuning that never reaches them + // tunes nothing. + "PCHRONICLE_OBJECT_STORE_CONCURRENCY", "RUST_LOG", ] { if let Some(value) = std::env::var_os(key) { @@ -197,7 +333,13 @@ impl Worker { .map(PathBuf::from) .or_else(|| dirs::cache_dir().map(|p| p.join("pchronicle"))) .context("no catalog worker cache directory")?; - let cache = std::path::absolute(root)?.join("workers").join(scope); + let root = std::path::absolute(root)?; + let cache = root.join("workers").join(scope); + // Blocks are keyed by store, object version and size, so every worker + // and every scope can share them. Keeping them beside the per-scope + // caches rather than inside one means a reader does not refetch what + // another worker already paid for. + let blocks = root.join("blocks"); let mut builder = std::fs::DirBuilder::new(); builder.recursive(true); #[cfg(unix)] @@ -206,7 +348,8 @@ impl Worker { builder.mode(0o700); } builder.create(&cache)?; - let mut child = command(std::env::current_exe()?, home.path(), &cache).spawn()?; + builder.create(&blocks)?; + let mut child = command(std::env::current_exe()?, home.path(), &cache, &blocks).spawn()?; let input = child.stdin.take().context("worker stdin missing")?; let output = child.stdout.take().context("worker stdout missing")?; let mut worker = Self { @@ -238,7 +381,16 @@ impl Worker { async fn exchange(&mut self, request: &WorkerRequest) -> Result { self.send(request).await?; - self.receive().await + loop { + match self.receive::().await? { + WorkerEvent::Progress(snapshot) => { + if let Some(p) = super::request_progress::current() { + p.worker(snapshot); + } + } + WorkerEvent::Response(response) => return Ok(response), + } + } } } @@ -290,7 +442,11 @@ pub(super) fn validate_backends(mounts: &[CatalogLibrary]) -> Result<()> { /// Called before main constructs any runtime or threads. Credentials arrive /// only over stdin, and remain fixed for the lifetime of this process. -pub(crate) fn run() -> Result<()> { +pub(crate) fn run(level: crate::LogLevel) -> Result<()> { + // Handlers execute here, so `ApiError::internal` emits its `root_cause` + // line in this process. Without a subscriber the inherited stderr stayed + // empty and every worker-side failure reached the browser as a bare 500. + super::request_log::init_warehouse_tracing(level); let mut input = std::io::stdin().lock(); let mut output = std::io::stdout().lock(); let bootstrap: Bootstrap = read_frame(&mut input)?.context("missing worker bootstrap")?; @@ -318,7 +474,20 @@ pub(crate) fn run() -> Result<()> { job.body.len() <= 1024 * 1024, "worker request body too large" ); + let id = job + .headers + .iter() + .find(|(name, _)| name == "x-request-id") + .map(|(_, v)| v.clone()) + .unwrap_or_default(); + let progress = super::request_progress::Progress::new( + id, + job.method.clone(), + job.uri.split('?').next().unwrap_or_default().to_owned(), + true, + ); let result = runtime.block_on(async { + let operation = async { let mut builder = axum::http::Request::builder() .method(job.method.as_str()) .uri(job.uri); @@ -327,7 +496,8 @@ pub(crate) fn run() -> Result<()> { } let response = warehouse .router() - .oneshot(builder.body(axum::body::Body::from(job.body))?) + .oneshot({ let mut request=builder.body(axum::body::Body::from(job.body))?; + request.extensions_mut().insert(progress.clone()); request }) .await?; let status = response.status().as_u16(); let headers = response @@ -359,8 +529,18 @@ pub(crate) fn run() -> Result<()> { headers, body, }) + }; + tokio::pin!(operation); + let mut tick = tokio::time::interval(Duration::from_millis(250)); + loop { + tokio::select! { + result = &mut operation => break result, + _ = tick.tick() => { output.write_all(&encode(&WorkerEvent::Progress(progress.snapshot()))?)?; output.flush()?; } + } + } })?; - output.write_all(&encode(&result)?)?; + output.write_all(&encode(&WorkerEvent::Progress(progress.snapshot()))?)?; + output.write_all(&encode(&WorkerEvent::Response(result))?)?; output.flush()?; } Ok(()) @@ -369,6 +549,7 @@ pub(crate) fn run() -> Result<()> { #[cfg(test)] mod tests { use super::*; + use std::collections::HashMap; #[test] fn frames_reject_truncation_and_oversize_and_preserve_boundaries() { @@ -383,21 +564,67 @@ mod tests { } #[tokio::test] - async fn pool_bounds_admission_and_never_evicts_queued_scopes() { + async fn pool_scales_reuses_bounds_and_wakes_waiters() { let pool = WorkerPool::default(); let permits: Vec<_> = (0..MAX_REQUESTS).map(|_| pool.admit().unwrap()).collect(); assert!(pool.admit().is_err()); drop(permits); assert!(pool.admit().is_ok()); - let mut slots = Vec::new(); - for index in 0..MAX_WORKERS { - slots.push(pool.slot(&index.to_string()).await.unwrap()); + + let first = pool.lease("same").await.unwrap(); + let second = pool.lease("same").await.unwrap(); + assert_eq!(pool.state.slots.lock().await.len(), 2); + let first_slot = OwnedMutexGuard::mutex(first.guard.as_ref().unwrap()).clone(); + drop(first); + let reused = pool.lease("same").await.unwrap(); + assert!(Arc::ptr_eq( + &first_slot, + OwnedMutexGuard::mutex(reused.guard.as_ref().unwrap()) + )); + let mut busy = vec![second, reused]; + for _ in busy.len()..MAX_WORKERS_PER_SCOPE { + busy.push(pool.lease("same").await.unwrap()); + } + assert!(pool.state.try_lease("same").await.unwrap().is_none()); + for index in MAX_WORKERS_PER_SCOPE..MAX_WORKERS { + busy.push(pool.lease(&format!("other-{index}")).await.unwrap()); } - assert!(Arc::ptr_eq(&slots[0], &pool.slot("0").await.unwrap())); - assert!(pool.slot("overflow").await.is_err()); - slots.remove(0); - assert!(pool.slot("replacement").await.is_ok()); - assert_eq!(pool.slots.lock().await.len(), MAX_WORKERS); + assert!(pool.state.try_lease("overflow").await.unwrap().is_none()); + let waiting = pool.lease("same"); + tokio::pin!(waiting); + assert!( + tokio::time::timeout(Duration::from_millis(10), &mut waiting) + .await + .is_err() + ); + busy.remove(0); + let lease = tokio::time::timeout(Duration::from_secs(1), waiting) + .await + .unwrap() + .unwrap(); + assert_eq!(pool.state.slots.lock().await.len(), MAX_WORKERS); + drop(lease); + let replacement = pool.lease("replacement").await.unwrap(); + assert_eq!(pool.state.slots.lock().await.len(), MAX_WORKERS); + drop(replacement); + } + + #[tokio::test] + async fn reaper_removes_only_idle_expired_slots() { + let pool = WorkerPool::default(); + let mut busy = pool.lease("scope").await.unwrap(); + busy.guard.as_mut().unwrap().idle_since = Instant::now() - WORKER_IDLE_TIMEOUT; + let idle = pool.lease("scope").await.unwrap(); + drop(idle); + { + let slots = pool.state.slots.lock().await; + slots[1].state.lock().await.idle_since = Instant::now() - WORKER_IDLE_TIMEOUT; + } + pool.state.reap_idle().await; + assert_eq!(pool.state.slots.lock().await.len(), 1); + drop(busy); + pool.state.reap_idle().await; + assert_eq!(pool.state.slots.lock().await.len(), 1); } #[cfg(unix)] @@ -415,13 +642,14 @@ mod tests { let input = child.stdin.take().unwrap(); let output = child.stdout.take().unwrap(); let pool = WorkerPool::default(); - let slot = pool.slot("test").await.unwrap(); - *slot.lock().await = Some(Worker { + let mut lease = pool.lease("test").await.unwrap(); + lease.guard.as_mut().unwrap().worker = Some(Worker { child, input, output, _home: home, }); + drop(lease); let result = tokio::time::timeout( Duration::from_millis(30), pool.execute( @@ -437,7 +665,8 @@ mod tests { ) .await; assert!(result.is_err()); - assert!(slot.lock().await.is_none()); + let lease = pool.lease("test").await.unwrap(); + assert!(lease.guard.as_ref().unwrap().worker.is_none()); tokio::time::timeout(Duration::from_secs(5), async { loop { // Signal zero observes process existence without sending a signal. @@ -454,7 +683,12 @@ mod tests { #[test] fn exec_environment_excludes_ambient_credentials() { let home = tempfile::tempdir().unwrap(); - let cmd = command(PathBuf::from("pchronicle"), home.path(), home.path()); + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + home.path(), + ); let env: HashMap<_, _> = cmd.as_std().get_envs().collect(); for key in [ "AWS_ACCESS_KEY_ID", @@ -476,6 +710,82 @@ mod tests { ); } + #[test] + fn exec_environment_names_a_surviving_block_cache() { + let home = tempfile::tempdir().unwrap(); + let blocks = tempfile::tempdir().unwrap(); + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + blocks.path(), + ); + let env: HashMap<_, _> = cmd.as_std().get_envs().collect(); + // Without this the cache resolves under the worker's throwaway HOME, so + // each worker refetches every index page the last one already read. + assert_eq!( + env[std::ffi::OsStr::new("PCHRONICLE_LANCE_CACHE_DIR")], + Some(blocks.path().as_os_str()) + ); + assert_ne!( + env[std::ffi::OsStr::new("PCHRONICLE_LANCE_CACHE_DIR")], + Some(home.path().as_os_str()) + ); + } + + #[test] + fn exec_environment_forwards_object_store_admission_tuning() { + let home = tempfile::tempdir().unwrap(); + // Workers issue the object-store reads, and `command` clears the + // environment. A knob missing from the allowlist silently tunes only + // the parent, which reads almost nothing. + // SAFETY: single-threaded test asserting how `command` forwards it. + unsafe { std::env::set_var("PCHRONICLE_OBJECT_STORE_CONCURRENCY", "8") }; + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + home.path(), + ); + let env: HashMap<_, _> = cmd.as_std().get_envs().collect(); + assert_eq!( + env[std::ffi::OsStr::new("PCHRONICLE_OBJECT_STORE_CONCURRENCY")], + Some(std::ffi::OsStr::new("8")) + ); + unsafe { std::env::remove_var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") }; + } + + #[test] + fn exec_arguments_carry_the_serve_log_level() { + let home = tempfile::tempdir().unwrap(); + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + home.path(), + ); + let args: Vec<_> = cmd + .as_std() + .get_args() + .map(|arg| arg.to_string_lossy().into_owned()) + .collect(); + // Handlers run in the child, so it must install a subscriber at the + // serve level; otherwise `root_cause` diagnostics are dropped there. + let level = args + .iter() + .position(|arg| arg == "--log-level") + .map(|index| args[index + 1].clone()); + assert_eq!( + level.as_deref(), + Some(super::super::request_log::initialized_log_level().as_arg()), + "{args:?}" + ); + assert!( + args.contains(&"--catalog-query-worker".to_owned()), + "{args:?}" + ); + } + #[test] fn mixed_s3_credentials_are_rejected_instead_of_using_first_key() { let library = CatalogLibrary { diff --git a/crates/persisting-pchronicle-cli/src/server/explorer.rs b/crates/persisting-pchronicle-cli/src/server/explorer.rs index d142fdf81..a5be52612 100644 --- a/crates/persisting-pchronicle-cli/src/server/explorer.rs +++ b/crates/persisting-pchronicle-cli/src/server/explorer.rs @@ -1,4 +1,4 @@ -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use persisting_pchronicle::model::EventRecord; use persisting_pchronicle::storage::{ @@ -217,6 +217,8 @@ pub(crate) struct TurnSearchStatus { #[derive(Clone, Debug, Serialize)] pub(crate) struct TurnExplorerPage { + #[serde(skip_serializing_if = "Option::is_none")] + pub(crate) analysis: Option, pub(crate) snapshot: PageSnapshot, pub(crate) records: Vec, pub(crate) search: TurnSearchStatus, @@ -423,7 +425,9 @@ pub(crate) fn run_page_with_fts( }) .collect::>(); let path_index_limit = 2_000usize; - let path_index = if records.len() > path_index_limit { + let path_index = if !needle.is_empty() { + Vec::new() + } else if records.len() > path_index_limit { // Huge compact-jsonl sources must not ship every identity into the // browser path explorer. Keep one representative per file plus a // bounded sample so WASM stays responsive. @@ -482,8 +486,12 @@ pub(crate) fn run_page_with_fts( ); RunExplorerPage { snapshot: page.snapshot, + path_index: if needle.is_empty() { + path_index + } else { + page.records.iter().map(|item| item.run.clone()).collect() + }, records: page.records, - path_index, search, } } @@ -513,15 +521,13 @@ pub(crate) fn analyze( let mut kinds = BTreeMap::::new(); let mut model_groups = BTreeMap::::new(); let mut error_count = 0usize; + let index = EventIndex::new(events); for item in turns { if let Some(timestamp) = item.turn.timestamp.as_ref() { timestamps.push(timestamp.clone()); } - let linked = events - .iter() - .filter(|event| item.event_seqs.contains(&event.seq)) - .collect::>(); + let linked = index.linked(&item.event_seqs, events); let mut values = Vec::new(); if let Some(value) = &item.turn.metrics { values.push(value); @@ -675,53 +681,89 @@ pub(crate) fn analyze( } } -pub(crate) fn turn_page( +pub(crate) fn turn_list_with_search( turns: &[TrajectoryTurnView], events: &[EventRecord], q: Option<&str>, source: Option<&str>, - offset: usize, - limit: usize, -) -> ExplorerPage { + search: TurnSearchStatus, +) -> TurnExplorerPage { let needle = q.unwrap_or_default().trim().to_ascii_lowercase(); + let index = EventIndex::new(events); let records = turns .iter() - .filter(|item| source.is_none_or(|source| source == "all" || item.turn.source == source)) - .filter(|item| needle.is_empty() || searchable_turn(item).contains(&needle)) - .map(|item| turn_summary(item, events)) - .collect(); - paginate(records, offset, limit.clamp(1, 500)) -} - -pub(crate) fn turn_page_with_search( - turns: &[TrajectoryTurnView], - events: &[EventRecord], - q: Option<&str>, - source: Option<&str>, - offset: usize, - limit: usize, - search: TurnSearchStatus, -) -> TurnExplorerPage { - let page = turn_page(turns, events, q, source, offset, limit); + .filter(|item| { + source.is_none_or(|source| { + source.is_empty() || source == "all" || item.turn.source == source + }) && (needle.is_empty() || searchable_turn(item).contains(&needle)) + }) + .map(|item| turn_summary(item, events, &index)) + .collect::>(); + let total = records.len(); TurnExplorerPage { - snapshot: page.snapshot, - records: page.records, + analysis: None, + // Preserve the response envelope for existing clients; turns are + // returned in full, with no offset or per-request row cap. + snapshot: PageSnapshot { + offset: 0, + next_offset: total, + total, + has_more: false, + limit: total, + }, + records, search, } } +/// Positions of every event sequence inside a trajectory's event list. +/// +/// Turns reference their events by sequence, so resolving them by scanning the +/// whole list once per turn costs `turns × events` and dominates long +/// trajectories. Repeated sequences are all kept: a source may emit an id more +/// than once, and collapsing them would change token and tool aggregates. +struct EventIndex(HashMap>); + +impl EventIndex { + fn new(events: &[EventRecord]) -> Self { + let mut positions = HashMap::>::with_capacity(events.len()); + for (position, event) in events.iter().enumerate() { + positions.entry(event.seq).or_default().push(position); + } + Self(positions) + } + + fn linked<'a>(&self, seqs: &[u64], events: &'a [EventRecord]) -> Vec<&'a EventRecord> { + let mut positions = seqs + .iter() + .filter_map(|seq| self.0.get(seq)) + .flatten() + .copied() + .collect::>(); + // A turn may list its sequences out of order or repeat one; callers + // expect each event once, in event-list order, as a scan produced them. + positions.sort_unstable(); + positions.dedup(); + positions + .into_iter() + .map(|position| &events[position]) + .collect() + } +} + pub(crate) fn turn_detail( item: &TrajectoryTurnView, events: &[EventRecord], event_provenance: CatalogEventProvenance, ) -> TurnDetail { - let linked = events - .iter() - .filter(|event| item.event_seqs.contains(&event.seq)) + let index = EventIndex::new(events); + let linked = index + .linked(&item.event_seqs, events) + .into_iter() .cloned() .collect::>(); TurnDetail { - summary: turn_summary(item, events), + summary: turn_summary(item, events, &index), turn: item.turn.clone(), wire_tool_calls: item.wire_tool_calls.clone(), event_provenance, @@ -729,11 +771,12 @@ pub(crate) fn turn_detail( } } -fn turn_summary(item: &TrajectoryTurnView, events: &[EventRecord]) -> TurnSummary { - let linked = events - .iter() - .filter(|event| item.event_seqs.contains(&event.seq)) - .collect::>(); +fn turn_summary( + item: &TrajectoryTurnView, + events: &[EventRecord], + index: &EventIndex, +) -> TurnSummary { + let linked = index.linked(&item.event_seqs, events); let mut values = Vec::new(); if let Some(value) = &item.turn.metrics { values.push(value); @@ -1237,6 +1280,44 @@ mod tests { ); } + #[test] + fn event_index_matches_a_full_scan_for_duplicate_and_unordered_sequences() { + let events = [7u64, 3, 7, 5, 3, 9] + .into_iter() + .map(|seq| EventRecord { + identity: Default::default(), + seq, + source: "test".into(), + kind: "note".into(), + timestamp: None, + session_id: None, + agent_id: None, + parent_uuid: None, + trace_id: None, + call_id: None, + subagent_id: None, + parent_agent_id: None, + branch: None, + parent_call_id: None, + payload: serde_json::json!({ "seq": seq }), + }) + .collect::>(); + let index = EventIndex::new(&events); + for seqs in [ + vec![], + vec![3], + vec![9, 3, 7], + vec![5, 5], + vec![3, 4, 7, 11], + ] { + let scanned = events + .iter() + .filter(|event| seqs.contains(&event.seq)) + .collect::>(); + assert_eq!(index.linked(&seqs, &events), scanned, "{seqs:?}"); + } + } + #[test] fn percentiles_report_coverage_without_inventing_missing_samples() { let stats = metric_stats(vec![10.0, 20.0, 30.0, 40.0], 8); diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 157701369..db27c1db6 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -9,11 +9,12 @@ mod physical; pub(crate) mod problem; mod query_admission; pub(crate) mod request_log; +mod request_progress; mod ui_cache; -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use std::net::SocketAddr; -use std::sync::Arc; +use std::sync::{Arc, Mutex, Weak}; use std::time::{Duration, Instant}; use anyhow::Context; @@ -33,15 +34,17 @@ use persisting_pchronicle::search::storyline_steps_fts_available; #[cfg(test)] use persisting_pchronicle::storage::StoryCoords; use persisting_pchronicle::storage::{ - CatalogErrorPolicy, CatalogEventProvenance, CatalogSnapshotOptions, CatalogStorylineKey, - DEFAULT_DATASET_NAME, DatasetCatalogSnapshot, DatasetMount, + CatalogConsistency, CatalogErrorPolicy, CatalogEventProvenance, CatalogSnapshotOptions, + CatalogStorylineKey, DEFAULT_DATASET_NAME, DatasetCatalogSnapshot, DatasetMount, + with_background_object_store_io, }; use serde::{Deserialize, Serialize}; use serde_json::{Value, json}; use acceleration::{AccelerationStatus, ServerAcceleration}; use problem::{ - ApiError, CHAIN_LIMIT, LOG_TARGET, QUERY_LOG_LIMIT, ROOT_CAUSE_LIMIT, truncate_utf8, + ApiError, CHAIN_LIMIT, ExecutionStage, LOG_TARGET, QUERY_LOG_LIMIT, ROOT_CAUSE_LIMIT, + truncate_utf8, }; use request_log::{FtsDiagnostics, RequestId, RequestMetrics}; @@ -52,6 +55,9 @@ fn fail(request_id: &RequestId, handler: &'static str, error: anyhow::Error) -> #[cfg(test)] use problem::BoundaryCode; +type CachedTrajectory = (String, Arc); +type TrajectoryCache = Arc>>; + #[derive(Clone)] struct AppState { config: Arc, @@ -59,7 +65,8 @@ struct AppState { /// Serializes global refreshes and stores the next automatic retry time. catalog_refresh: Arc>, catalog_refresh_interval: Duration, - trajectory_cache: Arc>>, + trajectory_cache: TrajectoryCache, + trajectory_flights: Arc>>>, /// Gateway-backed Warehouses read canonical events from the latest /// manifest for single-trace observation, independent of projection idle. live_reads: bool, @@ -67,11 +74,34 @@ struct AppState { catalog_query_worker: bool, catalog_workers: Arc, browse_mounts: Arc>, + request_progress: Arc, browse: Arc>, scoped_queries: Arc, } const DEFAULT_CATALOG_REFRESH_INTERVAL: Duration = Duration::from_secs(30); +// Return a structured timeout before the catalog worker's 60s transport cutoff. +const RUNS_SCAN_TIMEOUT: Duration = Duration::from_secs(50); +const TRAJECTORY_READ_TIMEOUT: Duration = Duration::from_secs(50); +// Backstop for the cached scan, which also runs as a background refresh with no +// request deadline. Kept above `RUNS_SCAN_TIMEOUT` so a foreground request +// reports the endpoint's own timeout instead of this inner one. +const RUNS_SCAN_BACKSTOP: Duration = Duration::from_secs(55); + +async fn with_deadline( + timeout: Duration, + request_id: &RequestId, + timeout_error: fn() -> ApiError, + operation: impl std::future::Future>, +) -> Result { + tokio::time::timeout(timeout, operation) + .await + .map_err(|_| { + timeout_error() + .with_request_id(request_id.as_str()) + .with_stage(ExecutionStage::Query) + })? +} #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct HomeLink { @@ -217,11 +247,13 @@ fn app_state_with_catalog_refresh_interval( catalog_refresh: Arc::new(tokio::sync::Mutex::new(Instant::now())), catalog_refresh_interval, trajectory_cache: Arc::new(tokio::sync::RwLock::new(None)), + trajectory_flights: Arc::new(Mutex::new(HashMap::new())), live_reads: false, catalog_acl: None, catalog_query_worker: false, catalog_workers: Arc::new(catalog_worker::WorkerPool::default()), browse_mounts: Arc::new(browse_mounts), + request_progress: Arc::new(request_progress::Registry::default()), browse: Arc::new(tokio::sync::OnceCell::new()), scoped_queries: Arc::new(query_admission::ScopedQueries::default()), } @@ -344,6 +376,7 @@ async fn browse_coordinator(state: &AppState) -> &ui_cache::BrowseCoordinator { fn api_routes() -> Router { Router::new() .route("/health", get(warehouse_health)) + .route("/requests/{id}", get(request_progress::get)) .route("/ui", get(ui_config)) .route("/runs", get(runs)) .route("/explorer/runs", get(explorer_runs)) @@ -382,7 +415,8 @@ fn finish_routes(state: AppState) -> Router { state.clone(), catalog::catalog_data_plane_layer, )) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + state.clone(), request_log::warehouse_request_layer, )) .with_state(state) @@ -468,6 +502,7 @@ async fn ui_config(State(state): State) -> Json { async fn build_catalog_runtime( config: &ChronicleServerConfig, ) -> anyhow::Result> { + request_progress::phase("source_metadata"); let snapshot = Arc::new( DatasetCatalogSnapshot::discover( config.datasets.clone(), @@ -476,6 +511,7 @@ async fn build_catalog_runtime( ) .await?, ); + request_progress::phase("query"); let engine = Arc::new( snapshot .clone() @@ -495,6 +531,7 @@ async fn build_scoped_query_runtime( scope: persisting_pchronicle::storage::QueryScope, cached_files: Vec, ) -> anyhow::Result> { + request_progress::phase("source_metadata"); let target = match scope.source_file.clone() { Some(file) => persisting_pchronicle::storage::ResolveTarget::Dataset { mount: scope.dataset.clone(), @@ -522,6 +559,7 @@ async fn build_scoped_query_runtime( config.catalog_options, ); let snapshot = Arc::new(resolver.resolve(target, mode, &cached_datasets).await?); + request_progress::phase("query"); let engine = Arc::new( snapshot .clone() @@ -543,17 +581,42 @@ async fn current_catalog( if let Some(runtime) = state.catalog.read().await.as_ref() { return Ok(Arc::clone(runtime)); } + request_progress::phase("catalog_wait"); let _refresh = state.catalog_refresh.lock().await; if let Some(runtime) = state.catalog.read().await.as_ref() { return Ok(Arc::clone(runtime)); } let runtime = build_catalog_runtime(&state.config) .await - .map_err(|error| fail(request_id, "current_catalog", error))?; + .map_err(|error| { + fail(request_id, "current_catalog", error).with_stage(ExecutionStage::Catalog) + })?; *state.catalog.write().await = Some(Arc::clone(&runtime)); Ok(runtime) } +// An exact file request must not wait for discovery of unrelated sources. +async fn catalog_for_source( + state: &AppState, + dataset: Option<&str>, + file: &str, + request_id: &RequestId, +) -> Result, ApiError> { + let Some(dataset) = dataset else { + return current_catalog(state, request_id).await; + }; + build_scoped_query_runtime( + &state.config, + persisting_pchronicle::storage::QueryScope { + dataset: dataset.to_owned(), + source_file: Some(file.to_owned()), + }, + Vec::new(), + ) + .await + .map_err(|error| fail(request_id, "catalog_for_source", error)) +} + async fn current_catalog_for_runs( state: &AppState, request_id: &RequestId, @@ -621,7 +684,7 @@ async fn rebuild_catalog_for_runs( #[derive(Debug, Serialize)] struct CatalogResponse { - consistency: &'static str, + consistency: CatalogConsistency, snapshot_id: String, created_at: String, default_dataset: Option, @@ -632,7 +695,7 @@ struct CatalogResponse { fn catalog_response(state: &AppState, runtime: &CatalogRuntime) -> CatalogResponse { CatalogResponse { - consistency: "per_source_pinned", + consistency: CatalogConsistency::Pinned, snapshot_id: runtime.snapshot.snapshot_id().to_string(), created_at: runtime.snapshot.created_at().to_string(), default_dataset: runtime.snapshot.default_dataset().map(str::to_owned), @@ -660,10 +723,9 @@ async fn refresh_catalog( let warehouse = PreparedWarehouse { state: state.clone(), }; - let runtime = warehouse - .refresh_runtime() - .await - .map_err(|error| fail(&request_id, "refresh_catalog", error))?; + let runtime = warehouse.refresh_runtime().await.map_err(|error| { + fail(&request_id, "refresh_catalog", error).with_stage(ExecutionStage::Catalog) + })?; Ok(Json(catalog_response(&state, &runtime))) } @@ -673,7 +735,13 @@ async fn runs( metrics: RequestMetrics, ) -> Result>, ApiError> { let started = Instant::now(); - let summaries = load_run_summaries(&state, None, None, &request_id, Some(&metrics)).await?; + let summaries = with_deadline( + RUNS_SCAN_TIMEOUT, + &request_id, + ApiError::runs_timeout, + load_run_summaries(&state, None, None, &request_id, Some(&metrics)), + ) + .await?; metrics.record("summary_total", started); Ok(Json(summaries)) } @@ -693,45 +761,52 @@ async fn load_run_summaries( }; let config = state.config.clone(); let query_scope = scope.clone(); + let mount = config + .datasets + .iter() + .find(|mount| mount.name == dataset) + .ok_or_else(|| { + ApiError::not_found("dataset was not found").with_request_id(request_id.as_str()) + })?; + // Browse initialization and its disk projection are optional for Runs. let cached_files = if file.is_none() && !state.live_reads { - browse_coordinator(state) - .await - .cached_source_paths( - config - .datasets - .iter() - .find(|mount| mount.name == dataset) - .expect("validated dataset mount"), - ) - .await + match state.browse.get() { + Some(browse) => browse.cached_source_paths(mount).await, + None => Vec::new(), + } } else { Vec::new() }; let query_metrics = metrics.cloned(); let execute = move |background: bool| async move { - // Background work must not be attributed to the triggering HTTP request. - let metrics = query_metrics.filter(|_| !background); - let phase = Instant::now(); - let runtime = - build_scoped_query_runtime(&config, query_scope.clone(), cached_files.clone()) + let summaries = tokio::time::timeout(RUNS_SCAN_BACKSTOP, async { + // Background work must not be attributed to the triggering HTTP request. + let metrics = query_metrics.filter(|_| !background); + let phase = Instant::now(); + let runtime = + build_scoped_query_runtime(&config, query_scope.clone(), cached_files.clone()) + .await?; + if let Some(metrics) = &metrics { + metrics.record("summary_catalog", phase); + } + let phase = Instant::now(); + let summaries = runtime + .acceleration + .scoped_run_summaries( + &runtime.snapshot, + &runtime.engine, + Some(&query_scope.dataset), + query_scope.source_file.as_deref(), + ) .await?; - if let Some(metrics) = &metrics { - metrics.record("summary_catalog", phase); - } - let phase = Instant::now(); - let summaries = runtime - .acceleration - .scoped_run_summaries( - &runtime.snapshot, - &runtime.engine, - Some(&query_scope.dataset), - query_scope.source_file.as_deref(), - ) - .await; - if let Some(metrics) = &metrics { - metrics.record("summary_sql", phase); - } - summaries + if let Some(metrics) = &metrics { + metrics.record("summary_sql", phase); + } + anyhow::Ok(summaries) + }) + .await + .context("runs scan timed out")??; + Ok(summaries) }; let started = Instant::now(); let result = if state.live_reads { @@ -765,13 +840,12 @@ async fn load_run_summaries( let summaries = runtime .acceleration .scoped_run_summaries(&runtime.snapshot, &runtime.engine, dataset, file) - .await; + .await + .map_err(|error| fail(request_id, "load_run_summaries", error))?; if let Some(metrics) = metrics { metrics.record("summary_sql", phase); } - summaries - .map(|summaries| summaries.as_ref().clone()) - .map_err(|error| fail(request_id, "load_run_summaries", error)) + Ok(summaries.as_ref().clone()) } fn api_query(query: Result, QueryRejection>) -> Result { @@ -781,7 +855,6 @@ fn api_query(query: Result, QueryRejection>) -> Result } const EXPLORER_RUN_MATCH_IDENTITY_MAX_ROWS: u64 = 50_000; -const EXPLORER_RUN_MATCH_PREVIEW_LIMIT: u64 = 512; fn explorer_run_identity_sql(dataset: &str, table: &str, predicate: &str) -> String { format!( @@ -792,11 +865,54 @@ fn explorer_run_identity_sql(dataset: &str, table: &str, predicate: &str) -> Str fn explorer_run_preview_sql( dataset: &str, table: &str, - select: &str, + columns: &[&str], predicate: &str, - limit: u64, + runs: &[&RunSummary], ) -> String { - format!("SELECT {select} FROM {dataset}.{table} WHERE ({predicate}) LIMIT {limit}") + let scope = runs + .iter() + .map(|run| { + format!( + "(_file_ = {} AND document_id = {})", + crate::sql_string(&run.file), + crate::sql_string(&run.document_id), + ) + }) + .collect::>() + .join(" OR "); + let columns = columns.join(", "); + let order = if table == "steps" { + "step_id" + } else { + "document_id" + }; + // One matching row per visible run. A global step LIMIT lets one long run + // consume every preview and leaves the rest of the page without evidence. + format!( + "SELECT source_path, document_id, {columns} FROM (SELECT _file_ AS source_path, document_id, {columns}, ROW_NUMBER() OVER (PARTITION BY _file_, document_id ORDER BY {order}) AS match_rank FROM {dataset}.{table} WHERE ({predicate}) AND ({scope})) AS matched WHERE match_rank = 1 LIMIT {}", + runs.len(), + ) +} + +fn explorer_preview_columns( + expression: &persisting_pchronicle::search::FindExpr, +) -> Vec<&'static str> { + let mut predicates = Vec::new(); + crate::collect_text_predicates(expression, &mut predicates); + let mut columns = predicates + .iter() + .flat_map(|item| item.field.columns().iter().copied()) + .collect::>(); + if columns.is_empty() { + columns.extend(if expression.has_step_json() { + &["message_value", "metrics"][..] + } else { + &["task", "prompt", "notes", "agent_name", "agent_model_name"][..] + }); + } + columns.sort_unstable(); + columns.dedup(); + columns } async fn explorer_query_jsonl( @@ -847,7 +963,7 @@ async fn try_compact_jsonl_runs_page( .map(str::trim) .filter(|value| !value.is_empty() && *value != "all"); - let runtime = current_catalog(state, request_id).await?; + let runtime = catalog_for_source(state, dataset_filter, file_filter, request_id).await?; let mut matched: Option<(String, String, Option)> = None; for dataset in runtime.snapshot.datasets() { if dataset_filter.is_some_and(|filter| dataset.mount.name != filter) { @@ -950,6 +1066,45 @@ async fn try_compact_jsonl_runs_page( })) } +async fn open_storyline_source( + state: &AppState, + dataset: &str, + file: &str, + request_id: &RequestId, +) -> Result, ApiError> { + request_progress::phase("source_metadata"); + let Some(mount) = state + .config + .datasets + .iter() + .find(|mount| mount.name == dataset) + else { + return Ok(None); + }; + let location = persisting_pchronicle::storage::DatasetLocation::parse_with_backend( + &mount.uri, + mount.backend().cloned(), + ) + .map_err(|error| fail(request_id, "storyline_source", error))?; + if location + .probe_nav_dataset_kind(file) + .await + .map_err(|error| fail(request_id, "storyline_source", error))? + != Some("storyline") + { + return Ok(None); + } + let uri = format!( + "{}/{}", + mount.uri.trim_end_matches('/'), + file.trim_matches('/') + ); + persisting_pchronicle::storage::StorylineLanceStore::open_uri(&uri) + .await + .map(Some) + .map_err(|error| fail(request_id, "storyline_source", error)) +} + /// Directory mounts only expose immediate children in the catalog. Nested /// Storyline leaves reached via explorer navigation are therefore absent from /// SQL acceleration. When the client asks for an exact `file=` that is a @@ -983,42 +1138,10 @@ async fn try_on_demand_storyline_runs_page( return Ok(None); }; - let runtime = current_catalog(state, request_id).await?; - let Some(dataset) = runtime.snapshot.dataset(dataset_name) else { + let Some(store) = open_storyline_source(state, dataset_name, file, request_id).await? else { return Ok(None); }; - // Prefer catalog-backed sources; only fall through for nested Directory paths. - if dataset.sources.iter().any(|source| { - source.kind != persisting_pchronicle::storage::CatalogSourceKind::Directory - && (source.file == file || source.file.starts_with(&format!("{file}/"))) - }) { - return Ok(None); - } - let under_directory = dataset.sources.iter().any(|source| { - source.kind == persisting_pchronicle::storage::CatalogSourceKind::Directory - && (file == source.file || file.starts_with(&format!("{}/", source.file))) - }); - if !under_directory && !dataset.sources.is_empty() { - return Ok(None); - } - - let location = persisting_pchronicle::storage::DatasetLocation::parse(&dataset.mount.uri) - .map_err(|error| fail(request_id, "explorer_runs", error))?; - let kind = location - .probe_nav_dataset_kind(file) - .await - .map_err(|error| fail(request_id, "explorer_runs", error))?; - if kind != Some("storyline") { - return Ok(None); - } - let uri = format!( - "{}/{}", - dataset.mount.uri.trim_end_matches('/'), - file.trim_matches('/') - ); - let store = persisting_pchronicle::storage::StorylineLanceStore::open_uri(&uri) - .await - .map_err(|error| fail(request_id, "explorer_runs", error))?; + request_progress::phase("storage_read"); let Some((_generation, ids)) = store .document_ids_snapshot() .await @@ -1105,11 +1228,28 @@ async fn explorer_runs( fts: FtsDiagnostics, query: Result, QueryRejection>, ) -> Result, ApiError> { + with_deadline( + RUNS_SCAN_TIMEOUT, + &request_id, + ApiError::runs_timeout, + explorer_runs_inner(State(state), request_id.clone(), metrics, fts, query), + ) + .await +} + +async fn explorer_runs_inner( + State(state): State, + request_id: RequestId, + metrics: RequestMetrics, + fts: FtsDiagnostics, + query: Result, QueryRejection>, +) -> Result, ApiError> { + request_progress::phase("query"); let query = api_query(query)?; - if let Some(page) = try_compact_jsonl_runs_page(&state, &query, &request_id).await? { + if let Some(page) = try_on_demand_storyline_runs_page(&state, &query, &request_id).await? { return Ok(Json(page)); } - if let Some(page) = try_on_demand_storyline_runs_page(&state, &query, &request_id).await? { + if let Some(page) = try_compact_jsonl_runs_page(&state, &query, &request_id).await? { return Ok(Json(page)); } let dataset_filter = query @@ -1127,12 +1267,23 @@ async fn explorer_runs( ) .await?; metrics.record("summary_total", started); + let mut preview_plan = None; let (fts_matches, fts_available, search_mode) = if query .q .as_deref() .is_some_and(|value| !value.trim().is_empty()) { - let runtime = current_catalog(&state, &request_id).await?; + let file_filter = query + .file + .as_deref() + .map(str::trim) + .filter(|file| !file.is_empty()); + let runtime = match (dataset_filter, file_filter) { + (Some(dataset), Some(file)) => { + catalog_for_source(&state, Some(dataset), file, &request_id).await? + } + _ => current_catalog(&state, &request_id).await?, + }; let raw = query.q.as_deref().unwrap_or_default().trim(); let expression = crate::combine_match_expressions(&[raw.to_owned()]) .map_err(|error| ApiError::invalid_request(error.to_string()))? @@ -1140,7 +1291,11 @@ async fn explorer_runs( let (predicate, fts_available, fts_errors) = crate::find_expression_predicate_for_dataset( &runtime.snapshot, &expression, - None, + query + .file + .as_deref() + .map(str::trim) + .filter(|file| !file.is_empty()), dataset_filter, ) .await @@ -1163,14 +1318,6 @@ async fn explorer_runs( if dataset_filter.is_some_and(|filter| dataset.mount.name != filter) { continue; } - let select = if table == "steps" { - // Keep all searchable step fields available to the preview - // selector. A COALESCE expression would hide a hit in (for - // example) reasoning_content behind a non-empty message. - "_file_ AS source_path, document_id, message_value, reasoning_content, observation, prompt, model_name" - } else { - "_file_ AS source_path, document_id, task, prompt, notes, agent_name, agent_model_name" - }; let identity_sql = explorer_run_identity_sql(&dataset.mount.name, table, &predicate); let identity_jsonl = explorer_query_jsonl( &runtime.engine, @@ -1199,46 +1346,14 @@ async fn explorer_runs( let identity = format!("{}\u{1f}{}\u{1f}{}", dataset.mount.name, file, document_id); matches.entry(identity).or_insert_with(String::new); } - let preview_sql = explorer_run_preview_sql( - &dataset.mount.name, - table, - select, - &predicate, - EXPLORER_RUN_MATCH_PREVIEW_LIMIT, - ); - let preview_jsonl = explorer_query_jsonl( - &runtime.engine, - &preview_sql, - EXPLORER_RUN_MATCH_PREVIEW_LIMIT, - &request_id, - ) - .await?; - for line in preview_jsonl.lines().filter(|line| !line.trim().is_empty()) { - let row: Value = serde_json::from_str(line).map_err(|error| { - fail( - &request_id, - "explorer_runs", - anyhow::anyhow!("decode run search preview: {error}"), - ) - })?; - let Some(file) = row.get("source_path").and_then(Value::as_str) else { - continue; - }; - let Some(document_id) = row.get("document_id").and_then(Value::as_str) else { - continue; - }; - let identity = format!("{}\u{1f}{}\u{1f}{}", dataset.mount.name, file, document_id); - if !matches.contains_key(&identity) { - continue; - } - let preview = search_preview_from_row(&row, raw, table); - if let Some(existing) = matches.get_mut(&identity) - && existing.is_empty() - { - *existing = preview; - } - } } + preview_plan = Some(( + runtime, + table, + explorer_preview_columns(&expression), + predicate, + raw.to_owned(), + )); let mode = if expression.has_text() && expression.has_json() { "fts+json" } else if expression.has_text() { @@ -1273,7 +1388,7 @@ async fn explorer_runs( } (BTreeMap::new(), fts_available, "none") }; - Ok(Json(explorer::run_page_with_fts( + let mut page = explorer::run_page_with_fts( summaries, &query, &fts_matches, @@ -1282,14 +1397,60 @@ async fn explorer_runs( mode: search_mode, tokenizer: fts_available.then_some("jieba"), }, - ))) + ); + if let Some((runtime, table, columns, predicate, raw)) = preview_plan { + for dataset in runtime.snapshot.datasets() { + let runs = page + .records + .iter() + .filter(|item| item.run.dataset == dataset.mount.name) + .map(|item| &item.run) + .collect::>(); + if runs.is_empty() { + continue; + } + let sql = + explorer_run_preview_sql(&dataset.mount.name, table, &columns, &predicate, &runs); + let jsonl = + explorer_query_jsonl(&runtime.engine, &sql, runs.len() as u64, &request_id).await?; + for line in jsonl.lines().filter(|line| !line.trim().is_empty()) { + let row: Value = serde_json::from_str(line) + .map_err(|error| fail(&request_id, "explorer_runs_preview", error.into()))?; + let file = row.get("source_path").and_then(Value::as_str); + let document = row.get("document_id").and_then(Value::as_str); + if let Some(item) = page.records.iter_mut().find(|item| { + item.run.dataset == dataset.mount.name + && Some(item.run.file.as_str()) == file + && Some(item.run.document_id.as_str()) == document + }) { + item.search_preview = Some(search_preview_from_row(&row, &raw, table)); + } + } + } + } + Ok(Json(page)) } -fn search_preview_text(raw: &str) -> String { - // The API deliberately returns the complete normalized field. The Web - // client owns the viewport-sized excerpt so it can guarantee that the - // matched term remains visible and highlighted. - crate::find_preview_text(raw) +fn search_preview_text(raw: &str, query: &str) -> String { + const MAX_CHARS: usize = 320; + let text = crate::find_preview_text(raw); + let needle = preview_needle(query).to_ascii_lowercase(); + let hit = if needle.is_empty() { + None + } else { + text.to_ascii_lowercase().find(&needle) + }; + let start = hit + .map(|offset| text[..offset].chars().count().saturating_sub(MAX_CHARS / 3)) + .unwrap_or(0); + let mut excerpt: String = text.chars().skip(start).take(MAX_CHARS).collect(); + if start + excerpt.chars().count() < text.chars().count() { + excerpt.push('…'); + } + if start > 0 { + excerpt.insert(0, '…'); + } + excerpt } fn search_preview_from_row(row: &Value, query: &str, table: &str) -> String { @@ -1300,6 +1461,8 @@ fn search_preview_from_row(row: &Value, query: &str, table: &str) -> String { "observation", "prompt", "model_name", + "env", + "metrics", ] } else { &["task", "prompt", "notes", "agent_name", "agent_model_name"] @@ -1310,7 +1473,7 @@ fn search_preview_from_row(row: &Value, query: &str, table: &str) -> String { let Some(value) = row.get(*column).and_then(search_preview_raw_value) else { continue; }; - let preview = search_preview_text(&value); + let preview = search_preview_text(&value, query); if fallback.is_none() && !preview.is_empty() { fallback = Some(preview.clone()); } @@ -1378,6 +1541,7 @@ async fn explorer_tree( ); let Some(name) = dataset else { let started = Instant::now(); + request_progress::phase("browse_cache"); let view = browse_coordinator(&state) .await .roots(&state.browse_mounts) @@ -1412,7 +1576,9 @@ async fn explorer_tree( .await .tree(mount, prefix) .await - .map_err(|error| fail(&request_id, "explorer_tree", error))?; + .map_err(|error| { + fail(&request_id, "explorer_tree", error).with_stage(ExecutionStage::Manifest) + })?; metrics.record("browse", started); Ok(Json(serde_json::to_value(view).unwrap())) } @@ -1528,54 +1694,17 @@ async fn try_resolve_on_demand_storyline_run( return Ok(None); } - let runtime = current_catalog(state, request_id).await?; - let Some(dataset) = runtime.snapshot.dataset(dataset_name) else { + let Some(store) = open_storyline_source(state, dataset_name, file, request_id).await? else { return Ok(None); }; - let exact_source = dataset.sources.iter().any(|source| { - source.kind != persisting_pchronicle::storage::CatalogSourceKind::Directory - && source.file == file - }); - if dataset.sources.iter().any(|source| { - source.kind != persisting_pchronicle::storage::CatalogSourceKind::Directory - && source.file.starts_with(&format!("{file}/")) - }) { - return Ok(None); - } - let under_directory = dataset.sources.iter().any(|source| { - source.kind == persisting_pchronicle::storage::CatalogSourceKind::Directory - && (file == source.file || file.starts_with(&format!("{}/", source.file))) - }); - if !under_directory && !exact_source && !dataset.sources.is_empty() { - return Ok(None); - } - - let location = persisting_pchronicle::storage::DatasetLocation::parse(&dataset.mount.uri) - .map_err(|error| fail(request_id, "resolve_run", error))?; - if location - .probe_nav_dataset_kind(file) - .await - .map_err(|error| fail(request_id, "resolve_run", error))? - != Some("storyline") - { - return Ok(None); - } - let uri = format!( - "{}/{}", - dataset.mount.uri.trim_end_matches('/'), - file.trim_matches('/') - ); - let store = persisting_pchronicle::storage::StorylineLanceStore::open_uri(&uri) - .await - .map_err(|error| fail(request_id, "resolve_run", error))?; - let Some((_generation, ids)) = store - .document_ids_snapshot() + let Some((_generation, present)) = store + .contains_document(session_id) .await .map_err(|error| fail(request_id, "resolve_run", error))? else { return Ok(None); }; - if !ids.iter().any(|id| id == session_id) { + if !present { return Ok(None); } let path = explorer::explorer_run_path(dataset_name, file, session_id, session_id, None, None); @@ -1601,12 +1730,11 @@ async fn try_resolve_on_demand_storyline_run( } async fn load_on_demand_storyline_bundle( - state: &AppState, + runtime: &CatalogRuntime, run: &RunSummary, request_id: &RequestId, op: &'static str, ) -> Result, ApiError> { - let runtime = current_catalog(state, request_id).await?; let Some(dataset) = runtime.snapshot.dataset(&run.dataset) else { return Ok(None); }; @@ -1659,11 +1787,12 @@ async fn load_on_demand_storyline_bundle( async fn catalog_or_on_demand_trajectory_bundle( state: &AppState, + runtime: &CatalogRuntime, run: &RunSummary, request_id: &RequestId, op: &'static str, ) -> Result { - let runtime = current_catalog(state, request_id).await?; + request_progress::phase("storage_read"); let key = catalog_storyline_key(run); let catalog_result = if state.live_reads { runtime.snapshot.load_live_trajectory_bundle(&key).await @@ -1672,12 +1801,12 @@ async fn catalog_or_on_demand_trajectory_bundle( }; match catalog_result { Ok(Some(bundle)) => Ok(bundle), - Ok(None) => load_on_demand_storyline_bundle(state, run, request_id, op) + Ok(None) => load_on_demand_storyline_bundle(runtime, run, request_id, op) .await? .ok_or_else(|| ApiError::not_found("run was not found")), Err(error) => { if let Some(bundle) = - load_on_demand_storyline_bundle(state, run, request_id, op).await? + load_on_demand_storyline_bundle(runtime, run, request_id, op).await? { Ok(bundle) } else { @@ -1733,8 +1862,10 @@ async fn load_events( request_id: &RequestId, ) -> Result { let run = resolve_run_summary(state, query, request_id, None).await?; + let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; let bundle = - catalog_or_on_demand_trajectory_bundle(state, &run, request_id, "load_events").await?; + catalog_or_on_demand_trajectory_bundle(state, &runtime, &run, request_id, "load_events") + .await?; let document = bundle.event_view; let offset = query .offset @@ -1796,8 +1927,10 @@ async fn storyline( ) -> Result, ApiError> { let query = api_query(query)?; let run = resolve_run_summary(&state, &query, &request_id, None).await?; + let runtime = catalog_for_source(&state, Some(&run.dataset), &run.file, &request_id).await?; let bundle = - catalog_or_on_demand_trajectory_bundle(&state, &run, &request_id, "storyline").await?; + catalog_or_on_demand_trajectory_bundle(&state, &runtime, &run, &request_id, "storyline") + .await?; Ok(Json( serde_json::to_value(bundle.storyline) .map_err(anyhow::Error::from) @@ -1924,126 +2057,171 @@ fn event_seqs_for_turn(turn: &StorylineTurn, by_call: &BTreeMap #[derive(Clone)] struct LoadedTrajectory { + runtime: Arc, run: RunSummary, event_provenance: CatalogEventProvenance, records: Vec, turns: Vec, } +type TrajectoryFlight = tokio::sync::OnceCell>; + async fn load_trajectory( state: &AppState, query: &SessionQuery, request_id: &RequestId, metrics: &RequestMetrics, -) -> Result { +) -> Result, ApiError> { let phase = Instant::now(); let run = resolve_run_summary(state, query, request_id, Some(metrics)).await?; metrics.record("resolve", phase); let phase = Instant::now(); - let runtime = current_catalog(state, request_id).await?; + let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; metrics.record("catalog", phase); let cache_key = format!( - "{}\u{1f}{}\u{1f}{}\u{1f}{}", + "{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}", runtime.snapshot.snapshot_id(), run.dataset, run.file, - run.session_id + run.session_id, + run.document_id ); - if !state.live_reads - && let Some((_, loaded)) = state - .trajectory_cache - .read() - .await - .as_ref() - .filter(|(key, _)| key == &cache_key) + if let Some((_, loaded)) = state + .trajectory_cache + .read() + .await + .as_ref() + .filter(|(key, _)| key == &cache_key) { metrics.record("trajectory_cache", Instant::now()); return Ok(loaded.clone()); } - if run.format.as_deref() == Some("compact-jsonl/v1") { - return Ok(LoadedTrajectory { - run, - event_provenance: CatalogEventProvenance::SyntheticFromStoryline, - records: Vec::new(), - turns: Vec::new(), - }); - } - let phase = Instant::now(); - let bundle = - catalog_or_on_demand_trajectory_bundle(state, &run, request_id, "load_trajectory").await?; - metrics.record("trajectory_read", phase); - let event_provenance = bundle.event_view.provenance; - let records = bundle.event_view.document.events; - let document = bundle.storyline; - // ACTF step records carry the first user input at document level when it - // is the baseline prompt. Preserve it on the first turn so Explorer can - // render the user side of the conversation without changing storage. - let document_prompt = document.prompt.clone(); - let mut by_call = BTreeMap::>::new(); - for event in &records { - if let Some(call_id) = event.call_id.as_ref().filter(|id| !id.is_empty()) { - by_call.entry(call_id.clone()).or_default().push(event.seq); - } - } - let turns = document - .turns - .into_iter() - .enumerate() - .map(|(turn_index, mut turn)| { - if turn_index == 0 && turn.source == "agent" && turn.prompt.is_none() { - turn.prompt = document_prompt.clone(); + let flight = { + let mut flights = state + .trajectory_flights + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + flights.retain(|_, entry| entry.strong_count() > 0); + if let Some(flight) = flights.get(&cache_key).and_then(Weak::upgrade) { + flight + } else { + let flight = Arc::new(tokio::sync::OnceCell::new()); + if flights.len() < 64 { + flights.insert(cache_key.clone(), Arc::downgrade(&flight)); } - let call_id = turn_call_id(&turn); - let event_seqs = event_seqs_for_turn(&turn, &by_call); - let mut wire_tool_calls = Vec::new(); - for event in records - .iter() - .filter(|event| event_seqs.contains(&event.seq)) - { - collect_wire_tool_calls(&event.payload, &mut wire_tool_calls); + flight + } + }; + // Cancelled initialization can be taken over by a waiter. Failures are not cached. + let loaded = flight + .get_or_try_init(|| async { + if run.format.as_deref() == Some("compact-jsonl/v1") { + return Ok(Arc::new(LoadedTrajectory { + runtime, + run, + event_provenance: CatalogEventProvenance::SyntheticFromStoryline, + records: Vec::new(), + turns: Vec::new(), + })); } - let mut seen = BTreeSet::new(); - wire_tool_calls.retain(|call| { - seen.insert(( - call.id.clone(), - call.name.clone(), - serde_json::to_string(&call.arguments).unwrap_or_default(), - )) - }); - // Tool outputs are canonicalized on the Storyline tool call, not - // necessarily on the event payload that supplied the call. Carry - // that result onto the wire call so AgenticMD can render it next - // to the matching command even when no observation envelope - // exists. - if let Some(native_calls) = turn.tool_calls.as_ref() { - for wire_call in &mut wire_tool_calls { - if wire_call.result.is_none() { - wire_call.result = wire_call.id.as_deref().and_then(|id| { - native_calls - .iter() - .find(|call| call.tool_call_id == id) - .and_then(|call| call.result.clone()) - }); - } + let phase = Instant::now(); + let bundle = catalog_or_on_demand_trajectory_bundle( + state, + &runtime, + &run, + request_id, + "load_trajectory", + ) + .await?; + metrics.record("trajectory_read", phase); + let event_provenance = bundle.event_view.provenance; + let records = bundle.event_view.document.events; + let document = bundle.storyline; + // ACTF step records carry the first user input at document level when it + // is the baseline prompt. Preserve it on the first turn so Explorer can + // render the user side of the conversation without changing storage. + let document_prompt = document.prompt.clone(); + let mut by_call = BTreeMap::>::new(); + let mut by_seq = HashMap::>::with_capacity(records.len()); + for (position, event) in records.iter().enumerate() { + if let Some(call_id) = event.call_id.as_ref().filter(|id| !id.is_empty()) { + by_call.entry(call_id.clone()).or_default().push(event.seq); } + by_seq.entry(event.seq).or_default().push(position); } - TrajectoryTurnView { - turn, - call_id, - event_seqs, - wire_tool_calls, - } + let turns = document + .turns + .into_iter() + .enumerate() + .map(|(turn_index, mut turn)| { + if turn_index == 0 && turn.source == "agent" && turn.prompt.is_none() { + turn.prompt = document_prompt.clone(); + } + let call_id = turn_call_id(&turn); + let event_seqs = event_seqs_for_turn(&turn, &by_call); + // Resolve this turn's events through the sequence index. + // Scanning every record per turn is quadratic and dominates + // long trajectories. + let mut positions = event_seqs + .iter() + .filter_map(|seq| by_seq.get(seq)) + .flatten() + .copied() + .collect::>(); + positions.sort_unstable(); + positions.dedup(); + let mut wire_tool_calls = Vec::new(); + for position in positions { + collect_wire_tool_calls(&records[position].payload, &mut wire_tool_calls); + } + let mut seen = BTreeSet::new(); + wire_tool_calls.retain(|call| { + seen.insert(( + call.id.clone(), + call.name.clone(), + serde_json::to_string(&call.arguments).unwrap_or_default(), + )) + }); + // Tool outputs are canonicalized on the Storyline tool call, not + // necessarily on the event payload that supplied the call. Carry + // that result onto the wire call so AgenticMD can render it next + // to the matching command even when no observation envelope + // exists. + if let Some(native_calls) = turn.tool_calls.as_ref() { + for wire_call in &mut wire_tool_calls { + if wire_call.result.is_none() { + wire_call.result = wire_call.id.as_deref().and_then(|id| { + native_calls + .iter() + .find(|call| call.tool_call_id == id) + .and_then(|call| call.result.clone()) + }); + } + } + } + TrajectoryTurnView { + turn, + call_id, + event_seqs, + wire_tool_calls, + } + }) + .collect(); + let loaded = Arc::new(LoadedTrajectory { + runtime, + run, + event_provenance, + records, + turns, + }); + Ok::<_, ApiError>(loaded) }) - .collect(); - let loaded = LoadedTrajectory { - run, - event_provenance, - records, - turns, - }; - if !state.live_reads { - *state.trajectory_cache.write().await = Some((cache_key, loaded.clone())); - } + .await? + .clone(); + // Memoize even under live catalog reads. The turns request and the follow-up + // /run statistics request share a worker when the pool is idle; without this + // memo the second request re-scans the same document. + *state.trajectory_cache.write().await = Some((cache_key, loaded.clone())); Ok(loaded) } @@ -2066,11 +2244,11 @@ async fn trajectory_view( .map(|calls| calls.len()) .sum(); Ok(Json(TrajectoryView { - run: loaded.run, + run: loaded.run.clone(), event_provenance: loaded.event_provenance, event_kind_counts, tool_call_count, - turns: loaded.turns, + turns: loaded.turns.clone(), })) } @@ -2081,13 +2259,31 @@ async fn explorer_run( query: Result, QueryRejection>, ) -> Result, ApiError> { let query = api_query(query)?; - let loaded = load_trajectory(&state, &query, &request_id, &metrics).await?; - Ok(Json(explorer::analyze( - loaded.run, - &loaded.turns, - &loaded.records, - loaded.event_provenance, - ))) + with_deadline( + TRAJECTORY_READ_TIMEOUT, + &request_id, + ApiError::trajectory_timeout, + async { + // Statistics read the whole run, so they admit through the + // background gate. Sharing the foreground gate lets this scan + // occupy every slot and starve the step page the reader is + // actually looking at, which both requests then time out behind. + let loaded = with_background_object_store_io(load_trajectory( + &state, + &query, + &request_id, + &metrics, + )) + .await?; + Ok(Json(explorer::analyze( + loaded.run.clone(), + &loaded.turns, + &loaded.records, + loaded.event_provenance, + ))) + }, + ) + .await } #[derive(Debug, Serialize)] @@ -2107,7 +2303,7 @@ async fn explorer_record( return Err(ApiError::not_found("run is not a compact JSONL record")); } let key = catalog_storyline_key(&run); - let record = current_catalog(&state, &request_id) + let record = catalog_for_source(&state, Some(&run.dataset), &run.file, &request_id) .await? .snapshot .compact_record(&key) @@ -2119,6 +2315,8 @@ async fn explorer_record( #[derive(Debug, Deserialize)] struct TurnsQuery { + #[serde(default)] + include_analysis: bool, dataset: Option, file: Option, run_id: Option, @@ -2127,8 +2325,6 @@ struct TurnsQuery { root_session_id: Option, q: Option, source: Option, - offset: Option, - limit: Option, } impl TurnsQuery { @@ -2152,12 +2348,30 @@ async fn explorer_turns( metrics: RequestMetrics, fts: FtsDiagnostics, query: Result, QueryRejection>, +) -> Result, ApiError> { + with_deadline( + TRAJECTORY_READ_TIMEOUT, + &request_id, + ApiError::trajectory_timeout, + explorer_turns_inner(State(state), request_id.clone(), metrics, fts, query), + ) + .await +} + +async fn explorer_turns_inner( + State(state): State, + request_id: RequestId, + metrics: RequestMetrics, + fts: FtsDiagnostics, + query: Result, QueryRejection>, ) -> Result, ApiError> { let query = api_query(query)?; + // Timeline and statistics are separate HTTP requests. Embedding analysis + // here forces the step list to wait on the same full-run scan. let session = query.session(); let loaded = load_trajectory(&state, &session, &request_id, &metrics).await?; let phase = Instant::now(); - let runtime = current_catalog(&state, &request_id).await?; + let runtime = &loaded.runtime; metrics.record("turn_catalog", phase); // Nested Directory Storylines are opened on-demand and are absent from the // prepared catalog; skip FTS path probing and keep in-memory turn pages. @@ -2203,7 +2417,6 @@ async fn explorer_turns( let expression = crate::combine_match_expressions(&[needle.to_owned()]) .map_err(|error| ApiError::invalid_request(error.to_string()))? .ok_or_else(|| ApiError::invalid_request("search query must not be empty"))?; - let runtime = current_catalog(&state, &request_id).await?; let phase = Instant::now(); let (predicate, available, fts_errors) = crate::find_expression_predicate_for_dataset( &runtime.snapshot, @@ -2271,19 +2484,25 @@ async fn explorer_turns( (loaded.turns.clone(), query.q.as_deref()) }; let phase = Instant::now(); - let page = explorer::turn_page_with_search( + let mut page = explorer::turn_list_with_search( &turns, &loaded.records, search_query, query.source.as_deref(), - query.offset.unwrap_or(0), - query.limit.unwrap_or(100), explorer::TurnSearchStatus { fts_available, mode: search_mode, tokenizer: fts_available.then_some("jieba"), }, ); + if query.include_analysis { + page.analysis = Some(explorer::analyze( + loaded.run.clone(), + &loaded.turns, + &loaded.records, + loaded.event_provenance, + )); + } metrics.record("turn_projection", phase); Ok(Json(page)) } @@ -2342,6 +2561,8 @@ struct QueryCatalog { #[derive(Debug, Serialize)] struct QueryDatasetSummary { + #[serde(skip_serializing_if = "Option::is_none")] + browse: Option, name: String, uri: String, ready_sources: usize, @@ -2675,7 +2896,7 @@ struct QueryTablesQuery { } async fn ui_query_catalog(state: &AppState) -> Result, ApiError> { - let browse = browse_coordinator(state).await; + let browse = state.browse.get(); let default_name = state .config .default_dataset @@ -2697,8 +2918,12 @@ async fn ui_query_catalog(state: &AppState) -> Result, ApiErr .unwrap_or_default(); let mut datasets = Vec::with_capacity(state.config.datasets.len()); for mount in &state.config.datasets { - let tree = browse.cached_dataset(mount).await; + let tree = match browse { + Some(browse) => browse.cached_dataset(mount).await, + None => None, + }; let (ready_sources, error_sources) = tree + .as_ref() .map(|tree| { tree.children.iter().fold((0, 0), |(ready, failed), child| { ( @@ -2709,6 +2934,7 @@ async fn ui_query_catalog(state: &AppState) -> Result, ApiErr }) .unwrap_or_default(); datasets.push(QueryDatasetSummary { + browse: None, name: mount.name.clone(), uri: mount.uri.clone(), ready_sources, @@ -2767,6 +2993,7 @@ async fn query_tables( .datasets() .iter() .map(|dataset| QueryDatasetSummary { + browse: None, name: dataset.mount.name.clone(), uri: dataset.mount.uri.clone(), ready_sources: dataset.ready_source_count(), diff --git a/crates/persisting-pchronicle-cli/src/server/problem.rs b/crates/persisting-pchronicle-cli/src/server/problem.rs index 2c3abf862..b93e0fccb 100644 --- a/crates/persisting-pchronicle-cli/src/server/problem.rs +++ b/crates/persisting-pchronicle-cli/src/server/problem.rs @@ -21,6 +21,16 @@ pub(crate) enum BoundaryCode { Internal, } +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub(crate) enum ExecutionStage { + Admission, + Catalog, + Manifest, + Worker, + Query, +} + impl BoundaryCode { pub(crate) const fn as_str(self) -> &'static str { match self { @@ -73,6 +83,8 @@ pub(super) struct ApiError { pub(super) code: BoundaryCode, message: String, request_id: String, + #[serde(skip_serializing_if = "Option::is_none")] + stage: Option, #[serde(skip)] root_cause: Option, } @@ -84,6 +96,7 @@ impl ApiError { code, message: message.into(), request_id: String::new(), + stage: None, root_cause: None, } } @@ -93,6 +106,11 @@ impl ApiError { self } + pub(super) fn with_stage(mut self, stage: ExecutionStage) -> Self { + self.stage = Some(stage); + self + } + pub(super) fn with_4xx_root_cause(self, error: &anyhow::Error) -> Self { if error.source().is_some() { self.with_4xx_root_cause_text(truncate_utf8( @@ -119,6 +137,15 @@ impl ApiError { let request_id = request_id.as_ref(); let deeper = error.source().is_some(); let root_cause = truncate_utf8(&error.root_cause().to_string(), ROOT_CAUSE_LIMIT); + // An elapsed internal budget is a timeout, not a defect. Reporting it + // as `internal` hid both the retry advice and the fact that the read + // was merely too slow for the request's time budget. + if error + .chain() + .any(|cause| cause.is::()) + { + return Self::deadline_exceeded(request_id, handler, error); + } let api = if let Some(boundary) = error.downcast_ref::() { Self::from_boundary(request_id, boundary.code, boundary.message.clone()) } else { @@ -203,6 +230,43 @@ impl ApiError { ) } + pub(super) fn runs_timeout() -> Self { + Self::public( + StatusCode::GATEWAY_TIMEOUT, + BoundaryCode::Unavailable, + "Runs request timed out; narrow the dataset or file scope and retry", + ) + } + + /// A budget inside the handler elapsed. Operators still need the chain, so + /// log it like an internal failure but answer with the retryable status. + fn deadline_exceeded(request_id: &str, handler: &'static str, error: anyhow::Error) -> Self { + tracing::warn!( + target: LOG_TARGET, + request_id = %request_id, + code = "unavailable", + handler = %handler, + root_cause = %truncate_utf8(&error.root_cause().to_string(), ROOT_CAUSE_LIMIT), + chain = %truncate_utf8(&format!("{error:#}"), CHAIN_LIMIT), + "warehouse request exceeded its deadline" + ); + Self::public( + StatusCode::GATEWAY_TIMEOUT, + BoundaryCode::Unavailable, + "Request exceeded its time budget; narrow the dataset or file scope and retry", + ) + .with_request_id(request_id) + .with_stage(ExecutionStage::Query) + } + + pub(super) fn trajectory_timeout() -> Self { + Self::public( + StatusCode::GATEWAY_TIMEOUT, + BoundaryCode::Unavailable, + "Trajectory request timed out; retry this run", + ) + } + pub(super) fn input(issue: InputIssue) -> Self { let message = issue.message().to_owned(); match issue.kind() { diff --git a/crates/persisting-pchronicle-cli/src/server/query_admission.rs b/crates/persisting-pchronicle-cli/src/server/query_admission.rs index 3c2002a1a..13ea2ee9f 100644 --- a/crates/persisting-pchronicle-cli/src/server/query_admission.rs +++ b/crates/persisting-pchronicle-cli/src/server/query_admission.rs @@ -3,7 +3,7 @@ use std::collections::HashMap; use std::future::Future; -use std::sync::{Arc, Mutex, Weak}; +use std::sync::{Arc, Mutex, MutexGuard, Weak}; use std::time::{Duration, Instant}; use anyhow::Result; @@ -20,6 +20,12 @@ const MAX_CACHE_BYTES: usize = 64 * 1024 * 1024; // One background summary refresh per process, with no waiting task queue. pub(super) static REFRESH_SLOT: Semaphore = Semaphore::const_new(1); +fn lock_recover(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) +} + struct CachedSummaries { summaries: Summaries, built_at: Instant, @@ -83,16 +89,16 @@ impl Default for ScopedQueries { impl ScopedQueries { pub(super) fn invalidate(&self) { - let mut cache = self.cache.lock().unwrap(); + let mut cache = lock_recover(&self.cache); cache.generation += 1; cache.entries.clear(); // Requests started after explicit refresh must not join an older build. - self.flights.lock().unwrap().clear(); + lock_recover(&self.flights).clear(); } fn publish(&self, scope: QueryScope, generation: u64, summaries: Summaries) { let bytes = summary_bytes(&summaries); - let mut cache = self.cache.lock().unwrap(); + let mut cache = lock_recover(&self.cache); if cache.generation != generation { return; } @@ -140,7 +146,7 @@ impl ScopedQueries { Fut: Future> + Send + 'static, { let (generation, cached) = { - let mut cache = self.cache.lock().unwrap(); + let mut cache = lock_recover(&self.cache); let generation = cache.generation; let cached = cache.entries.get_mut(&scope).map(|entry| { entry.last_used = Instant::now(); @@ -163,7 +169,7 @@ impl ScopedQueries { match queries.run(scope.clone(), || execute(true)).await { Ok(summaries) => queries.publish(scope, generation, summaries), Err(error) => { - let mut cache = queries.cache.lock().unwrap(); + let mut cache = lock_recover(&queries.cache); if cache.generation == generation && let Some(entry) = cache.entries.get_mut(&scope) { @@ -195,7 +201,7 @@ impl ScopedQueries { Fut: Future>, { let flight = { - let mut flights = self.flights.lock().unwrap(); + let mut flights = lock_recover(&self.flights); flights.retain(|_, entry| entry.strong_count() > 0); if let Some(flight) = flights.get(&scope).and_then(Weak::upgrade) { flight @@ -206,12 +212,14 @@ impl ScopedQueries { flight } }; + super::request_progress::phase("query_queue"); // OnceCell transfers initialization to a waiter if the initializing // request is cancelled. No detached tasks or permanently owned pins. flight .result .get_or_init(|| async { let _slot = self.slots.acquire().await.expect("admission never closes"); + super::request_progress::phase("query"); execute().await.map_err(SharedAccelerationFailure::new) }) .await @@ -234,7 +242,7 @@ mod tests { } fn expire(queries: &ScopedQueries, key: &QueryScope) { - let mut cache = queries.cache.lock().unwrap(); + let mut cache = lock_recover(&queries.cache); let entry = cache.entries.get_mut(key).unwrap(); entry.built_at = Instant::now() - REFRESH_INTERVAL; entry.refresh_after = Instant::now(); diff --git a/crates/persisting-pchronicle-cli/src/server/request_log.rs b/crates/persisting-pchronicle-cli/src/server/request_log.rs index d077696a9..f73b9ed3a 100644 --- a/crates/persisting-pchronicle-cli/src/server/request_log.rs +++ b/crates/persisting-pchronicle-cli/src/server/request_log.rs @@ -113,10 +113,17 @@ where } } -pub(crate) async fn warehouse_request_layer( +pub(super) async fn warehouse_request_layer( + axum::extract::State(state): axum::extract::State, mut request: Request, next: Next, ) -> Response { + // Progress polling must stay cheap and must not generate more request records. + if request.uri().path().starts_with("/api/requests/") + || request.uri().path().starts_with("/api/v1/requests/") + { + return next.run(request).await; + } let incoming = request .headers() .get("x-request-id") @@ -143,8 +150,36 @@ pub(crate) async fn warehouse_request_layer( query = %truncate_utf8(&query, QUERY_LOG_LIMIT), "warehouse request start" ); - let response = next.run(request).await; + // Worker requests carry their trace over private IPC; public observers + // never acquire a worker slot just to inspect a stalled request. + let progress = request + .extensions() + .get::() + .cloned() + .unwrap_or_else(|| { + super::request_progress::Progress::new( + request_id.clone(), + method.clone(), + path.clone(), + false, + ) + }); + if !path.contains("/requests/") && path.starts_with("/api/") { + state + .request_progress + .insert(request.headers(), progress.clone()); + } + let _cancel = super::request_progress::CancelOnDrop(progress.clone()); + progress.phase(if state.catalog_query_worker { + "execution" + } else { + "authentication" + }); + let response = super::request_progress::scope(progress.clone(), next.run(request)).await; let status = response.status(); + if status.is_success() { + progress.phase("response"); + } let root_cause = response .extensions() .get::() @@ -165,6 +200,8 @@ pub(crate) async fn warehouse_request_layer( query = %truncate_utf8(&query, QUERY_LOG_LIMIT), "warehouse request" ); + let diagnostic_error = error_fields.as_ref().map(|(_, message)| message.clone()); + progress.finish(Some(status.as_u16()), diagnostic_error); if (400..500).contains(&status.as_u16()) { let (code, message) = error_fields.unwrap_or_default(); let fts_errors = fts.joined(); @@ -267,7 +304,19 @@ pub(crate) fn tracing_filter(level: crate::LogLevel) -> String { } } +/// Level the process initialized tracing with, so spawned catalog workers can +/// be started at the same verbosity instead of defaulting to their own. +static INITIALIZED_LEVEL: std::sync::OnceLock = std::sync::OnceLock::new(); + +pub(crate) fn initialized_log_level() -> crate::LogLevel { + INITIALIZED_LEVEL + .get() + .copied() + .unwrap_or(crate::LogLevel::Info) +} + pub(crate) fn init_warehouse_tracing(level: crate::LogLevel) { + let _ = INITIALIZED_LEVEL.set(level); // Synchronous stderr is enough: lines are short. Do not wrap this in an // async logger while `main` holds `stdout.lock()` — on macOS `Stderr` // writes take that same lock and deadlock Tokio workers. diff --git a/crates/persisting-pchronicle-cli/src/server/request_progress.rs b/crates/persisting-pchronicle-cli/src/server/request_progress.rs new file mode 100644 index 000000000..70892c1ae --- /dev/null +++ b/crates/persisting-pchronicle-cli/src/server/request_progress.rs @@ -0,0 +1,432 @@ +//! Bounded, opt-in request diagnostics. The observer token is a capability, +//! separate from catalog credentials, so even authentication failures are visible. +use super::problem::ApiError; +use axum::{ + Json, + extract::{Path, State}, + http::HeaderMap, +}; +use serde::{Deserialize, Serialize}; +use std::{ + collections::HashMap, + future::Future, + sync::{Arc, Mutex}, + time::{Duration, Instant}, +}; + +pub const OBSERVER_HEADER: &str = "x-pchronicle-observer"; +const LIMIT: usize = 512; +const TTL: Duration = Duration::from_secs(600); + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct Phase { + pub name: String, + pub state: String, + pub elapsed_ms: u64, +} +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct Snapshot { + pub request_id: String, + pub method: String, + pub path: String, + pub state: String, + pub elapsed_ms: u64, + pub status: Option, + pub error: Option, + pub note: Option, + pub phases: Vec, + pub worker: Option>, +} +struct Running { + snapshot: Snapshot, + started: Instant, + phase_started: Instant, + worker_updated: Instant, +} +#[derive(Clone)] +pub struct Progress(Arc>); +#[derive(Default)] +pub struct Registry(Mutex>); + +tokio::task_local! { static CURRENT: Progress; } +pub async fn scope(progress: Progress, work: F) -> F::Output { + CURRENT.scope(progress, work).await +} +pub fn current() -> Option { + CURRENT.try_with(Clone::clone).ok() +} +pub fn phase(name: &str) { + if let Some(p) = current() { + p.phase(name); + } +} +fn ms(d: Duration) -> u64 { + d.as_millis().min(u64::MAX as u128) as u64 +} + +impl Progress { + pub fn new(id: String, method: String, path: String, worker: bool) -> Self { + let mut names = if worker { + vec!["execution"] + } else { + vec![ + "authentication", + "worker_queue", + "worker_start", + "worker_execution", + ] + }; + if path.ends_with("/explorer/tree") { + names.extend(["browse_cache", "directory_wait", "manifest_summary"]); + } else { + names.extend(["source_metadata", "storage_read", "query"]); + } + names.push("response"); + let now = Instant::now(); + Self(Arc::new(Mutex::new(Running { + snapshot: Snapshot { + request_id: id, + method, + path, + state: "running".into(), + elapsed_ms: 0, + status: None, + error: None, + note: None, + phases: names + .into_iter() + .map(|name| Phase { + name: name.into(), + state: "pending".into(), + elapsed_ms: 0, + }) + .collect(), + worker: None, + }, + started: now, + phase_started: now, + worker_updated: now, + }))) + } + pub fn note(&self, note: &str) { + self.0 + .lock() + .unwrap_or_else(|e| e.into_inner()) + .snapshot + .note = Some(note.into()); + } + pub fn phase(&self, name: &str) { + let mut v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + if v.snapshot.state != "running" { + return; + } + if v.snapshot + .phases + .iter() + .any(|p| p.name == name && p.state == "running") + { + return; + } + let elapsed = ms(v.phase_started.elapsed()); + for p in &mut v.snapshot.phases { + if p.state == "running" { + p.state = "completed".into(); + p.elapsed_ms += elapsed; + } + } + if let Some(p) = v.snapshot.phases.iter_mut().find(|p| p.name == name) { + p.state = "running".into(); + } else { + v.snapshot.phases.push(Phase { + name: name.into(), + state: "running".into(), + elapsed_ms: 0, + }); + } + v.phase_started = Instant::now(); + } + pub fn snapshot(&self) -> Snapshot { + let v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + let mut s = v.snapshot.clone(); + if s.state == "running" { + s.elapsed_ms = ms(v.started.elapsed()); + for p in &mut s.phases { + if p.state == "running" { + p.elapsed_ms += ms(v.phase_started.elapsed()); + } + } + if let Some(w) = s.worker.as_mut() + && w.state == "running" + { + let elapsed = ms(v.worker_updated.elapsed()); + w.elapsed_ms += elapsed; + for p in &mut w.phases { + if p.state == "running" { + p.elapsed_ms += elapsed; + } + } + } + } + s + } + pub fn worker(&self, snapshot: Snapshot) { + let mut v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + v.snapshot.worker = Some(Box::new(snapshot)); + v.worker_updated = Instant::now(); + } + pub fn finish(&self, status: Option, error: Option) { + let mut v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + if v.snapshot.state != "running" { + return; + } + let state = match status { + None => "cancelled", + Some(code) if code >= 400 || error.is_some() => "failed", + Some(_) => "completed", + }; + let elapsed = ms(v.phase_started.elapsed()); + for p in &mut v.snapshot.phases { + if p.state == "running" { + p.state = state.into(); + p.elapsed_ms += elapsed; + } else if p.state == "pending" { + p.state = "skipped".into(); + } + } + if let Some(w) = v.snapshot.worker.as_mut() + && w.state == "running" + && state != "completed" + { + w.state = state.into(); + for p in &mut w.phases { + if p.state == "running" { + p.state = state.into(); + } + } + } + v.snapshot.state = state.into(); + v.snapshot.elapsed_ms = ms(v.started.elapsed()); + v.snapshot.status = status; + v.snapshot.error = error; + } +} +pub struct CancelOnDrop(pub Progress); +impl Drop for CancelOnDrop { + fn drop(&mut self) { + self.0 + .finish(None, Some("Request execution was cancelled".into())); + } +} + +fn observer(headers: &HeaderMap) -> Option { + let token = headers.get(OBSERVER_HEADER)?.to_str().ok()?; + if token.len() != 32 || !token.bytes().all(|b| b.is_ascii_hexdigit()) { + return None; + } + Some(blake3::hash(token.as_bytes()).to_hex().to_string()) +} +impl Registry { + pub fn insert(&self, headers: &HeaderMap, progress: Progress) { + let Some(owner) = observer(headers) else { + return; + }; + let mut entries = self.0.lock().unwrap_or_else(|e| e.into_inner()); + entries.retain(|_, p| { + p.0.lock() + .unwrap_or_else(|e| e.into_inner()) + .started + .elapsed() + < TTL + }); + if entries.len() >= LIMIT + && let Some(oldest) = entries + .iter() + .max_by_key(|(_, p)| { + p.0.lock() + .unwrap_or_else(|e| e.into_inner()) + .started + .elapsed() + }) + .map(|(k, _)| k.clone()) + { + entries.remove(&oldest); + } + let id = progress.snapshot().request_id; + entries.insert((owner, id), progress); + } + fn get(&self, headers: &HeaderMap, id: &str) -> Option { + let owner = observer(headers)?; + let entries = self.0.lock().unwrap_or_else(|e| e.into_inner()); + let p = entries.get(&(owner, id.to_owned()))?; + if p.0 + .lock() + .unwrap_or_else(|e| e.into_inner()) + .started + .elapsed() + >= TTL + { + return None; + } + Some(p.snapshot()) + } +} +pub(super) async fn get( + State(state): State, + Path(id): Path, + headers: HeaderMap, +) -> Result, ApiError> { + state + .request_progress + .get(&headers, &id) + .map(Json) + .ok_or_else(|| { + ApiError::not_found("Request diagnostics expired or are not available to this browser") + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn diagnostics_respond_while_request_is_blocked_and_preserve_failure() { + use axum::{Router, body::Body, http::Request, middleware, routing::get}; + use tower::ServiceExt; + let state = super::super::app_state(super::super::ChronicleServerConfig::front_only()); + let entered = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let app = Router::new() + .route( + "/api/slow", + get({ + let entered = entered.clone(); + let release = release.clone(); + move || { + let entered = entered.clone(); + let release = release.clone(); + async move { + phase("storage_read"); + entered.notify_one(); + release.notified().await; + Err::<(), _>(ApiError::invalid_request("source format is invalid")) + } + } + }), + ) + .route("/api/requests/{id}", get(super::get)) + .layer(middleware::from_fn_with_state( + state.clone(), + super::super::request_log::warehouse_request_layer, + )) + .with_state(state.clone()); + let token = "0123456789abcdef0123456789abcdef"; + let request = |uri: &str, owner: &str| { + Request::builder() + .uri(uri) + .header(OBSERVER_HEADER, owner) + .header("x-request-id", "live-test") + .body(Body::empty()) + .unwrap() + }; + let task = tokio::spawn( + app.clone() + .oneshot(request("/api/slow?secret=not-recorded", token)), + ); + entered.notified().await; + let response = tokio::time::timeout( + Duration::from_secs(1), + app.clone() + .oneshot(request("/api/requests/live-test", token)), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(response.status(), 200); + let bytes = axum::body::to_bytes(response.into_body(), 65536) + .await + .unwrap(); + assert!(!String::from_utf8_lossy(&bytes).contains("not-recorded")); + let snapshot: Snapshot = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(snapshot.state, "running"); + assert!( + snapshot + .phases + .iter() + .any(|p| p.name == "storage_read" && p.state == "running") + ); + let denied = app + .clone() + .oneshot(request( + "/api/requests/live-test", + "ffffffffffffffffffffffffffffffff", + )) + .await + .unwrap(); + assert_eq!(denied.status(), 404); + release.notify_one(); + assert_eq!(task.await.unwrap().unwrap().status(), 400); + let response = app + .oneshot(request("/api/requests/live-test", token)) + .await + .unwrap(); + let bytes = axum::body::to_bytes(response.into_body(), 65536) + .await + .unwrap(); + let snapshot: Snapshot = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(snapshot.state, "failed"); + assert_eq!(snapshot.error.as_deref(), Some("source format is invalid")); + assert!( + snapshot + .phases + .iter() + .any(|p| p.name == "storage_read" && p.state == "failed") + ); + assert_eq!( + state.request_progress.0.lock().unwrap().len(), + 1, + "polls are never registered" + ); + } + + #[test] + fn progress_is_live_isolated_bounded_and_finishes_on_cancel() { + let registry = Registry::default(); + let mut headers = HeaderMap::new(); + headers.insert( + OBSERVER_HEADER, + "0123456789abcdef0123456789abcdef".parse().unwrap(), + ); + let p = Progress::new( + "id".into(), + "GET".into(), + "/api/explorer/runs".into(), + false, + ); + registry.insert(&headers, p.clone()); + p.phase("authentication"); + p.phase("worker_queue"); + assert!(registry.get(&HeaderMap::new(), "id").is_none()); + let snapshot = registry.get(&headers, "id").unwrap(); + assert_eq!(snapshot.phases[0].state, "completed"); + assert_eq!(snapshot.phases[1].state, "running"); + let mut other = headers.clone(); + other.insert( + OBSERVER_HEADER, + "ffffffffffffffffffffffffffffffff".parse().unwrap(), + ); + assert!(registry.get(&other, "id").is_none()); + drop(CancelOnDrop(p)); + assert_eq!(registry.get(&headers, "id").unwrap().state, "cancelled"); + for i in 0..LIMIT + 5 { + registry.insert( + &headers, + Progress::new(i.to_string(), "GET".into(), "/api/runs".into(), false), + ); + } + assert_eq!(registry.0.lock().unwrap().len(), LIMIT); + let expired = Progress::new("expired".into(), "GET".into(), "/api/runs".into(), false); + registry.insert(&headers, expired.clone()); + expired.0.lock().unwrap().started = Instant::now() - TTL; + assert!(registry.get(&headers, "expired").is_none()); + } +} diff --git a/crates/persisting-pchronicle-cli/src/server/tests.rs b/crates/persisting-pchronicle-cli/src/server/tests.rs index 71c44a081..a1388abc3 100644 --- a/crates/persisting-pchronicle-cli/src/server/tests.rs +++ b/crates/persisting-pchronicle-cli/src/server/tests.rs @@ -51,26 +51,24 @@ fn explorer_run_identity_sql_does_not_project_step_payloads() { } #[test] -fn explorer_run_preview_sql_is_row_bounded() { - let sql = explorer_run_preview_sql( - "dataset", - "steps", - "_file_ AS source_path, document_id, message_value", - "step_id = 1", - 512, - ); - let lowered = sql.to_ascii_lowercase(); - assert!(lowered.contains("limit 512"), "{sql}"); - assert!(lowered.contains("message_value"), "{sql}"); +fn search_preview_is_bounded_around_the_hit() { + let raw = format!("{} ipython {}", "前缀 ".repeat(400), "suffix ".repeat(400)); + let preview = search_preview_text(&raw, "IPYTHON"); + assert!(preview.contains("ipython")); + assert!(preview.chars().count() <= 322); + assert!(preview.starts_with('…') && preview.ends_with('…')); } #[test] -fn search_preview_returns_the_complete_normalized_field() { - let raw = format!("{} ipython {}", "prefix ".repeat(80), "suffix ".repeat(80)); - let preview = search_preview_text(&raw); - assert!(preview.contains("ipython")); - assert!(preview.starts_with("prefix prefix")); - assert!(preview.ends_with("suffix suffix ")); +fn default_search_projects_only_message_body() { + let expression = crate::combine_match_expressions(&["rust".into()]) + .unwrap() + .unwrap(); + assert_eq!(explorer_preview_columns(&expression), ["message_value"]); + let expression = crate::combine_match_expressions(&["#observation(rust)".into()]) + .unwrap() + .unwrap(); + assert_eq!(explorer_preview_columns(&expression), ["observation"]); } #[test] @@ -334,6 +332,59 @@ async fn internal_error_logs_root_cause_and_redacts_json() { ); } +#[tokio::test] +async fn elapsed_internal_budget_reports_a_retryable_timeout() { + let events = std::sync::Arc::new(std::sync::Mutex::new(Vec::::new())); + let _guard = tracing::subscriber::set_default(CapturingSubscriber::new(events.clone())); + let elapsed = tokio::time::timeout(Duration::from_millis(1), std::future::pending::<()>()) + .await + .expect_err("pending future must time out"); + let error = anyhow::Error::new(elapsed) + .context("runs scan timed out") + .context("cached acceleration build failure"); + let response = super::fail( + &RequestId("rid-deadline".into()), + "load_run_summaries", + error, + ) + .into_response(); + + assert_eq!(response.status(), StatusCode::GATEWAY_TIMEOUT); + let body = response_json(response).await; + assert_eq!(body["code"], "unavailable"); + assert_eq!(body["stage"], "query"); + assert_eq!(body["request_id"], "rid-deadline"); + assert!( + body["message"].as_str().unwrap().contains("time budget"), + "{body}" + ); + + let logged = events.lock().unwrap().clone(); + assert!( + logged + .iter() + .all(|event| event.level != tracing::Level::ERROR), + "a timeout must not be logged as an internal defect: {logged:?}" + ); + let warning = logged + .iter() + .find(|event| event.message.contains("exceeded its deadline")) + .unwrap_or_else(|| panic!("{logged:?}")); + assert!( + warning + .fields + .get("chain") + .unwrap() + .contains("runs scan timed out"), + "{:?}", + warning.fields + ); + assert_eq!( + warning.fields.get("handler").map(String::as_str), + Some("load_run_summaries") + ); +} + #[tokio::test] async fn middleware_echoes_request_id_on_json_errors() { use tower::ServiceExt; @@ -343,7 +394,8 @@ async fn middleware_echoes_request_id_on_json_errors() { } let app = axum::Router::new() .route("/api/boom", axum::routing::get(boom)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let response = app @@ -387,7 +439,8 @@ async fn four_xx_warn_includes_root_cause_when_chain_is_deeper() { } let app = axum::Router::new() .route("/api/boom", axum::routing::get(boom)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let response = app @@ -451,7 +504,8 @@ async fn middleware_rejects_illegal_incoming_id() { } let app = axum::Router::new() .route("/api/boom", axum::routing::get(boom)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let response = app @@ -489,7 +543,8 @@ async fn middleware_info_logs_static_assets() { } let app = axum::Router::new() .route("/assets/app.css", axum::routing::get(missing)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let _ = app @@ -2458,6 +2513,68 @@ async fn physical_api_lists_empty_sources_for_json_catalog_and_rejects_non_lance assert_eq!(body["code"], "not_found"); } +#[tokio::test] +async fn search_pages_have_body_previews_for_every_visible_run() { + use persisting_pchronicle::storage::StorylineLanceStore; + let root = tempfile::tempdir().unwrap(); + let store = StorylineLanceStore::open(root.path().join("story")) + .await + .unwrap(); + // More than the old 512-row global preview budget in the first run. + for (session, count) in [("a", 513), ("b", 1), ("c", 1), ("metadata", 1)] { + let mut document = storyline_document(session, session); + let template = document.turns[0].clone(); + document.turns = (0..count) + .map(|id| { + let mut turn = template.clone(); + turn.id = id; + turn.message = json!(if session == "metadata" { + "unrelated text".to_owned() + } else { + format!("{} rust {session}", "padding ".repeat(60)) + }); + turn.observation = Some(json!("rust metadata")); + turn.prompt = + persisting_pchronicle::model::StorylinePrompt::from_pair("rust prompt", ""); + turn + }) + .collect(); + store.replace_storyline(&document).await.unwrap(); + } + let app = router(root.path().to_string_lossy().to_string()); + let dataset = encode_query(DEFAULT_DATASET_NAME); + for (offset, expected) in [(0, vec!["a", "b"]), (2, vec!["c"])] { + let (status, body) = get_json( + &app, + &format!("/api/explorer/runs?dataset={dataset}&q=rust&offset={offset}&limit=2"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], 3, "{body}"); + assert_eq!(body["snapshot"]["has_more"], offset == 0); + let records = body["records"].as_array().unwrap(); + assert_eq!(records.len(), expected.len(), "{body}"); + assert_eq!(body["path_index"].as_array().unwrap().len(), expected.len()); + for (record, session) in records.iter().zip(expected) { + assert_eq!(record["session_id"], session); + let preview = record["search_preview"].as_str().unwrap(); + assert!(preview.contains(&format!("rust {session}")), "{record}"); + assert!(preview.chars().count() <= 322); + } + } + let (status, body) = get_json( + &app, + &format!( + "/api/explorer/runs?dataset={dataset}&q={}&limit=2", + encode_query("#observation(rust)") + ), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], 4, "{body}"); + assert_eq!(body["records"][0]["search_preview"], "rust metadata"); +} + #[tokio::test] async fn physical_api_inspects_storyline_lance_layout_file_and_page() { use persisting_pchronicle::storage::StorylineLanceStore; @@ -2640,6 +2757,75 @@ async fn physical_api_inspects_storyline_lance_layout_file_and_page() { ); } +#[tokio::test] +async fn exact_runs_request_does_not_wait_for_global_catalog() -> anyhow::Result<()> { + let root = tempfile::tempdir()?; + let store = + persisting_pchronicle::storage::StorylineLanceStore::open(root.path().join("nested/story")) + .await?; + store + .replace_storyline(&storyline_document("session-a", "run-a")) + .await?; + let state = app_state(ChronicleServerConfig::mounted(vec![DatasetMount::new( + "prod2", + root.path().to_string_lossy(), + )?])?); + // A slow unrelated catalog refresh must not block an exact source request. + let _refresh = state.catalog_refresh.lock().await; + let app = finish_routes(state.clone()); + let (status, page) = tokio::time::timeout( + Duration::from_secs(5), + get_json( + &app, + "/api/explorer/runs?dataset=prod2&file=nested/story&limit=50", + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{page}"); + assert_eq!(page["snapshot"]["total"], 1, "{page}"); + assert!(state.catalog.read().await.is_none()); + Ok(()) +} + +#[tokio::test] +async fn exact_trajectory_endpoints_do_not_wait_for_global_catalog() -> anyhow::Result<()> { + let root = tempfile::tempdir()?; + let store = + persisting_pchronicle::storage::StorylineLanceStore::open(root.path().join("nested/story")) + .await?; + store + .replace_storyline(&storyline_document("session-a", "run-a")) + .await?; + let state = app_state(ChronicleServerConfig::mounted(vec![DatasetMount::new( + "prod2", + root.path().to_string_lossy(), + )?])?); + let _refresh = state.catalog_refresh.lock().await; + let app = finish_routes(state.clone()); + let coords = "dataset=prod2&file=nested/story&agent_id=storyline&session_id=session-a"; + for endpoint in [ + "explorer/run", + "explorer/turns", + "explorer/turn", + "trajectory-view", + "events", + "storyline", + ] { + let (status, body) = tokio::time::timeout( + Duration::from_secs(5), + get_json(&app, &format!("/api/{endpoint}?{coords}&turn_id=1")), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{endpoint}: {body}"); + if endpoint == "explorer/run" { + assert_eq!(body["event_provenance"], "synthetic_from_storyline"); + assert!(body["turn_count"].as_u64().unwrap() > 0); + } + } + assert!(state.catalog.read().await.is_none()); + Ok(()) +} + #[tokio::test] async fn browse_tree_does_not_build_query_runtime() -> anyhow::Result<()> { use http_body_util::BodyExt; @@ -2676,3 +2862,256 @@ async fn browse_tree_does_not_build_query_runtime() -> anyhow::Result<()> { ); Ok(()) } + +#[tokio::test] +async fn runs_metadata_does_not_initialize_browse_or_resolve_sources() { + let root = tempfile::tempdir().unwrap(); + let config = ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.path().join("missing").to_string_lossy().to_string()).unwrap(), + ]) + .unwrap(); + let state = app_state(config); + let Json(catalog) = ui_query_catalog(&state).await.unwrap(); + assert_eq!(catalog.datasets.len(), 1); + assert_eq!(catalog.datasets[0].name, "dataset"); + assert!(state.browse.get().is_none()); + assert!(state.catalog.read().await.is_none()); +} + +#[tokio::test] +async fn runs_unknown_dataset_is_a_structured_error_without_panicking() { + let root = tempfile::tempdir().unwrap(); + let app = router(root.path().to_string_lossy().to_string()); + let (status, response) = get_json(&app, "/api/explorer/runs?dataset=missing").await; + assert_eq!(status, StatusCode::NOT_FOUND, "{response}"); + assert_eq!(response["code"], "not_found"); + let (status, _) = get_json(&app, "/api/query/tables?ui=true").await; + assert_eq!(status, StatusCode::OK); +} + +#[tokio::test] +async fn runs_deadline_includes_catalog_lock_and_releases_cancelled_work() { + let root = tempfile::tempdir().unwrap(); + write_gateway_fixture(root.path(), "run.json", "session", "job"); + let config = ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.path().to_string_lossy().to_string()).unwrap(), + ]) + .unwrap(); + let state = app_state(config); + let held = state.catalog_refresh.lock().await; + let request_id = RequestId("runs-timeout-test".into()); + let error = with_deadline( + Duration::from_millis(20), + &request_id, + ApiError::runs_timeout, + load_run_summaries(&state, None, None, &request_id, None), + ) + .await + .unwrap_err(); + assert_eq!(error.status, StatusCode::GATEWAY_TIMEOUT); + let body = serde_json::to_value(error).unwrap(); + assert_eq!(body["code"], "unavailable"); + assert_eq!(body["request_id"], "runs-timeout-test"); + assert_eq!(body["stage"], "query"); + drop(held); + let summaries = with_deadline( + Duration::from_secs(5), + &request_id, + ApiError::runs_timeout, + load_run_summaries(&state, None, None, &request_id, None), + ) + .await + .unwrap(); + assert_eq!(summaries.len(), 1); +} + +#[tokio::test] +async fn full_turn_lists_and_scoped_search_keep_source_boundaries() { + use persisting_pchronicle::storage::StorylineLanceStore; + let root = tempfile::tempdir().unwrap(); + for (file, session) in [ + ("group/one", "one"), + ("group/two", "two"), + ("group-other", "other"), + ] { + let store = StorylineLanceStore::open(root.path().join(file)) + .await + .unwrap(); + let mut story = storyline_document(session, session); + let template = story.turns[0].clone(); + story.turns = (0..5) + .map(|index| { + let mut turn = template.clone(); + turn.id = 10 + index * 3; + turn.source = if index % 2 == 0 { "user" } else { "agent" }.into(); + turn.message = json!(format!("hello {session} step {index}")); + turn + }) + .collect(); + store.replace_storyline(&story).await.unwrap(); + } + let state = app_state( + ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.path().to_string_lossy()).unwrap(), + ]) + .unwrap(), + ); + let request = RequestId("search-scope-test".into()); + let runtime = current_catalog(&state, &request).await.unwrap(); + let app = finish_routes(state); + for (file, expected) in [("", 3), ("group", 2), ("group/one", 1)] { + let (status, body) = get_json( + &app, + &format!("/api/explorer/runs?dataset={DEFAULT_DATASET_NAME}&file={file}&q=hello"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], expected, "{body}"); + } + let coords = format!( + "dataset={DEFAULT_DATASET_NAME}&file=group/one&agent_id=agent&session_id=one&run_id=one" + ); + for (suffix, total, id) in [ + ("offset=1&limit=2", 5, 10), + ("source=agent&offset=1&limit=1", 2, 13), + ("q=hello&offset=2&limit=1", 5, 10), + ] { + let (status, body) = + get_json(&app, &format!("/api/explorer/turns?{coords}&{suffix}")).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], total, "{body}"); + assert_eq!(body["records"][0]["id"], id, "{body}"); + } + let (status, body) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&offset=99&limit=2"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], 5); + assert_eq!(body["snapshot"]["has_more"], false); + assert_eq!(body["records"].as_array().unwrap().len(), 5); + let (status, body) = get_json(&app, &format!("/api/explorer/turn?{coords}&turn_id=19")).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["turn"]["id"], 19, "{body}"); + // An unreadable sibling must never be opened by scoped FTS. + let sibling = runtime + .snapshot + .storyline_table_paths(DEFAULT_DATASET_NAME, "group-other") + .unwrap() + .unwrap(); + std::fs::remove_dir_all(&sibling.steps).unwrap(); + let expression = crate::combine_match_expressions(&["hello".into()]) + .unwrap() + .unwrap(); + let (_, available, errors) = crate::find_expression_predicate_for_dataset( + &runtime.snapshot, + &expression, + Some("group"), + Some(DEFAULT_DATASET_NAME), + ) + .await + .unwrap(); + assert!(available); + assert!(errors.is_empty(), "{errors:?}"); +} + +#[tokio::test] +async fn overlapping_trajectory_loads_share_immutable_result() { + let root = json_dataset_root(); + let state = app_state( + ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.to_string_lossy()).unwrap(), + ]) + .unwrap(), + ); + let request = RequestId("coalesced-trajectory".into()); + let metrics = RequestMetrics::default(); + let summaries = load_run_summaries(&state, None, None, &request, None) + .await + .unwrap(); + let run = &summaries[0]; + let query = SessionQuery { + dataset: Some(run.dataset.clone()), + file: Some(run.file.clone()), + run_id: run.run_id.clone(), + agent_id: run.agent_id.clone(), + session_id: run.session_id.clone(), + root_session_id: run.root_session_id.clone(), + offset: None, + limit: None, + }; + let (left, right) = tokio::join!( + load_trajectory(&state, &query, &request, &metrics), + load_trajectory(&state, &query, &request, &metrics) + ); + assert!(Arc::ptr_eq(&left.unwrap(), &right.unwrap())); +} + +#[tokio::test] +async fn explorer_loads_all_turns_with_offloaded_unknown_fields() { + use persisting_pchronicle::storage::StorylineLanceStore; + let root = tempfile::tempdir().unwrap(); + let store = StorylineLanceStore::open(root.path().join("story")) + .await + .unwrap(); + let mut story = storyline_document("all-turns", "all-turns"); + for index in 0..16 { + story + .unknown_fields + .insert( + "codex", + "source", + format!("/events/{index}"), + json!({"text": "x".repeat(8192)}), + ) + .unwrap(); + } + story.refresh_unknown_key_counts().unwrap(); + let template = story.turns[0].clone(); + story.turns = (0..501) + .map(|index| { + let mut turn = template.clone(); + turn.id = index; + turn + }) + .collect(); + store.replace_storyline(&story).await.unwrap(); + let app = router(root.path().to_string_lossy().into_owned()); + let coords = "dataset=dataset&file=story&agent_id=agent&session_id=all-turns&run_id=all-turns"; + // Even legacy pagination parameters must not truncate the result. + let (status, body) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&offset=100&limit=1"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + let records = body["records"].as_array().unwrap(); + assert_eq!(records.len(), 501); + assert_eq!(records[0]["id"], 0); + assert_eq!(records[500]["id"], 500); + assert_eq!(body["snapshot"]["has_more"], false); + assert!(body.get("analysis").is_none()); + let (status, combined) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&include_analysis=true"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{combined}"); + assert_eq!(combined["records"], body["records"]); + let (status, body) = get_json(&app, &format!("/api/explorer/run?{coords}")).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["turn_count"], 501); + assert_eq!(combined["analysis"], body); + let (status, filtered) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&include_analysis=true&source=nonexistent"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{filtered}"); + assert!(filtered["records"].as_array().unwrap().is_empty()); + assert_eq!( + filtered["analysis"], body, + "statistics must cover the entire run" + ); +} diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index 1f50ed030..10dd0ca43 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -3,21 +3,26 @@ use std::collections::{HashMap, HashSet, VecDeque}; use std::path::PathBuf; -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Mutex, MutexGuard}; use std::time::{Duration, Instant}; use anyhow::{Context, Result}; -use persisting_pchronicle::storage::{DatasetLocation, DatasetMount, ManifestCache}; +use persisting_pchronicle::storage::{ + CatalogConsistency, CatalogState, DatasetLocation, DatasetMount, ManifestCache, PathListKind, + wait_for_foreground_object_store_idle, with_background_object_store_io, +}; use serde::{Deserialize, Serialize}; use tokio::sync::{RwLock, mpsc, oneshot}; use super::explorer::{CatalogTree, catalog_tree_from_mount_specs, catalog_tree_from_path_list}; const REFRESH_INTERVAL: Duration = Duration::from_secs(30); +const FOREGROUND_REFRESH_TIMEOUT: Duration = Duration::from_millis(250); const QUEUE_CAPACITY: usize = 128; -// ponytail: one process-wide browse scan at a time; per-backend budgets if -// multiple independent stores need more throughput. This does not gate SQL. -static BROWSE_IO: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); +// Keep browse work bounded while allowing a foreground request to run beside +// one background walk. +static BROWSE_IO: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(2); +static BACKGROUND_IO: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); #[derive(Clone, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] struct TreeKey { @@ -44,7 +49,7 @@ impl TreeKey { } fn mount_fingerprint(mount: &DatasetMount) -> String { - let location = DatasetLocation::parse(&mount.uri).ok(); + let location = DatasetLocation::parse_with_backend(&mount.uri, mount.backend().cloned()).ok(); let identity = location .as_ref() .and_then(|l| l.local_path()) @@ -57,16 +62,28 @@ fn mount_fingerprint(mount: &DatasetMount) -> String { .unwrap_or_else(|| mount.uri.trim_end_matches('/').to_owned()); // S3-compatible endpoints can expose different data under the same URI. let endpoint = if mount.uri.starts_with("s3://") { - std::env::var("AWS_ENDPOINT_URL_S3") - .or_else(|_| std::env::var("AWS_ENDPOINT_URL")) - .or_else(|_| std::env::var("AWS_ENDPOINT")) - .unwrap_or_default() + match mount.backend() { + Some(config) => config.endpoint.clone().unwrap_or_default(), + None => std::env::var("AWS_ENDPOINT_URL_S3") + .or_else(|_| std::env::var("AWS_ENDPOINT")) + .or_else(|_| std::env::var("AWS_ENDPOINT_URL")) + .unwrap_or_default(), + } } else { String::new() }; - blake3::hash(&serde_json::to_vec(&(identity, endpoint)).unwrap()) - .to_hex() - .to_string() + blake3::hash( + &serde_json::to_vec(&( + identity, + endpoint, + mount + .backend() + .map(|b| (&b.region, &b.access_key, &b.secret_key)), + )) + .unwrap(), + ) + .to_hex() + .to_string() } #[derive(Clone, Debug, Serialize, Deserialize)] @@ -81,13 +98,15 @@ struct IndexEntry { pub(crate) struct BrowseSnapshot { // Preserve existing Tree wire fields for old Web clients. #[serde(flatten)] - tree: CatalogTree, - browse: BrowseStatus, + pub(super) tree: CatalogTree, + pub(super) browse: BrowseStatus, } #[derive(Clone, Debug, Serialize)] -struct BrowseStatus { - consistency: &'static str, +pub(super) struct BrowseStatus { + partial: bool, + consistency: CatalogConsistency, + state: CatalogState, generation: String, observed_at: i64, stale: bool, @@ -105,6 +124,12 @@ struct RefreshState { type Reply = oneshot::Sender>; type Pending = Arc>>>; +fn lock_recover(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) +} + /// One bounded worker per serve instance, shared by timer and HTTP requests. /// The task owns the index, not the coordinator; dropping the last AppState /// aborts it, including an in-flight list. No permanent process singleton. @@ -172,16 +197,6 @@ impl BrowseCoordinator { } } - pub(crate) async fn cached_dataset(&self, mount: &DatasetMount) -> Option { - let key = TreeKey::new(mount, "").ok()?; - self.index - .values - .read() - .await - .get(&key) - .map(|entry| entry.tree.clone()) - } - pub(crate) async fn cached_source_paths(&self, mount: &DatasetMount) -> Vec { let values = self.index.values.read().await; let fingerprint = mount_fingerprint(mount); @@ -197,26 +212,36 @@ impl BrowseCoordinator { paths } + pub(crate) async fn cached_dataset(&self, mount: &DatasetMount) -> Option { + let key = TreeKey::new(mount, "").ok()?; + self.index + .values + .read() + .await + .get(&key) + .map(|entry| entry.tree.clone()) + } + pub(crate) async fn roots(&self, mounts: &[DatasetMount]) -> BrowseSnapshot { let mut tree = catalog_tree_from_mount_specs(mounts); let values = self.index.values.read().await; let mut observed_at = now(); let mut complete = true; + let mut partial = false; let mut refreshing = false; let mut error = None; for mount in mounts { let key = TreeKey::new(mount, "").expect("root prefix"); - let _ = self.enqueue(&key, None); - refreshing |= self.pending.lock().unwrap().contains_key(&key); - if let Some(failure) = self - .refresh - .lock() - .unwrap() + // Mount names are local configuration; the root page must not + // promote every mount into foreground remote work. + refreshing |= lock_recover(&self.pending).contains_key(&key); + if let Some(failure) = lock_recover(&self.refresh) .get(&key) .and_then(|s| s.error.clone()) { error = Some(failure); } + partial |= projection_is_partial(&values, &key); if let Some(entry) = values.get(&key) { observed_at = observed_at.min(entry.observed_at); if let Some(child) = tree.children.iter_mut().find(|c| c.name == mount.name) { @@ -230,6 +255,11 @@ impl BrowseCoordinator { .manifests .summary(&format!("{}\0{}", mount.name, mount_fingerprint(mount))) .await; + partial |= summary.partial + || !self + .manifests + .is_complete_under(&manifest_key(&key, Some("")), "") + .await; if let Some(child) = tree.children.iter_mut().find(|c| c.name == mount.name) { child.run_count = summary.trajectories as usize; child.dataset_count = Some(summary.datasets as usize); @@ -253,12 +283,24 @@ impl BrowseCoordinator { .fold(0usize, |sum, c| sum.saturating_add(c.failed_count)); BrowseSnapshot { browse: BrowseStatus { - consistency: "best_effort", + partial, + consistency: CatalogConsistency::BestEffort, + state: if refreshing { + CatalogState::Refreshing + } else if partial { + CatalogState::Partial + } else if !complete && error.is_some() { + CatalogState::Unavailable + } else if !complete || error.is_some() || now() - observed_at >= 30 { + CatalogState::Stale + } else { + CatalogState::Ready + }, generation: blake3::hash(&serde_json::to_vec(&tree).unwrap()) .to_hex() .to_string(), observed_at: if complete { observed_at } else { 0 }, - stale: !complete || error.is_some() || now() - observed_at >= 30, + stale: complete && (error.is_some() || now() - observed_at >= 30), refreshing, last_error: error, }, @@ -267,6 +309,7 @@ impl BrowseCoordinator { } pub(crate) async fn tree(&self, mount: &DatasetMount, prefix: &str) -> Result { + super::request_progress::phase("browse_cache"); let key = TreeKey::new(mount, prefix)?; tracing::info!( target: "pchronicle.serve", @@ -275,41 +318,77 @@ impl BrowseCoordinator { "browse tree request" ); let existing = self.index.values.read().await.get(&key).cloned(); + let existing = match existing { + Some(entry) => Some(entry), + None => self + .manifests + .get(&manifest_key(&key, None)) + .await + .map(|listing| { + let tree = + catalog_tree_from_path_list(&key.dataset, &key.prefix, &listing.entries); + IndexEntry { + generation: blake3::hash(&serde_json::to_vec(&tree).unwrap()) + .to_hex() + .to_string(), + tree, + observed_at: listing.observed_at, + } + }), + }; if let Some(entry) = existing { - let _ = self.enqueue(&key, None); + if now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64 { + let _ = self.enqueue(&key, None); + } return Ok(self.snapshot_with_summary(&key, entry).await); } + // A cold page is user-visible work: attach to the same single-flight + // refresh as the background walker and return once this prefix exists. + super::request_progress::phase("directory_wait"); let (reply, wait) = oneshot::channel(); self.enqueue(&key, Some(reply))?; - // Bound cold requests even if many prefixes precede them in the queue. - tokio::time::timeout(Duration::from_secs(30), wait) - .await - .context("browse refresh timed out")? - .context("browse worker stopped")? - .map_err(anyhow::Error::msg)?; - let entry = self - .index - .values - .read() - .await + // enqueue may decline during cooldown; it drops the sender in that + // case. A closed reply is not a new internal error or an empty directory. + let _ = tokio::time::timeout(FOREGROUND_REFRESH_TIMEOUT, wait).await; + let existing = self.index.values.read().await.get(&key).cloned(); + if let Some(entry) = existing { + return Ok(self.snapshot_with_summary(&key, entry).await); + } + let refreshing = lock_recover(&self.pending).contains_key(&key); + if let Some(progress) = super::request_progress::current() { + progress.note(if refreshing { "Directory listing is still refreshing in the background; this response contains a partial view." } else { "Directory listing is unavailable. Check storage connectivity and the server log for this request." }); + } + let last_error = lock_recover(&self.refresh) .get(&key) - .cloned() - .context("browse refresh produced no view")?; - Ok(self.snapshot_with_summary(&key, entry).await) + .and_then(|s| s.error.clone()); + Ok(BrowseSnapshot { + tree: catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]), + browse: BrowseStatus { + partial: true, + consistency: CatalogConsistency::BestEffort, + state: if refreshing { + CatalogState::Refreshing + } else { + CatalogState::Unavailable + }, + generation: String::new(), + observed_at: 0, + stale: false, + refreshing, + last_error, + }, + }) } fn enqueue(&self, key: &TreeKey, reply: Option) -> Result<()> { - if let Some(state) = self.refresh.lock().unwrap().get(key) + if let Some(state) = lock_recover(&self.refresh).get(key) && state .retry_at .is_some_and(|deadline| deadline > Instant::now()) { - if let Some(error) = &state.error { - anyhow::bail!("{error}"); - } return Ok(()); } - let mut pending = self.pending.lock().unwrap(); + let mut pending = lock_recover(&self.pending); if let Some(waiters) = pending.get_mut(key) { waiters.retain(|reply| !reply.is_closed()); if let Some(reply) = reply { @@ -326,124 +405,119 @@ impl BrowseCoordinator { } fn snapshot(&self, key: &TreeKey, entry: IndexEntry) -> BrowseSnapshot { - let error = self - .refresh - .lock() - .unwrap() + let error = lock_recover(&self.refresh) .get(key) .and_then(|s| s.error.clone()); BrowseSnapshot { tree: entry.tree, browse: BrowseStatus { - consistency: "best_effort", + partial: false, + consistency: CatalogConsistency::BestEffort, + state: if lock_recover(&self.pending).contains_key(key) { + CatalogState::Refreshing + } else if error.is_some() + || now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64 + { + CatalogState::Stale + } else { + CatalogState::Ready + }, generation: entry.generation, observed_at: entry.observed_at, stale: error.is_some() || now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64, - refreshing: self.pending.lock().unwrap().contains_key(key), + refreshing: lock_recover(&self.pending).contains_key(key), last_error: error, }, } } async fn snapshot_with_summary(&self, key: &TreeKey, mut entry: IndexEntry) -> BrowseSnapshot { + super::request_progress::phase("manifest_summary"); let current_manifest_key = manifest_key(key, None); - let summary = self.manifests.summary_under(¤t_manifest_key).await; - entry.tree.dataset_count = Some(summary.datasets as usize); - entry.tree.trajectory_count = Some(summary.trajectories as usize); - entry.tree.run_count = summary.trajectories as usize; - let cached_trees = self - .index - .values - .read() - .await - .values() - .map(|entry| entry.tree.clone()) - .collect::>(); - let (cached_datasets, cached_trajectories) = cached_leaf_summary(&cached_trees, ""); - if cached_datasets > 0 { - entry.tree.dataset_count = Some(cached_datasets); - entry.tree.trajectory_count = Some(cached_trajectories); - entry.tree.run_count = cached_trajectories; + let observation = self.manifests.get(¤t_manifest_key).await; + // The directory projection supplies names; only local manifest + // observations supply dataset identity, formats and record counts. + if let Some(listing) = &observation + && listing.entries.iter().any(|leaf| { + leaf.kind == PathListKind::Dataset + && (leaf.path == key.prefix || (key.prefix.is_empty() && leaf.path == ".")) + }) + { + entry.tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &listing.entries); } for child in &mut entry.tree.children { if child.kind != "dir" { continue; } - let child_prefix = if key.prefix.is_empty() - || child.path == key.prefix - || child.path.starts_with(&format!("{}/", key.prefix)) - { - child.path.clone() + let child_prefix = child.path.clone(); + let child_manifest_key = manifest_key(key, Some(&child_prefix)); + let listing = self.manifests.get(&child_manifest_key).await; + let leaf = observation + .iter() + .flat_map(|listing| &listing.entries) + .chain(listing.iter().flat_map(|listing| &listing.entries)) + .find(|leaf| leaf.kind == PathListKind::Dataset && leaf.path == child_prefix); + if let Some(leaf) = leaf { + let name = child.name.clone(); + *child = catalog_tree_from_path_list( + &key.dataset, + &key.prefix, + std::slice::from_ref(leaf), + ) + .children + .remove(0); + child.name = name; + child.dataset_count = Some(1); + child.trajectory_count = Some(child.run_count); } else { - format!("{}/{}", key.prefix, child.path) - }; - let summary = self - .manifests - .summary_under(&manifest_key(key, Some(&child_prefix))) - .await; - child.dataset_count = (summary.datasets > 0).then_some(summary.datasets as usize); - child.trajectory_count = - (summary.trajectories > 0).then_some(summary.trajectories as usize); - let child_key = TreeKey { - dataset: key.dataset.clone(), - uri_fingerprint: key.uri_fingerprint.clone(), - prefix: child_prefix.trim_matches('/').to_owned(), - }; - // A directory can be discovered after the background walk started. - // Schedule it here so its descendant manifests become available to - // the next render without making the request wait on storage I/O. - let _ = self.enqueue(&child_key, None); - let cached = self.index.values.read().await.get(&child_key).cloned(); - tracing::info!( - target: "pchronicle.serve", - dataset = %key.dataset, - parent_prefix = %key.prefix, - child_prefix = %child.path, - manifest_datasets = summary.datasets, - manifest_trajectories = summary.trajectories, - projection_hit = cached.is_some(), - "catalog directory summary" - ); - if let Some(cached) = cached { - if let Some(count) = cached.tree.dataset_count.filter(|count| *count > 0) { - child.dataset_count = Some(count); - } - if let Some(count) = cached - .tree - .trajectory_count - .or(Some(cached.tree.run_count)) - .filter(|count| *count > 0) - { - child.trajectory_count = Some(count); - } - } - let (datasets, trajectories) = cached_leaf_summary(&cached_trees, &child_prefix); - if datasets > 0 { - child.dataset_count = Some(datasets); - child.trajectory_count = Some(trajectories); + let summary = self.manifests.summary_under(&child_manifest_key).await; + child.dataset_count = Some(summary.datasets as usize); + child.trajectory_count = Some(summary.trajectories as usize); } } - let child_summary = + let summary = self.manifests.summary_under(¤t_manifest_key).await; + entry.tree.dataset_count = Some(summary.datasets as usize); + entry.tree.trajectory_count = Some(summary.trajectories as usize); + entry.tree.run_count = summary.trajectories as usize; + let partial = !self + .manifests + .is_complete_under(&manifest_key(key, Some("")), &key.prefix) + .await; + let mut snapshot = self.snapshot(key, entry); + snapshot.browse.partial = partial; + if partial && snapshot.browse.state == CatalogState::Ready { + snapshot.browse.state = CatalogState::Partial; + } + snapshot + } +} + +// A root observation is not a complete descendant inventory. This also +// exposes the worker's traversal limit without treating unseen directories as empty. +fn projection_is_partial(values: &HashMap, key: &TreeKey) -> bool { + values + .iter() + .filter(|(other, _)| { + other.dataset == key.dataset + && other.uri_fingerprint == key.uri_fingerprint + && (key.prefix.is_empty() + || other.prefix == key.prefix + || other.prefix.starts_with(&format!("{}/", key.prefix))) + }) + .any(|(_, entry)| { entry .tree .children .iter() - .fold((0usize, 0usize), |(datasets, trajectories), child| { - let datasets = datasets.saturating_add( - child - .dataset_count - .unwrap_or_else(|| (child.kind == "dataset") as usize), - ); - let trajectories = trajectories - .saturating_add(child.trajectory_count.unwrap_or(child.run_count)); - (datasets, trajectories) - }); - entry.tree.dataset_count = Some((summary.datasets as usize).max(child_summary.0)); - entry.tree.trajectory_count = Some((summary.trajectories as usize).max(child_summary.1)); - entry.tree.run_count = entry.tree.trajectory_count.unwrap_or_default(); - self.snapshot(key, entry) - } + .filter(|child| child.kind == "dir") + .any(|child| { + !values.contains_key(&TreeKey { + prefix: child.path.clone(), + ..key.clone() + }) + }) + }) } fn manifest_key(key: &TreeKey, child_prefix: Option<&str>) -> String { @@ -455,24 +529,6 @@ fn manifest_key(key: &TreeKey, child_prefix: Option<&str>) -> String { } } -fn cached_leaf_summary(trees: &[CatalogTree], prefix: &str) -> (usize, usize) { - let mut leaves = HashMap::::new(); - for tree in trees { - for child in &tree.children { - if child.kind == "file" - && child.data_type != "other" - && (child.path == prefix || child.path.starts_with(&format!("{prefix}/"))) - { - leaves.insert(child.path.clone(), child.run_count); - } - } - } - ( - leaves.len(), - leaves.values().copied().fold(0usize, usize::saturating_add), - ) -} - async fn run_worker( mounts: Vec, index: Arc, @@ -483,118 +539,189 @@ async fn run_worker( ) { let mut interval = tokio::time::interval(REFRESH_INTERVAL); interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - // Incremental round over roots and previously browsed prefixes. Foreground - // requests are checked between each scan, not after the entire mount set. - // Breadth-first keeps shallow datasets visible while a deep subtree is - // still being indexed. let mut background = VecDeque::new(); let mut visited = HashSet::new(); + let mut background_budget = 0usize; + let mut background_running = false; + // Manifest failures/backoff never suppress an interactive directory list. + let background_states = Mutex::new(HashMap::::new()); + // FIFO is intentional: children are appended only after their parent + // completes, so the background walk is breadth-first (shallow to deep). + // JoinSet aborts outstanding work when the coordinator is dropped. + let mut jobs = tokio::task::JoinSet::<(TreeKey, bool, Result)>::new(); loop { - let key = tokio::select! { + let (key, is_background) = tokio::select! { biased; - _ = std::future::ready(()), if !background.is_empty() => background.pop_front().unwrap(), - request = receiver.recv() => match request { Some(key) => key, None => break }, + completed = jobs.join_next(), if !jobs.is_empty() => { + let (key, was_background, result) = match completed.unwrap() { + Ok(completed) => completed, + Err(error) => { + tracing::error!(target: "pchronicle.serve", %error, "browse refresh task stopped"); + for (_, waiters) in lock_recover(&pending).drain() { + for waiter in waiters { + let _ = waiter.send(Err(format!("browse refresh task stopped: {error}"))); + } + } + return; + } + }; + if was_background { background_running = false; } + let target_states = if was_background { &background_states } else { states.as_ref() }; + let outcome = match result { + Ok(tree) => { + if was_background { + visited.insert(key.clone()); + for child in &tree.children { + if child.kind == "dir" && visited.len() < 10_000 { + let next = TreeKey { prefix: child.path.clone(), ..key.clone() }; + if visited.insert(next.clone()) { background.push_back(next); } + } + } + } else if visited.len() < 10_000 && visited.insert(key.clone()) { + // An opened directory becomes a manifest observation; + // its descendants still enter the FIFO breadth-first. + background.push_back(key.clone()); + } + lock_recover(target_states).insert(key.clone(), RefreshState { + retry_at: Some(Instant::now() + Duration::from_secs(2)), + ..Default::default() + }); + Ok(()) + } + Err(error) => { + let error = format!("{error:#}"); + let cached_view = index.values.read().await.contains_key(&key); + let mut states = lock_recover(target_states); + let state = states.entry(key.clone()).or_default(); + state.failures = state.failures.saturating_add(1); + state.retry_at = Some(Instant::now() + Duration::from_secs( + (30u64 * (1u64 << state.failures.min(4))).min(300))); + state.error = Some(error.clone()); + tracing::warn!(target: "pchronicle.serve", dataset = %key.dataset, + prefix = %key.prefix, background = was_background, cached_view, %error, "browse refresh failed"); + Err(error) + } + }; + if !was_background { finish(&pending, &key, outcome); } + continue; + } + request = receiver.recv(), if jobs.len() < 2 => match request { + Some(key) => (key, false), None => break + }, + _ = std::future::ready(()), if !background.is_empty() + && background_budget > 0 && !background_running && jobs.len() < 2 => { + background_budget -= 1; + (background.pop_front().unwrap(), true) + }, _ = interval.tick() => { - tracing::info!(target: "pchronicle.serve", mounts = mounts.len(), "browse manifest refresh round started"); - if background.is_empty() { + if background.is_empty() && !background_running { visited.clear(); background.extend(mounts.iter().filter_map(|m| TreeKey::new(m, "").ok())); visited.extend(background.iter().cloned()); } + background_budget = 32; continue; } }; let Some(mount) = mounts .iter() .find(|m| TreeKey::new(m, &key.prefix).ok().as_ref() == Some(&key)) + .cloned() else { continue; }; - if states - .lock() - .unwrap() + let target_states = if is_background { + &background_states + } else { + states.as_ref() + }; + if lock_recover(target_states) .get(&key) .is_some_and(|s| s.retry_at.is_some_and(|deadline| deadline > Instant::now())) { - finish(&pending, &key, Ok(())); + if !is_background { + finish(&pending, &key, Ok(())); + } continue; } - pending.lock().unwrap().entry(key.clone()).or_default(); - let result = async { - let _permit = BROWSE_IO.acquire().await?; - // Bound prefix-list start rate as well as concurrency. One list may - // contain several storage requests; backend I/O gates still apply. - tokio::time::sleep(Duration::from_millis(100)).await; - let location = DatasetLocation::parse(&mount.uri)?; - if let Some(root) = location.local_path() { - // Navigation's list API treats a missing path as empty. For a - // cached view that could erase an offline mount's descendants. - // Deletions are instead established by a successful parent list. - tokio::fs::metadata(root.join(&key.prefix)) - .await - .context("browse path unavailable")?; + if !is_background { + lock_recover(&pending).entry(key.clone()).or_default(); + } + if is_background { + background_running = true; + } + let index = index.clone(); + let manifests = manifests.clone(); + jobs.spawn(async move { + // Reserve at least one global browse slot for foreground work. + let _background = if is_background { + BACKGROUND_IO.acquire().await.ok() + } else { + None + }; + // Interactive turns/run hold foreground object-store demand; pause + // browse refresh so it does not share the S3 pipe with the user. + if is_background { + wait_for_foreground_object_store_idle().await; } - let manifest_key = if key.prefix.is_empty() { - format!("{}\0{}", key.dataset, key.uri_fingerprint) + let work = async { + tokio::time::timeout( + Duration::from_secs(10), + refresh_tree(&mount, &key, &index, &manifests, is_background), + ) + .await + .context("browse list timed out") + .and_then(|result| result) + }; + let result = if is_background { + with_background_object_store_io(work).await } else { - format!("{}\0{}\0{}", key.dataset, key.uri_fingerprint, key.prefix) + work.await }; - let listing = tokio::time::timeout( - Duration::from_secs(20), - manifests.refresh(manifest_key, &location, &key.prefix), - ) + (key, is_background, result) + }); + } +} + +async fn refresh_tree( + mount: &DatasetMount, + key: &TreeKey, + index: &BrowseTreeProjection, + manifests: &ManifestCache, + background: bool, +) -> Result { + let _permit = BROWSE_IO.acquire().await?; + #[cfg(test)] + let block = lock_recover(&index.refresh_blocks).get(key).cloned(); + #[cfg(test)] + let _block = if let Some(block) = block { + Some(block.acquire_owned().await?) + } else { + None + }; + let location = DatasetLocation::parse_with_backend(&mount.uri, mount.backend().cloned())?; + if let Some(root) = location.local_path() { + tokio::fs::metadata(root.join(&key.prefix)) .await - .context("browse list timed out")??; - let entries = listing.entries; - let tree = catalog_tree_from_path_list(&mount.name, &key.prefix, &entries); - // Walk only navigational directories; a Dataset leaf is opaque. - // A bounded frontier prevents the background walk growing without limit. - visited.insert(key.clone()); - for child in &tree.children { - if child.kind == "dir" && visited.len() < 10_000 { - let next = TreeKey::new(mount, &child.path)?; - if visited.insert(next.clone()) { - background.push_back(next); - } - } - } - index.put(key.clone(), tree).await; - Ok::<_, anyhow::Error>(()) - } - .await; - let outcome = match result { - Ok(()) => { - tracing::info!(target: "pchronicle.serve", dataset = %key.dataset, prefix = %key.prefix, "browse manifest level refreshed"); - states.lock().unwrap().insert( - key.clone(), - RefreshState { - retry_at: Some(Instant::now() + Duration::from_secs(2)), - ..Default::default() - }, - ); - Ok(()) - } - Err(error) => { - let error = format!("{error:#}"); - let cached_view = index.values.read().await.contains_key(&key); - let mut states = states.lock().unwrap(); - let state = states.entry(key.clone()).or_default(); - state.failures = state.failures.saturating_add(1); - state.retry_at = Some( - Instant::now() - + Duration::from_secs((30u64 * (1u64 << state.failures.min(4))).min(300)), - ); - state.error = Some(error.clone()); - tracing::warn!(target: "pchronicle.serve", dataset = %key.dataset, prefix = %key.prefix, cached_view, error = %error, "browse refresh failed"); - Err(error) - } - }; - finish(&pending, &key, outcome); + .context("browse path unavailable")?; } + let entries = if background { + manifests + .refresh_for_browse(manifest_key(key, None), &location, &key.prefix) + .await? + .entries + } else { + // Foreground LIST never waits for a background manifest refresh lock, + // probes a sidecar, or replaces the cached manifest observation. + location.list_directory(&key.prefix).await? + }; + let tree = catalog_tree_from_path_list(&mount.name, &key.prefix, &entries); + index.put(key.clone(), tree.clone()).await; + Ok(tree) } fn finish(pending: &Pending, key: &TreeKey, outcome: std::result::Result<(), String>) { - if let Some(waiters) = pending.lock().unwrap().remove(key) { + if let Some(waiters) = lock_recover(pending).remove(key) { for waiter in waiters { let _ = waiter.send(outcome.clone()); } @@ -607,6 +734,8 @@ fn now() -> i64 { /// UI-only tree projection over the core ManifestCache. It preserves the legacy tree wire format. struct BrowseTreeProjection { + #[cfg(test)] + refresh_blocks: Mutex>>, disk: Arc, values: RwLock>, } @@ -628,6 +757,8 @@ impl BrowseTreeProjection { Self { disk, values: RwLock::new(values), + #[cfg(test)] + refresh_blocks: Mutex::new(HashMap::new()), } } @@ -708,6 +839,340 @@ mod tests { DatasetMount::new("test", path.to_string_lossy()).unwrap() } + #[tokio::test] + async fn failed_cold_browse_stays_unavailable_during_cooldown_then_recovers() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("not-yet-created"); + let mount = mount(&source); + let coordinator = + BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; + let key = TreeKey::new(&mount, "").unwrap(); + let (reply, wait) = oneshot::channel(); + coordinator.enqueue(&key, Some(reply)).unwrap(); + assert!(wait.await.unwrap().is_err()); + for _ in 0..3 { + let view = + tokio::time::timeout(Duration::from_millis(100), coordinator.tree(&mount, "")) + .await + .unwrap() + .unwrap(); + assert_eq!(view.browse.state, CatalogState::Unavailable); + assert!(view.browse.partial); + assert!(!view.browse.refreshing); + assert_eq!(view.browse.observed_at, 0); + assert!( + view.browse + .last_error + .as_deref() + .unwrap() + .contains("browse path unavailable") + ); + } + assert!(!lock_recover(&coordinator.pending).contains_key(&key)); + assert_eq!(lock_recover(&coordinator.refresh)[&key].failures, 1); + std::fs::create_dir_all(source.join("recovered")).unwrap(); + lock_recover(&coordinator.refresh).remove(&key); + let (reply, wait) = oneshot::channel(); + coordinator.enqueue(&key, Some(reply)).unwrap(); + wait.await.unwrap().unwrap(); + let view = coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(view.tree.children[0].name, "recovered"); + assert!(view.browse.last_error.is_none()); + assert!(view.browse.observed_at > 0); + } + + #[tokio::test] + async fn blocked_s3_endpoint_does_not_block_another_mount_with_the_same_bucket() { + use axum::{ + Router, + http::{Method, StatusCode}, + }; + use persisting_pchronicle::storage::StoreConfig; + let temp = tempfile::tempdir().unwrap(); + let blocked = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let healthy = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let blocked_endpoint = format!("http://{}", blocked.local_addr().unwrap()); + let healthy_endpoint = format!("http://{}", healthy.local_addr().unwrap()); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = started.clone(); + let mut servers = tokio::task::JoinSet::new(); + servers.spawn(async move { + axum::serve( + blocked, + Router::new().fallback(move || { + let notify = notify.clone(); + async move { + notify.notify_one(); + std::future::pending::().await + } + }), + ) + .await + .unwrap(); + }); + servers.spawn(async move { + axum::serve(healthy, Router::new().fallback(|method: Method| async move { + if method == Method::HEAD { + (StatusCode::NOT_FOUND, "") + } else { + (StatusCode::OK, r#"same-bucketfalsenested/"#) + } + })).await.unwrap(); + }); + let make_mount = |name: &str, endpoint: String| { + DatasetMount::new(name, "s3://same-bucket") + .unwrap() + .with_backend(StoreConfig { + endpoint: Some(endpoint), + region: Some("us-east-1".into()), + access_key: Some("test-key".into()), + secret_key: Some("test-secret".into()), + }) + }; + let blocked_mount = make_mount("prod", blocked_endpoint); + let healthy_mount = make_mount("prod2", healthy_endpoint); + let coordinator = BrowseCoordinator::start_at( + vec![blocked_mount.clone(), healthy_mount.clone()], + temp.path().join("cache"), + ) + .await; + tokio::time::timeout(Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let view = tokio::time::timeout(Duration::from_secs(3), async { + loop { + let view = coordinator.tree(&healthy_mount, "").await.unwrap(); + if view.browse.observed_at > 0 { + break view; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("healthy endpoint must not wait for blocked endpoint's retry budget"); + assert_eq!(view.tree.children[0].name, "nested"); + // A cold request returns a truthful loading view promptly, even while + // its HTTP request is hung. It must not create a second remote refresh. + let view = + tokio::time::timeout(Duration::from_secs(1), coordinator.tree(&blocked_mount, "")) + .await + .unwrap() + .unwrap(); + assert_eq!(view.browse.state, CatalogState::Refreshing); + assert!(view.browse.partial); + assert!( + lock_recover(&coordinator.pending) + .contains_key(&TreeKey::new(&blocked_mount, "").unwrap()) + ); + } + + #[tokio::test] + async fn foreground_lists_while_same_prefix_manifest_is_blocked_then_uses_cached_metadata() { + use axum::{ + Router, + http::{Method, StatusCode, Uri}, + }; + use persisting_pchronicle::storage::StoreConfig; + let temp = tempfile::tempdir().unwrap(); + let manifest_dir = temp.path().join("manifest"); + std::fs::create_dir_all(&manifest_dir).unwrap(); + persisting_pchronicle::storage::write_compact_jsonl_manifest(&manifest_dir, 1, 42).unwrap(); + let manifest = std::fs::read_to_string(manifest_dir.join("chronicle.manifest")).unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let started = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let requests = Arc::new(Mutex::new(Vec::new())); + let (notify, resume, recorded) = (started.clone(), release.clone(), requests.clone()); + let mut servers = tokio::task::JoinSet::new(); + servers.spawn(async move { + axum::serve(listener, Router::new().fallback(move |method: Method, uri: Uri| { + let (notify, resume, recorded, manifest) = (notify.clone(), resume.clone(), recorded.clone(), manifest.clone()); + async move { + lock_recover(&recorded).push((method.clone(), uri.to_string())); + if uri.path() == "/bucket/chronicle.manifest" { + notify.notify_one(); + resume.notified().await; + } + if uri.path() == "/bucket/leaf/chronicle.manifest" { + return (StatusCode::OK, manifest); + } + if method == Method::HEAD { return (StatusCode::NOT_FOUND, String::new()); } + (StatusCode::OK, r#"bucketfalseleaf/"#.to_owned()) + } + })).await.unwrap(); + }); + let mount = DatasetMount::new("s3", "s3://bucket") + .unwrap() + .with_backend(StoreConfig { + endpoint: Some(endpoint), + region: Some("us-east-1".into()), + access_key: Some("key".into()), + secret_key: Some("secret".into()), + }); + let coordinator = + BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; + tokio::time::timeout(Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let view = tokio::time::timeout(Duration::from_secs(2), async { + loop { + let view = coordinator.tree(&mount, "").await.unwrap(); + if view.browse.observed_at > 0 { + break view; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("LIST must not join or wait for the blocked manifest job"); + assert_eq!(view.tree.children[0].kind, "dir"); + assert_eq!(view.tree.dataset_count, Some(0)); + assert!(view.browse.partial); + assert_eq!( + lock_recover(&requests) + .iter() + .filter(|(method, _)| method == Method::HEAD) + .count(), + 1, + "foreground must issue no marker probes" + ); + assert_eq!( + lock_recover(&requests).len(), + 2, + "only background HEAD and foreground LIST" + ); + // Even after the short request cooldown expires, UI polling should + // read the fresh local directory projection, not perform another LIST. + tokio::time::sleep(Duration::from_millis(2100)).await; + coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(lock_recover(&requests).len(), 2); + release.notify_one(); + let root_key = TreeKey::new(&mount, "").unwrap(); + tokio::time::timeout(Duration::from_secs(3), async { + while coordinator + .manifests + .get(&manifest_key(&root_key, Some("leaf"))) + .await + .is_none() + { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let view = coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(view.tree.children[0].kind, "file"); + assert_eq!(view.tree.children[0].data_type, "compact-jsonl"); + assert_eq!(view.tree.children[0].trajectory_count, Some(42)); + assert_eq!(view.tree.dataset_count, Some(1)); + assert_eq!(view.tree.trajectory_count, Some(42)); + assert!(!view.browse.partial); + // Local metadata remains useful after the remote endpoint disappears. + servers.abort_all(); + let view = coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(view.tree.trajectory_count, Some(42)); + } + + #[tokio::test] + async fn cold_tree_returns_loading_and_incomplete_descendants_are_partial() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + std::fs::create_dir_all(source.join("nested")).unwrap(); + let mount = mount(&source); + let coordinator = + BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; + let cold = tokio::time::timeout(Duration::from_secs(1), coordinator.tree(&mount, "")) + .await + .unwrap() + .unwrap(); + assert_eq!(cold.tree.dataset.as_deref(), Some("test")); + assert!(matches!( + cold.browse.state, + CatalogState::Ready | CatalogState::Partial + )); + assert!(cold.browse.observed_at > 0); + let key = TreeKey::new(&mount, "").unwrap(); + let mut entry = IndexEntry { + tree: CatalogTree::default(), + generation: String::new(), + observed_at: now(), + }; + entry + .tree + .children + .push(super::super::explorer::CatalogTreeChild { + kind: "dir".into(), + path: "nested".into(), + ..Default::default() + }); + let mut values = HashMap::from([(key.clone(), entry.clone())]); + assert!(projection_is_partial(&values, &key)); + entry.tree.children.clear(); + values.insert( + TreeKey { + prefix: "nested".into(), + ..key.clone() + }, + entry, + ); + assert!(!projection_is_partial(&values, &key)); + } + + #[tokio::test] + async fn foreground_finishes_while_background_is_blocked_and_drop_cancels_work() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + std::fs::create_dir_all(source.join("nested")).unwrap(); + let mount = mount(&source); + // Construct before starting the worker so the root background scan is + // deterministically blocked while a child foreground request arrives. + let index = Arc::new(BrowseTreeProjection::open(temp.path().join("tree.lance")).await); + let root_key = TreeKey::new(&mount, "").unwrap(); + let blocker = Arc::new(tokio::sync::Semaphore::new(0)); + lock_recover(&index.refresh_blocks).insert(root_key.clone(), blocker.clone()); + let manifests = Arc::new(ManifestCache::open(temp.path().join("manifest.lance")).await); + let pending = Arc::new(Mutex::new(HashMap::new())); + let states = Arc::new(Mutex::new(HashMap::new())); + let (sender, receiver) = mpsc::channel(QUEUE_CAPACITY); + let task = tokio::spawn(run_worker( + vec![mount.clone()], + index, + manifests, + pending.clone(), + states, + receiver, + )); + tokio::time::timeout(Duration::from_secs(5), async { + while BROWSE_IO.available_permits() != 1 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let child = TreeKey::new(&mount, "nested").unwrap(); + let (reply, wait) = oneshot::channel(); + lock_recover(&pending).insert(child.clone(), vec![reply]); + sender.send(child).await.unwrap(); + tokio::time::timeout(Duration::from_secs(5), wait) + .await + .unwrap() + .unwrap() + .unwrap(); + assert!( + BROWSE_IO.available_permits() == 1, + "background must still be blocked" + ); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + // Aborting the scheduler drops its JoinSet and releases active I/O. + let permits = tokio::time::timeout(Duration::from_secs(5), BROWSE_IO.acquire_many(2)) + .await + .unwrap() + .unwrap(); + drop(permits); + } + #[tokio::test] async fn directory_statistics_use_cached_descendants_without_child_projections() { let temp = tempfile::tempdir().unwrap(); @@ -731,29 +1196,14 @@ mod tests { .manifests .refresh_mount( &manifest_key(&root_key, None), - &DatasetLocation::parse(&mount.uri).unwrap(), + &DatasetLocation::parse_with_backend(&mount.uri, mount.backend().cloned()).unwrap(), ) .await .unwrap(); std::fs::remove_dir_all(&source).unwrap(); assert!(browse.index.values.read().await.is_empty()); for (prefix, datasets) in [("", 4), ("nested", 2)] { - let key = TreeKey::new(&mount, prefix).unwrap(); - let listing = browse - .manifests - .get(&manifest_key(&key, None)) - .await - .unwrap(); - let view = browse - .snapshot_with_summary( - &key, - IndexEntry { - tree: catalog_tree_from_path_list(&mount.name, prefix, &listing.entries), - generation: String::new(), - observed_at: now(), - }, - ) - .await; + let view = browse.tree(&mount, prefix).await.unwrap(); let json = serde_json::to_value(view).unwrap(); assert_eq!(json["dataset_count"], datasets); assert_eq!(json["trajectory_count"], 708); @@ -845,21 +1295,21 @@ mod tests { let coordinator = BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; // Hold the global I/O permit to make duplicate cold requests observable. - let gate = BROWSE_IO.acquire().await.unwrap(); + let gate = BROWSE_IO.acquire_many(2).await.unwrap(); let key = TreeKey::new(&mount, "").unwrap(); let (a, ar) = oneshot::channel(); let (b, br) = oneshot::channel(); coordinator.enqueue(&key, Some(a)).unwrap(); coordinator.enqueue(&key, Some(b)).unwrap(); - assert_eq!(coordinator.pending.lock().unwrap().len(), 1); - assert_eq!(coordinator.pending.lock().unwrap()[&key].len(), 2); + assert_eq!(lock_recover(&coordinator.pending).len(), 1); + assert_eq!(lock_recover(&coordinator.pending)[&key].len(), 2); drop(gate); ar.await.unwrap().unwrap(); br.await.unwrap().unwrap(); let old = coordinator.tree(&mount, "").await.unwrap(); assert_eq!(old.tree.children[0].name, "nested"); std::fs::remove_dir_all(&source).unwrap(); - coordinator.refresh.lock().unwrap().remove(&key); + lock_recover(&coordinator.refresh).remove(&key); let (reply, wait) = oneshot::channel(); coordinator.enqueue(&key, Some(reply)).unwrap(); assert!(wait.await.unwrap().is_err()); diff --git a/crates/persisting-pchronicle-cli/src/sync.rs b/crates/persisting-pchronicle-cli/src/sync.rs index d0cb1e84c..74f5e4d44 100644 --- a/crates/persisting-pchronicle-cli/src/sync.rs +++ b/crates/persisting-pchronicle-cli/src/sync.rs @@ -174,13 +174,7 @@ pub(crate) async fn run( let mut pending = BTreeSet::new(); let mut failures = 0u32; loop { - tokio::time::sleep(interval).await; - while let Ok(path) = changes_rx.try_recv() { - pending.insert(path); - } - if pending.is_empty() { - continue; - } + collect_pending_changes(&mut changes_rx, &mut pending).await?; match super::exchange::sync_snapshot( &source_uri, @@ -239,6 +233,21 @@ pub(crate) async fn run( } } +// A failed batch remains pending. Only idle syncs wait for a new change; +// retries also absorb any changes received during backoff. +async fn collect_pending_changes( + changes: &mut tokio::sync::mpsc::Receiver, + pending: &mut BTreeSet, +) -> Result<()> { + if pending.is_empty() { + pending.insert(changes.recv().await.context("sync watcher stopped")?); + } + while let Ok(path) = changes.try_recv() { + pending.insert(path); + } + Ok(()) +} + fn prepare_destination(uri: &str, name: &str) -> Result { anyhow::ensure!(!uri.is_empty(), "sync {name} target must not be empty"); let location = DatasetLocation::parse(uri)?; @@ -348,6 +357,41 @@ fn changed_paths( mod tests { use super::*; + #[tokio::test] + async fn failed_sync_retries_without_new_changes_and_coalesces_backoff_changes() { + let (sender, mut receiver) = tokio::sync::mpsc::channel(4); + sender.send(PathBuf::from("a.json")).await.unwrap(); + let mut pending = BTreeSet::new(); + collect_pending_changes(&mut receiver, &mut pending) + .await + .unwrap(); + // Simulate a failed attempt: retain the batch, with no new file event. + tokio::time::timeout( + Duration::from_millis(100), + collect_pending_changes(&mut receiver, &mut pending), + ) + .await + .unwrap() + .unwrap(); + sender.send(PathBuf::from("b.json")).await.unwrap(); + collect_pending_changes(&mut receiver, &mut pending) + .await + .unwrap(); + assert_eq!( + pending, + BTreeSet::from([PathBuf::from("a.json"), PathBuf::from("b.json")]) + ); + pending.clear(); + assert!( + tokio::time::timeout( + Duration::from_millis(20), + collect_pending_changes(&mut receiver, &mut pending) + ) + .await + .is_err() + ); + } + #[test] fn prepare_destination_preserves_object_store_uri() { assert_eq!( diff --git a/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs b/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs index add3c50c5..c8a6ccfb9 100644 --- a/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs +++ b/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs @@ -96,7 +96,6 @@ permissions = ["read"] .map(|v| v.to_str().unwrap_or("")) .collect::>() .join(","); - assert!(timings.contains("parent_total"), "{timings}"); assert!(timings.contains("total;"), "{timings}"); let body: Value = response.json().await?; assert_eq!(body["children"].as_array().map(Vec::len), Some(1)); @@ -109,6 +108,14 @@ permissions = ["read"] .send() .await?; assert_eq!(query.status(), 200); + let timings = query + .headers() + .get_all("server-timing") + .iter() + .map(|v| v.to_str().unwrap_or("")) + .collect::>() + .join(","); + assert!(timings.contains("parent_total"), "{timings}"); let query: Value = query.json().await?; assert_eq!( query["rows"], @@ -116,6 +123,36 @@ permissions = ["read"] "{query}" ); } + // Concurrent requests in one scope may now use different exec workers. + // Each process must retain the same grants and storage identity. + let mut concurrent = tokio::task::JoinSet::new(); + for user in ["alice", "bob", "alice", "bob"] { + let client = client.clone(); + let endpoint = endpoint.to_owned(); + concurrent.spawn(async move { + let response = client + .post(format!("http://{endpoint}/api/query/evidence")) + .header("x-pchronicle-access-key", format!("{user}-ak")) + .header("x-pchronicle-secret-key", format!("{user}-sk")) + .json( + &json!({"sql":"SELECT session_id FROM runs", "max_rows":10, "max_bytes":4096}), + ) + .send() + .await?; + assert_eq!(response.status(), 200); + let body: Value = response.json().await?; + let expected = if user == "alice" { + "left-session" + } else { + "right-session" + }; + assert_eq!(body["rows"], json!([{"session_id":expected}])); + Ok::<_, anyhow::Error>(()) + }); + } + while let Some(result) = concurrent.join_next().await { + result??; + } let response = client .get(format!("{url}?dataset=right")) .header("x-pchronicle-access-key", "alice-ak") @@ -214,7 +251,17 @@ async fn exec_worker_handles_multiple_frames_and_exits_on_eof() -> Result<()> { json!({"method":"GET", "uri":"/api/health", "headers":[], "body":[]}), ) .await?; - assert_eq!(read_frame(&mut output).await?["status"], 200); + loop { + let event = read_frame(&mut output).await?; + match event["type"].as_str() { + Some("Progress") => continue, + Some("Response") => { + assert_eq!(event["value"]["status"], 200); + break; + } + _ => anyhow::bail!("unexpected worker frame: {event}"), + } + } assert_eq!(child.id(), pid); assert!(child.try_wait()?.is_none()); } diff --git a/crates/persisting-pchronicle/Cargo.toml b/crates/persisting-pchronicle/Cargo.toml index 48bb277a3..ba813798d 100644 --- a/crates/persisting-pchronicle/Cargo.toml +++ b/crates/persisting-pchronicle/Cargo.toml @@ -18,7 +18,9 @@ lance-store = [ "dep:lance-datafusion", "dep:lance-file", "dep:lance-index", + "dep:lance-io", "dep:opendal", + "dep:object_store", ] # Remote object stores are separated from the local Lance engine so pPilot and # other local-only consumers do not compile cloud SDKs. `s3-store` remains a @@ -40,6 +42,7 @@ proptest = ["lance-store"] [dependencies] anyhow.workspace = true async-trait.workspace = true +bytes.workspace = true base64.workspace = true blake3.workspace = true chrono.workspace = true @@ -53,9 +56,11 @@ lance-core = { workspace = true, optional = true } lance-file = { workspace = true, optional = true } lance-datafusion = { workspace = true, optional = true } lance-index = { workspace = true, optional = true } +lance-io = { workspace = true, optional = true } lance-linalg = { workspace = true, optional = true } lance-table = { workspace = true, optional = true } opendal = { workspace = true, optional = true } +object_store = { workspace = true, optional = true } persisting-agentctl.workspace = true persisting-events.workspace = true serde = { workspace = true, features = ["derive"] } @@ -375,3 +380,7 @@ required-features = ["proptest"] name = "proptest_unknown_fields_roundtrip" path = "tests/proptests/unknown_fields_roundtrip.rs" required-features = ["proptest"] + +[[bench]] +name = "lance_cache" +harness = false diff --git a/crates/persisting-pchronicle/benches/lance_cache.rs b/crates/persisting-pchronicle/benches/lance_cache.rs new file mode 100644 index 000000000..72641f012 --- /dev/null +++ b/crates/persisting-pchronicle/benches/lance_cache.rs @@ -0,0 +1,24 @@ +use bytes::Bytes; +use criterion::{Criterion, criterion_group, criterion_main}; +use persisting_pchronicle::storage::{BlockCache, CacheConfig}; +use std::hint::black_box; +use tempfile::tempdir; + +fn block_cache(c: &mut Criterion) { + let rt = tokio::runtime::Runtime::new().unwrap(); + let dir = tempdir().unwrap(); + let cache = BlockCache::new(CacheConfig::new( + dir.path().into(), + 64 * 1024 * 1024, + 64 * 1024, + )); + let path = dir.path().join("block"); + rt.block_on(cache.get_or_fetch(&path, 4, async { Ok(Bytes::from_static(b"data")) })) + .unwrap(); + c.bench_function("block_cache_hit", |b| { + b.iter(|| rt.block_on(cache.get_or_fetch(black_box(&path), 4, async { unreachable!() }))) + }); +} + +criterion_group!(benches, block_cache); +criterion_main!(benches); diff --git a/crates/persisting-pchronicle/src/revision.rs b/crates/persisting-pchronicle/src/revision.rs index b59fe6ee9..411e2d9f9 100644 --- a/crates/persisting-pchronicle/src/revision.rs +++ b/crates/persisting-pchronicle/src/revision.rs @@ -4,7 +4,6 @@ use std::sync::Arc; use anyhow::{Context, Result}; use futures::TryStreamExt; -use lance::Dataset; use lance::dataset::{InsertBuilder, MergeInsertBuilder, WhenMatched, WhenNotMatched}; use lance::deps::arrow_array::{Array, RecordBatch, RecordBatchIterator, StringArray}; use lance::deps::arrow_schema::{DataType, Field, Schema}; @@ -94,7 +93,7 @@ fn text(batch: &RecordBatch, column: &str, row: usize) -> Result { pub async fn read_revisions(session: &StoryCoords) -> Result> { let uri = revision_dataset_path(session)?; - let dataset = match Dataset::open(&uri).await { + let dataset = match crate::storage::open_lance_dataset(&uri).await { Ok(dataset) => dataset, Err(lance::Error::DatasetNotFound { .. }) => return Ok(Vec::new()), Err(error) => return Err(anyhow::anyhow!(error)).context("open revisions.lance"), @@ -139,7 +138,7 @@ pub async fn write_revisions(session: &StoryCoords, rows: &[RevisionRow]) -> Res } let _guard = crate::store::dataset_write_lock::acquire(&uri).await?; let batch = batch(rows)?; - match Dataset::open(&uri).await { + match crate::storage::open_lance_dataset(&uri).await { Ok(dataset) => { let reader = Box::new(RecordBatchIterator::new(vec![Ok(batch)], schema())); MergeInsertBuilder::try_new(Arc::new(dataset), vec!["revision_id".into()])? diff --git a/crates/persisting-pchronicle/src/search/find.rs b/crates/persisting-pchronicle/src/search/find.rs index 312909d9c..9cb05599b 100644 --- a/crates/persisting-pchronicle/src/search/find.rs +++ b/crates/persisting-pchronicle/src/search/find.rs @@ -66,12 +66,11 @@ impl FindTextField { pub fn columns(&self) -> &'static [&'static str] { match self { Self::Reasoning => &["reasoning_content"], - Self::Message | Self::User | Self::Assistant => &["message_value"], + Self::Content | Self::Message | Self::User | Self::Assistant => &["message_value"], Self::Observation => &["observation"], Self::Prompt | Self::System => &["prompt", "message_value"], Self::Model => &["model_name"], Self::Env => &["env"], - Self::Content => &["message_value", "observation", "prompt"], Self::All => &[ "message_value", "reasoning_content", diff --git a/crates/persisting-pchronicle/src/search/storyline.rs b/crates/persisting-pchronicle/src/search/storyline.rs index 8eaca396f..8f9171fe1 100644 --- a/crates/persisting-pchronicle/src/search/storyline.rs +++ b/crates/persisting-pchronicle/src/search/storyline.rs @@ -45,14 +45,7 @@ const STORYLINE_FTS_COLUMNS: &[&str] = &[ ]; /// Default Storyline step columns searched by an unqualified text query. -pub const STORYLINE_STEP_SEARCH_COLUMNS: &[&str] = &[ - "message_value", - "reasoning_content", - "model_name", - "observation", - "env", - "prompt", -]; +pub const STORYLINE_STEP_SEARCH_COLUMNS: &[&str] = &["message_value"]; /// Ensure all FTS and JSON search indexes supported by a Storyline table. pub(crate) async fn ensure_storyline_search_indexes(dataset: &mut Dataset) -> Result<()> { diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 096522bc7..a1c93e03f 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -1,6 +1,10 @@ //! pChronicle 的持久化存储入口。 pub type Result = anyhow::Result; +#[cfg(feature = "lance-store")] +pub use crate::store::opendal_store::StoreConfig; +#[cfg(feature = "lance-store")] +pub use crate::store::opendal_store::{RetryPatience, set_retry_patience}; /// Parse an integer byte size with binary IEC suffixes. pub fn parse_byte_size(value: &str) -> std::result::Result { @@ -49,6 +53,12 @@ pub use crate::discovery::{ drop_lifecycle_run_partitions, expand_story_locations, expand_story_locations_blocking, }; +#[cfg(feature = "lance-store")] +pub use crate::store::blockcache::{ + BlockCache, CacheConfig, CacheStats, CachedObjectStore, DEFAULT_BLOCK_SIZE_BYTES, + DEFAULT_CAPACITY_BYTES, LanceCacheWrapper, SERVE_CAPACITY_BYTES, capacity_for_serve, + configured_capacity_bytes, default_cache_dir, lance_store_params, +}; #[cfg(feature = "lance-store")] pub use crate::store::index_build_progress::{ Guard as IndexBuildProgressGuard, install as install_index_build_progress, @@ -56,29 +66,42 @@ pub use crate::store::index_build_progress::{ #[cfg(feature = "lance-store")] pub use crate::store::object_store_io_gate::{ IoKind as ObjectStoreIoKind, ObjectStoreGateSnapshot, ObjectStoreThrottleEvent, - ObjectStoreThrottleHookGuard, format_aimd_flow_label as format_object_store_aimd_flow_label, + ObjectStoreThrottleHookGuard, foreground_object_store_demand, + format_aimd_flow_label as format_object_store_aimd_flow_label, install_throttle_hook as install_object_store_throttle_hook, - snapshot as object_store_gate_snapshot, + snapshot as object_store_gate_snapshot, wait_for_foreground_object_store_idle, + with_background_object_store_io, }; +#[cfg(feature = "lance-store")] +pub async fn open_lance_dataset(uri: &str) -> lance::Result { + lance::dataset::builder::DatasetBuilder::from_uri(uri) + .with_store_params(crate::store::blockcache::lance_store_params( + crate::store::blockcache::configured_capacity_bytes(), + )) + .load() + .await +} + #[cfg(feature = "lance-store")] pub use crate::store::{ AppendOutcome, AttemptRecord, AttemptRecordState, AttemptRegistry, CachedDataset, - CatalogDataset, CatalogErrorPolicy, CatalogEventProvenance, CatalogEventView, CatalogNamespace, - CatalogPage, CatalogProjectionStatus, CatalogSnapshotOptions, CatalogSourceDescription, - CatalogSourceKind, CatalogSourceRevision, CatalogSourceStatus, CatalogStorylineKey, - CatalogTrajectoryBundle, ChronicleManifest, CommitRunOutcome, CompactJsonlBuildPhase, - CompactJsonlColumn, CompactJsonlImportEvent, CompactJsonlOffload, CompactJsonlOptions, - CompactJsonlRecord, CompactJsonlStore, DEFAULT_CONTENT_OFFLOAD_THRESHOLD, - DEFAULT_CONTENT_PREVIEW_BYTES, DEFAULT_DATASET_NAME, DEFAULT_MAX_CHUNK_BYTES, - DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, DEFAULT_PHYSICAL_PAGE_LIMIT, - Dataset, DatasetCatalogSnapshot, DatasetLocation, DatasetLocationKind, DatasetMount, - DatasetResolver, DiscoveredSource, EventFactSnapshot, EventLogLayoutStats, EventWriterFence, - ExportOutcome, ImportableObjectEvent, LanceMaintenanceOptions, LanceMaintenanceReport, - LeaseAcquireOutcome, LocationSummary, ManifestCache, ManifestKind, ManifestListing, - ManifestReadMode, ManifestStats, NamespacePath, PathListEntry, PathListKind, PersistentCache, - PhysicalColumn, PhysicalDataFile, PhysicalFileLayout, PhysicalFragment, PhysicalLayout, - PhysicalPage, PhysicalPagePreview, PhysicalPageQuery, PhysicalSource, PhysicalTable, + CatalogConsistency, CatalogDataset, CatalogErrorPolicy, CatalogEventProvenance, + CatalogEventView, CatalogNamespace, CatalogPage, CatalogProjectionStatus, + CatalogSnapshotOptions, CatalogSourceDescription, CatalogSourceKind, CatalogSourceRevision, + CatalogSourceStatus, CatalogState, CatalogStatus, CatalogStorylineKey, CatalogTrajectoryBundle, + ChronicleManifest, CommitRunOutcome, CompactJsonlBuildPhase, CompactJsonlColumn, + CompactJsonlImportEvent, CompactJsonlOffload, CompactJsonlOptions, CompactJsonlRecord, + CompactJsonlStore, DEFAULT_CONTENT_OFFLOAD_THRESHOLD, DEFAULT_CONTENT_PREVIEW_BYTES, + DEFAULT_DATASET_NAME, DEFAULT_MAX_CHUNK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_BYTES, + DEFAULT_MAX_EVENT_FALLBACK_ROWS, DEFAULT_PHYSICAL_PAGE_LIMIT, Dataset, DatasetCatalogSnapshot, + DatasetLocation, DatasetLocationKind, DatasetMount, DatasetResolver, DiscoveredSource, + EventFactSnapshot, EventLogLayoutStats, EventWriterFence, ExportOutcome, ImportableObjectEvent, + LanceMaintenanceOptions, LanceMaintenanceReport, LeaseAcquireOutcome, LocationSummary, + ManifestCache, ManifestKind, ManifestListing, ManifestReadMode, ManifestRefreshReport, + ManifestStats, NamespacePath, PathListEntry, PathListKind, PersistentCache, PhysicalColumn, + PhysicalDataFile, PhysicalFileLayout, PhysicalFragment, PhysicalLayout, PhysicalPage, + PhysicalPagePreview, PhysicalPageQuery, PhysicalSource, PhysicalTable, ProjectionSourceSnapshot, QueryScope, RawEventLanceAppender, RawEventLanceStore, ReplayOutcome, ResolveMode, ResolveTarget, RunControlStore, ShallowNavEntry, StorylineContentOptions, StorylineContentReadMode, StorylineDataSource, StorylineDataSourceOptions, StorylineLanceStore, diff --git a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs new file mode 100644 index 000000000..772473084 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs @@ -0,0 +1,619 @@ +//! Lance bridge and ObjectStore read-through adapter. Only validated range reads +//! are cached; HEAD and full GET retain the backend's native semantics/stream. + +use super::{BlockCache, CacheConfig}; +use crate::store::object_store_io_gate::{self as io_gate, IoKind}; +use async_trait::async_trait; +use bytes::Bytes; +use futures::{StreamExt, stream::BoxStream}; +use object_store::{ + Attributes, CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, + ObjectMeta, ObjectStore, PutMultipartOptions, PutOptions, PutPayload, PutResult, + Result as ObjectResult, UploadPart, path::Path, +}; +use std::{ + ops::Range, + sync::{Arc, Mutex}, + time::{Duration, Instant}, +}; + +/// A round trip slower than this is worth naming next to the request that +/// caused it; the enclosing span already carries the path and byte range. +const SLOW_REQUEST: Duration = Duration::from_millis(200); + +/// Admission and feedback live at the object-request boundary. The backend +/// already retries transport errors; do not replay an entire Lance open here. +async fn remote_request( + uri: &str, + request: impl std::future::Future>, +) -> ObjectResult { + let _permit = io_gate::acquire(uri, IoKind::Read).await; + // Time the transfer alone. Admission queueing is reported by the gate, so + // separating the two keeps a slow backend from reading as contention. + let started = Instant::now(); + let result = request.await; + let elapsed = started.elapsed(); + if elapsed >= SLOW_REQUEST { + tracing::debug!( + target: "pchronicle.object_store_gate", + elapsed_ms = elapsed.as_millis() as u64, + "remote object read" + ); + } + match &result { + Ok(_) => io_gate::note_success(uri), + Err(error) if io_gate::is_transient_error(error) => { + io_gate::note_failure(uri, IoKind::Read) + } + Err(_) => {} + } + result +} + +async fn remote_write_request( + uri: &str, + request: impl std::future::Future>, +) -> ObjectResult { + let _permit = io_gate::acquire(uri, IoKind::Write).await; + let result = request.await; + match &result { + Ok(_) => io_gate::note_success(uri), + Err(error) if io_gate::is_transient_error(error) => { + io_gate::note_failure(uri, IoKind::Write) + } + Err(_) => {} + } + result +} + +struct GatedMultipartUpload { + inner: Arc>>, + io_scope: String, +} + +impl std::fmt::Debug for GatedMultipartUpload { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("GatedMultipartUpload") + .field("io_scope", &self.io_scope) + .finish() + } +} + +#[async_trait] +impl MultipartUpload for GatedMultipartUpload { + fn put_part(&mut self, data: PutPayload) -> UploadPart { + let inner = Arc::clone(&self.inner); + let scope = self.io_scope.clone(); + Box::pin(async move { + // ponytail: serialize multipart parts because object_store's trait + // returns a 'static future from &mut self; parallelism can be added + // when the trait exposes owned part handles. + let mut inner = inner.lock().await; + remote_write_request(&scope, inner.put_part(data)).await + }) + } + + async fn complete(&mut self) -> ObjectResult { + let mut inner = self.inner.lock().await; + remote_write_request(&self.io_scope, inner.complete()).await + } + + async fn abort(&mut self) -> ObjectResult<()> { + let mut inner = self.inner.lock().await; + remote_write_request(&self.io_scope, inner.abort()).await + } +} + +#[derive(Debug, Clone)] +pub struct LanceCacheWrapper { + pub capacity_bytes: u64, + pub block_size_bytes: u64, + pub root: std::path::PathBuf, +} + +fn backend_identity_from_env() -> String { + let backend: std::collections::BTreeMap<_, _> = std::env::vars() + .filter(|(key, _)| { + key.starts_with("AWS_") || key.starts_with("AZURE_") || key.starts_with("GOOGLE_") + }) + .collect(); + blake3::hash(&serde_json::to_vec(&backend).unwrap_or_default()) + .to_hex() + .to_string() +} + +impl lance_io::object_store::WrappingObjectStore for LanceCacheWrapper { + fn wrap(&self, prefix: &str, original: Arc) -> Arc { + // Lance uses e.g. s3$bucket, not a URI. Convert only its scheme delimiter. + let uri = prefix.replacen('$', "://", 1); + if !io_gate::is_remote_uri(&uri) { + return original; + } + let config = CacheConfig::new( + self.root.clone(), + self.capacity_bytes, + self.block_size_bytes, + ); + Arc::new(CachedObjectStore::new( + original, + config, + format!("{uri}#{}", backend_identity_from_env()), + )) + } +} + +pub fn lance_wrapper(capacity_bytes: u64) -> Arc { + let mut config = CacheConfig::from_env(); + config.capacity_bytes = capacity_bytes; + Arc::new(LanceCacheWrapper { + capacity_bytes, + block_size_bytes: config.block_size_bytes, + root: config.root, + }) +} + +pub fn lance_store_params(capacity_bytes: u64) -> lance_io::object_store::ObjectStoreParams { + lance_io::object_store::ObjectStoreParams { + object_store_wrapper: Some(lance_wrapper(capacity_bytes)), + ..Default::default() + } +} + +/// How long a ranged read may reuse metadata it already fetched. +/// +/// A ranged read cannot name its cache block until it knows the object's +/// version, etag and size, so it HEADs the backend first. Asking every time +/// means a cache hit still pays a full round trip and the block cache buys no +/// latency at all. Lance never reuses a path for different bytes -- data files, +/// index pages and manifests are all named per commit -- so the answer is +/// stable for far longer than this. Keep the window short anyway so a backend +/// that does rewrite a path recovers on its own. +const HEAD_TTL: Duration = Duration::from_secs(60); + +/// Objects to remember at once. A reader touches the fragments and index pages +/// of a handful of datasets, so this holds a whole working set. +const HEAD_CAPACITY: usize = 8192; + +/// Metadata for objects this store has already looked up. +#[derive(Debug, Default)] +struct HeadMemo { + entries: std::collections::HashMap, +} + +/// A path plus the version asked for, which is all that changes the answer; +/// remaining conditions are checked against the metadata by the caller. +type HeadKey = (String, Option); + +impl HeadMemo { + fn get(&self, key: &HeadKey) -> Option<(ObjectMeta, Attributes)> { + let (meta, attributes, fetched) = self.entries.get(key)?; + (fetched.elapsed() < HEAD_TTL).then(|| (meta.clone(), attributes.clone())) + } + + fn forget(&mut self, path: &str) { + self.entries.retain(|(known, _), _| known != path); + } + + fn insert(&mut self, key: HeadKey, meta: ObjectMeta, attributes: Attributes) { + if self.entries.len() >= HEAD_CAPACITY { + self.entries + .retain(|_, (_, _, fetched)| fetched.elapsed() < HEAD_TTL); + if self.entries.len() >= HEAD_CAPACITY { + self.entries.clear(); + } + } + self.entries.insert(key, (meta, attributes, Instant::now())); + } +} + +#[derive(Debug, Clone)] +pub struct CachedObjectStore { + inner: Arc, + cache: BlockCache, + block_size_bytes: u64, + store_uri: String, + io_scope: String, + heads: Arc>, +} + +impl CachedObjectStore { + /// `store_uri` is the cache namespace: include the backend/credential scope + /// as well as bucket. Use a digest for any sensitive configuration. + pub fn new(inner: Arc, config: CacheConfig, store_uri: String) -> Self { + let block_size_bytes = config.block_size_bytes; + Self { + inner, + cache: BlockCache::new(config), + block_size_bytes, + io_scope: io_gate::scope_key(&store_uri), + store_uri, + heads: Arc::new(Mutex::new(HeadMemo::default())), + } + } + + fn heads(&self) -> std::sync::MutexGuard<'_, HeadMemo> { + self.heads + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + } + + async fn read_block( + &self, + path: &Path, + meta: &ObjectMeta, + options: GetOptions, + key: &str, + range: Range, + ) -> ObjectResult { + let block = range.start / self.block_size_bytes; + let file = self.cache.block_path(key, block); + self.cache + .get_or_fetch(&file, (range.end - range.start) as usize, async { + remote_request(&self.io_scope, async { + let result = self.inner.get_opts(path, options.clone()).await?; + // Do not publish mismatched bytes even if a compatible backend + // ignores If-Match or the requested VersionId. + if result.range != range + || result.meta.size != meta.size + || options + .version + .as_ref() + .is_some_and(|v| result.meta.version.as_ref() != Some(v)) + || options + .if_match + .as_ref() + .is_some_and(|tag| result.meta.e_tag.as_ref() != Some(tag)) + { + // Remembered metadata describes bytes the backend no + // longer serves; the next read must ask again. + self.heads().forget(path.as_ref()); + return Err(object_store::Error::Precondition { + path: path.to_string(), + source: "object changed while reading cached block".into(), + }); + } + result.bytes().await + }) + .await + }) + .await + } +} + +impl std::fmt::Display for CachedObjectStore { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "cached({})", self.inner) + } +} + +#[async_trait] +impl ObjectStore for CachedObjectStore { + async fn put_opts(&self, p: &Path, b: PutPayload, o: PutOptions) -> ObjectResult { + remote_write_request(&self.io_scope, self.inner.put_opts(p, b, o)).await + } + async fn put_multipart_opts( + &self, + p: &Path, + o: PutMultipartOptions, + ) -> ObjectResult> { + let upload = + remote_write_request(&self.io_scope, self.inner.put_multipart_opts(p, o)).await?; + Ok(Box::new(GatedMultipartUpload { + inner: Arc::new(tokio::sync::Mutex::new(upload)), + io_scope: self.io_scope.clone(), + })) + } + async fn get_opts(&self, p: &Path, o: GetOptions) -> ObjectResult { + if o.head || o.range.is_none() { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + } + // Metadata for the requested version. Conditions beyond the version are + // checked locally below, so remembered metadata reaches the same verdict + // a conditional round trip would have returned. + let head_key = (p.to_string(), o.version.clone()); + let remembered = self.heads().get(&head_key); + let (head_meta, head_attributes) = match remembered { + Some(head) => head, + None => { + let head = remote_request( + &self.io_scope, + self.inner.get_opts( + p, + GetOptions { + head: true, + range: None, + ..o.clone() + }, + ), + ) + .await?; + let head = (head.meta, head.attributes); + self.heads() + .insert(head_key.clone(), head.0.clone(), head.1.clone()); + head + } + }; + o.check_preconditions(&head_meta)?; + let version = head_meta + .version + .clone() + .filter(|v| !v.is_empty() && v != "null"); + let etag = head_meta + .e_tag + .clone() + .filter(|v| !v.is_empty() && !v.starts_with("W/")); + if version.is_none() && etag.is_none() { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + } + // A backend that did not identify the requested version cannot safely + // populate a version-keyed cache. Preserve its native GET semantics. + if o.version.is_some() && o.version != head_meta.version { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + } + let Some(range) = o.range.as_ref() else { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + }; + let range = + range + .as_range(head_meta.size) + .map_err(|source| object_store::Error::Generic { + store: "pchronicle-cache", + source: Box::new(source), + })?; + // v2 deliberately never reads the previous unnamespaced cache entries. + let key = serde_json::to_string(&( + "v2", + &self.store_uri, + p.as_ref(), + &version, + &etag, + head_meta.size, + self.block_size_bytes, + )) + .unwrap_or_else(|_| { + format!( + "v2:{}:{}:{:?}:{:?}:{}:{}", + self.store_uri, + p.as_ref(), + version, + etag, + head_meta.size, + self.block_size_bytes + ) + }); + let options = GetOptions { + version: version.or(o.version), + if_match: etag.or(o.if_match), + ..o + }; + let this = self.clone(); + let path = p.clone(); + let meta = head_meta.clone(); + let block_size = self.block_size_bytes; + let blocks = (range.start / block_size + ..range.end.saturating_add(block_size - 1) / block_size) + .map(|block| { + let block_start = block * block_size; + let block_end = block_start.saturating_add(block_size).min(meta.size); + let visible_start = range.start.max(block_start); + let visible_end = range.end.min(block_end); + (block_start, block_end, visible_start, visible_end) + }) + .collect::>(); + // Keep at most four misses in flight. The stream remains lazy, while + // adjacent uncached blocks overlap their S3 requests. + let stream = futures::stream::iter(blocks) + .map( + move |(block_start, block_end, visible_start, visible_end)| { + let this = this.clone(); + let path = path.clone(); + let meta = meta.clone(); + let key = key.clone(); + let mut options = options.clone(); + async move { + options.range = Some((block_start..block_end).into()); + let bytes = this + .read_block(&path, &meta, options, &key, block_start..block_end) + .await?; + Ok::<_, object_store::Error>(bytes.slice( + (visible_start - block_start) as usize + ..(visible_end - block_start) as usize, + )) + } + }, + ) + .buffered(4); + Ok(GetResult { + payload: GetResultPayload::Stream(stream.boxed()), + meta: head_meta, + range, + attributes: head_attributes, + }) + } + fn delete_stream( + &self, + p: BoxStream<'static, ObjectResult>, + ) -> BoxStream<'static, ObjectResult> { + let uri = self.io_scope.clone(); + let stream = self.inner.delete_stream(p); + futures::stream::unfold(stream, move |mut stream| { + let uri = uri.clone(); + async move { + let _permit = io_gate::acquire(&uri, IoKind::Write).await; + match stream.next().await { + Some(result) => { + if result.is_ok() { + io_gate::note_success(&uri); + } else if result + .as_ref() + .err() + .is_some_and(io_gate::is_transient_error) + { + io_gate::note_failure(&uri, IoKind::Write); + } + Some((result, stream)) + } + None => None, + } + } + }) + .boxed() + } + fn list(&self, p: Option<&Path>) -> BoxStream<'static, ObjectResult> { + let uri = self.io_scope.clone(); + let stream = self.inner.list(p); + futures::stream::unfold(stream, move |mut stream| { + let uri = uri.clone(); + async move { + let _permit = io_gate::acquire(&uri, IoKind::Read).await; + match stream.next().await { + Some(result) => { + if result.is_ok() { + io_gate::note_success(&uri); + } else if result + .as_ref() + .err() + .is_some_and(io_gate::is_transient_error) + { + io_gate::note_failure(&uri, IoKind::Read); + } + Some((result, stream)) + } + None => None, + } + } + }) + .boxed() + } + async fn list_with_delimiter(&self, p: Option<&Path>) -> ObjectResult { + remote_request(&self.io_scope, self.inner.list_with_delimiter(p)).await + } + async fn copy_opts(&self, a: &Path, b: &Path, o: CopyOptions) -> ObjectResult<()> { + remote_write_request(&self.io_scope, self.inner.copy_opts(a, b, o)).await + } +} + +#[cfg(test)] +mod tests { + use super::*; + use object_store::{ObjectStoreExt, memory::InMemory}; + + #[tokio::test] + async fn range_cache_preserves_head_and_full_get_semantics() { + let root = tempfile::tempdir().unwrap(); + let inner = Arc::new(InMemory::new()); + let path = Path::from("dataset/data.lance"); + inner + .put(&path, Bytes::from_static(b"0123456789").into()) + .await + .unwrap(); + let cached = CachedObjectStore::new( + inner.clone(), + CacheConfig::new(root.path().into(), 1024, 4), + "s3://test-bucket".into(), + ); + + let head = cached.head(&path).await.unwrap(); + assert_eq!(head.size, 10); + assert_eq!( + cached.get_range(&path, 3..8).await.unwrap(), + Bytes::from_static(b"34567") + ); + assert_eq!( + cached.get(&path).await.unwrap().bytes().await.unwrap(), + Bytes::from_static(b"0123456789") + ); + } + + /// Counts what actually leaves for the backend. + #[derive(Debug)] + struct Counted { + inner: Arc, + gets: Arc, + } + + impl std::fmt::Display for Counted { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "counted") + } + } + + #[async_trait] + impl ObjectStore for Counted { + async fn put_opts( + &self, + p: &Path, + b: PutPayload, + o: PutOptions, + ) -> ObjectResult { + self.inner.put_opts(p, b, o).await + } + async fn put_multipart_opts( + &self, + p: &Path, + o: PutMultipartOptions, + ) -> ObjectResult> { + self.inner.put_multipart_opts(p, o).await + } + async fn get_opts(&self, p: &Path, o: GetOptions) -> ObjectResult { + self.gets.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + self.inner.get_opts(p, o).await + } + fn delete_stream( + &self, + p: BoxStream<'static, ObjectResult>, + ) -> BoxStream<'static, ObjectResult> { + self.inner.delete_stream(p) + } + fn list(&self, p: Option<&Path>) -> BoxStream<'static, ObjectResult> { + self.inner.list(p) + } + async fn list_with_delimiter(&self, p: Option<&Path>) -> ObjectResult { + self.inner.list_with_delimiter(p).await + } + async fn copy_opts(&self, a: &Path, b: &Path, o: CopyOptions) -> ObjectResult<()> { + self.inner.copy_opts(a, b, o).await + } + } + + #[tokio::test] + async fn a_cached_range_read_reaches_the_backend_no_more() { + let root = tempfile::tempdir().unwrap(); + let memory = Arc::new(InMemory::new()); + let path = Path::from("dataset/data.lance"); + memory + .put(&path, Bytes::from_static(b"0123456789").into()) + .await + .unwrap(); + let gets = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let cached = CachedObjectStore::new( + Arc::new(Counted { + inner: memory, + gets: Arc::clone(&gets), + }), + CacheConfig::new(root.path().into(), 1024, 4), + "s3://test-bucket".into(), + ); + + assert_eq!( + cached.get_range(&path, 3..8).await.unwrap(), + Bytes::from_static(b"34567") + ); + let after_miss = gets.load(std::sync::atomic::Ordering::Relaxed); + assert!(after_miss > 0, "the first read must fill the cache"); + + // Re-reading served blocks used to re-HEAD the object, so a hit still + // cost a round trip and the cache bought no latency. + for _ in 0..5 { + assert_eq!( + cached.get_range(&path, 3..8).await.unwrap(), + Bytes::from_static(b"34567") + ); + } + assert_eq!( + gets.load(std::sync::atomic::Ordering::Relaxed), + after_miss, + "a fully cached range must not reach the backend" + ); + } +} diff --git a/crates/persisting-pchronicle/src/store/blockcache/block.rs b/crates/persisting-pchronicle/src/store/blockcache/block.rs new file mode 100644 index 000000000..05f68c5bb --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/block.rs @@ -0,0 +1,217 @@ +use super::config::CacheConfig; +use bytes::Bytes; +use std::{ + collections::HashMap, + future::Future, + path::{Path, PathBuf}, + sync::{ + Arc, + atomic::{AtomicU64, Ordering}, + }, +}; + +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +pub struct CacheStats { + pub hits: u64, + pub misses: u64, + pub evictions: u64, +} + +#[derive(Debug, Default)] +struct Counters { + hits: AtomicU64, + misses: AtomicU64, + evictions: AtomicU64, +} + +#[derive(Debug, Clone)] +pub struct BlockCache { + config: CacheConfig, + counters: Arc, + flights: Arc>>>>, + last_trim_ms: Arc, +} + +impl BlockCache { + pub fn new(config: CacheConfig) -> Self { + Self { + config, + counters: Arc::new(Counters::default()), + flights: Arc::new(tokio::sync::Mutex::new(HashMap::new())), + last_trim_ms: Arc::new(AtomicU64::new(0)), + } + } + pub fn stats(&self) -> CacheStats { + CacheStats { + hits: self.counters.hits.load(Ordering::Relaxed), + misses: self.counters.misses.load(Ordering::Relaxed), + evictions: self.counters.evictions.load(Ordering::Relaxed), + } + } + pub fn block_path(&self, key: &str, block: u64) -> PathBuf { + let hash = blake3::hash(format!("{key}:{block}").as_bytes()) + .to_hex() + .to_string(); + // Keep directory fan-out bounded; a single flat cache directory becomes + // expensive once a service has browsed many datasets and versions. + self.config + .root + .join(&hash[..2]) + .join(&hash[2..4]) + .join(hash) + } + pub async fn get_or_fetch( + &self, + path: &Path, + expected: usize, + fetch: F, + ) -> object_store::Result + where + F: Future>, + { + // Reclaim stale files after a restart as well as after new downloads. + self.schedule_trim(); + match tokio::fs::read(path).await { + Ok(bytes) if bytes.len() == expected => { + self.counters.hits.fetch_add(1, Ordering::Relaxed); + return Ok(Bytes::from(bytes)); + } + _ => { + self.counters.misses.fetch_add(1, Ordering::Relaxed); + } + }; + let flight = { + let mut flights = self.flights.lock().await; + flights + .entry(path.to_path_buf()) + .or_insert_with(|| Arc::new(tokio::sync::Mutex::new(()))) + .clone() + }; + let flight_guard = flight.lock().await; + let result = async { + if let Ok(bytes) = tokio::fs::read(path).await + && bytes.len() == expected + { + self.counters.hits.fetch_add(1, Ordering::Relaxed); + return Ok(Bytes::from(bytes)); + } + let bytes = fetch.await?; + if bytes.len() != expected { + return Err(object_store::Error::Generic { + store: "pchronicle-cache", + source: format!("block length {}, expected {expected}", bytes.len()).into(), + }); + } + if tokio::fs::create_dir_all(path.parent().unwrap_or(&self.config.root)) + .await + .is_ok() + { + let tmp = path.with_extension(format!("{}.tmp", std::process::id())); + if tokio::fs::write(&tmp, &bytes).await.is_ok() { + let _ = tokio::fs::rename(&tmp, path).await; + self.schedule_trim(); + } + } + Ok(bytes) + } + .await; + drop(flight_guard); + let mut flights = self.flights.lock().await; + if Arc::strong_count(&flight) == 2 + && flights + .get(path) + .is_some_and(|current| Arc::ptr_eq(current, &flight)) + { + flights.remove(path); + } + result + } + fn schedule_trim(&self) { + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |duration| duration.as_millis() as u64); + let last = self.last_trim_ms.load(Ordering::Relaxed); + if now.saturating_sub(last) < 5_000 + || self + .last_trim_ms + .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed) + .is_err() + { + return; + } + let cache = self.clone(); + tokio::spawn(async move { cache.trim().await }); + } + pub async fn trim(&self) { + let mut files = Vec::new(); + let mut total = 0; + let mut dirs = vec![self.config.root.clone()]; + while let Some(dir) = dirs.pop() { + let Ok(mut entries) = tokio::fs::read_dir(dir).await else { + continue; + }; + while let Ok(Some(entry)) = entries.next_entry().await { + let path = entry.path(); + if let Ok(meta) = entry.metadata().await { + if meta.is_dir() { + dirs.push(path); + } else if meta.is_file() && path.extension().is_none_or(|ext| ext != "tmp") { + total += meta.len(); + files.push((meta.modified().ok(), meta.len(), path)); + } + } + } + } + files.sort_by_key(|(mtime, _, _)| *mtime); + for (_, size, path) in files { + if total <= self.config.capacity_bytes { + break; + } + if tokio::fs::remove_file(path).await.is_ok() { + total -= size; + self.counters.evictions.fetch_add(1, Ordering::Relaxed); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[tokio::test] + async fn hit_miss_and_corrupt_are_observable() { + let dir = tempfile::tempdir().unwrap(); + let cache = BlockCache::new(CacheConfig::new(dir.path().into(), 100, 4)); + let p = dir.path().join("x"); + assert_eq!( + cache + .get_or_fetch(&p, 4, async { Ok(Bytes::from_static(b"abcd")) }) + .await + .unwrap(), + Bytes::from_static(b"abcd") + ); + assert_eq!( + cache + .get_or_fetch(&p, 4, async { panic!("cache miss") }) + .await + .unwrap(), + Bytes::from_static(b"abcd") + ); + tokio::fs::write(&p, b"bad").await.unwrap(); + assert_eq!( + cache + .get_or_fetch(&p, 4, async { Ok(Bytes::from_static(b"efgh")) }) + .await + .unwrap(), + Bytes::from_static(b"efgh") + ); + assert_eq!( + cache.stats(), + CacheStats { + hits: 1, + misses: 2, + evictions: 0 + } + ); + } +} diff --git a/crates/persisting-pchronicle/src/store/blockcache/config.rs b/crates/persisting-pchronicle/src/store/blockcache/config.rs new file mode 100644 index 000000000..6e7ac3c01 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/config.rs @@ -0,0 +1,51 @@ +use std::path::PathBuf; + +pub const DEFAULT_CAPACITY_BYTES: u64 = 512 * 1024 * 1024; +pub const SERVE_CAPACITY_BYTES: u64 = 8 * DEFAULT_CAPACITY_BYTES; +pub const DEFAULT_BLOCK_SIZE_BYTES: u64 = 8 * 1024 * 1024; + +#[derive(Debug, Clone)] +pub struct CacheConfig { + pub root: PathBuf, + pub capacity_bytes: u64, + pub block_size_bytes: u64, +} + +impl CacheConfig { + pub fn new(root: PathBuf, capacity_bytes: u64, block_size_bytes: u64) -> Self { + assert!(capacity_bytes > 0 && block_size_bytes > 0); + Self { + root, + capacity_bytes, + block_size_bytes, + } + } + + pub fn from_env() -> Self { + let capacity = std::env::var("PCHRONICLE_LANCE_CACHE_CAPACITY_BYTES") + .ok() + .and_then(|v| v.parse().ok()) + .filter(|v| *v > 0) + .unwrap_or(DEFAULT_CAPACITY_BYTES); + Self::new(default_cache_dir(), capacity, DEFAULT_BLOCK_SIZE_BYTES) + } +} + +pub fn default_cache_dir() -> PathBuf { + std::env::var("PCHRONICLE_LANCE_CACHE_DIR") + .map(PathBuf::from) + .unwrap_or_else(|_| { + dirs::cache_dir() + .unwrap_or_else(|| PathBuf::from(".")) + .join("pchronicle") + .join("blocks") + }) +} + +pub const fn capacity_for_serve(is_serve: bool) -> u64 { + if is_serve { + SERVE_CAPACITY_BYTES + } else { + DEFAULT_CAPACITY_BYTES + } +} diff --git a/crates/persisting-pchronicle/src/store/blockcache/mod.rs b/crates/persisting-pchronicle/src/store/blockcache/mod.rs new file mode 100644 index 000000000..8c78722d3 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/mod.rs @@ -0,0 +1,14 @@ +mod adapter; +mod block; +mod config; + +pub use adapter::{CachedObjectStore, LanceCacheWrapper, lance_store_params}; +pub use block::{BlockCache, CacheStats}; +pub use config::{ + CacheConfig, DEFAULT_BLOCK_SIZE_BYTES, DEFAULT_CAPACITY_BYTES, SERVE_CAPACITY_BYTES, + capacity_for_serve, default_cache_dir, +}; + +pub fn configured_capacity_bytes() -> u64 { + CacheConfig::from_env().capacity_bytes +} diff --git a/crates/persisting-pchronicle/src/store/catalog/discovery.rs b/crates/persisting-pchronicle/src/store/catalog/discovery.rs index 8ce877df1..5416cc999 100644 --- a/crates/persisting-pchronicle/src/store/catalog/discovery.rs +++ b/crates/persisting-pchronicle/src/store/catalog/discovery.rs @@ -603,10 +603,7 @@ pub(super) async fn discover_candidate_at( "invalid source scope" ); let Some(root) = local_mount_path(&mount.uri) else { - // A missing remote manifest is not proof that a prefix or JSON object - // is absent. Preserve full membership and projection binding here until - // object-store discovery supports a bounded ancestor-aware traversal. - return discover_candidates(mount, options).await; + return discover_object_candidate_at(&mount.uri, file, options).await; }; let root_metadata = fs::metadata(&root).context("inspect scoped Dataset root")?; if file == "." || root_metadata.is_file() { @@ -931,7 +928,7 @@ async fn is_compact_jsonl_directory(path: &Path) -> Result { if let Some(manifest) = try_load_manifest(path) { return Ok(manifest.is_compact_jsonl_leaf()); } - let dataset = match lance::Dataset::open(path.to_string_lossy().as_ref()).await { + let dataset = match crate::storage::open_lance_dataset(path.to_string_lossy().as_ref()).await { Ok(dataset) => dataset, Err(_) => return Ok(false), }; @@ -961,6 +958,65 @@ async fn discover_object_candidates( } } +// Follow only the requested ancestry. Opaque leaves and canonical event bundles +// keep their namespace semantics; unrelated siblings must never gate an exact read. +async fn discover_object_candidate_at( + uri: &str, + file: &str, + mut options: LocalQueryManifestOptions, +) -> Result> { + if file == "." { + return discover_object_candidates(uri, options).await; + } + let store = OpendalStore::from_uri(uri).await?; + let mut budget = DiscoveryBudget::new(options); + let mut current = String::new(); + let mut parts = file.split('/'); + loop { + budget.observe_entry()?; + if current == file + && is_json_candidate(Path::new(file)) + && let Some(entry) = store.stat_file(file).await? + { + budget.observe_source()?; + return Ok(vec![Candidate::RemoteFile { + file: file.into(), + store: store.clone(), + meta: RemoteObjectMeta::from(entry), + }]); + } + match probe_object_prefix(&store, uri, ¤t, root_source_path(¤t)).await? { + Some(ObjectProbe::Source(candidate)) => { + let sidecar = object_storyline_sidecar(&store, uri, &candidate).await?; + let mut candidates = vec![candidate]; + candidates.extend(sidecar); + // Both sides are needed for canonical projection binding before + // discover_impl filters to the requested source namespace. + for _ in &candidates { + budget.observe_source()?; + } + if !candidates.iter().any(|candidate| { + let source = candidate.source_stub().file; + source == file || source.starts_with(&format!("{file}/")) + }) { + return Ok(Vec::new()); + } + return Ok(candidates); + } + None if current.ends_with(".lance") => return Ok(Vec::new()), + Some(ObjectProbe::Branch) | None => {} + } + let Some(part) = parts.next() else { + options.max_entries = options.max_entries.saturating_sub(budget.entries); + return collect_object_virtual(&store, uri, file, options).await; + }; + if !current.is_empty() { + current.push('/'); + } + current.push_str(part); + } +} + enum ObjectProbe { Source(Candidate), Branch, diff --git a/crates/persisting-pchronicle/src/store/catalog/identity.rs b/crates/persisting-pchronicle/src/store/catalog/identity.rs index 1a5117342..a4f2466c9 100644 --- a/crates/persisting-pchronicle/src/store/catalog/identity.rs +++ b/crates/persisting-pchronicle/src/store/catalog/identity.rs @@ -2,6 +2,7 @@ use anyhow::Result; use serde::Serialize; use crate::DocumentFormat; +use crate::store::opendal_store::StoreConfig; use super::DEFAULT_DATASET_NAME; @@ -48,6 +49,8 @@ pub struct DatasetMount { pub namespace: NamespacePath, #[serde(skip)] pub(super) format_hint: Option, + #[serde(skip)] + pub(crate) backend: Option, } impl DatasetMount { @@ -73,9 +76,19 @@ impl DatasetMount { uri, namespace, format_hint: None, + backend: None, }) } + pub fn with_backend(mut self, backend: StoreConfig) -> Self { + self.backend = Some(backend); + self + } + + pub fn backend(&self) -> Option<&StoreConfig> { + self.backend.as_ref() + } + pub fn default(uri: impl Into) -> Result { Self::new(DEFAULT_DATASET_NAME, uri) } diff --git a/crates/persisting-pchronicle/src/store/catalog/location.rs b/crates/persisting-pchronicle/src/store/catalog/location.rs index e0e929ff7..d93c6a952 100644 --- a/crates/persisting-pchronicle/src/store/catalog/location.rs +++ b/crates/persisting-pchronicle/src/store/catalog/location.rs @@ -9,7 +9,7 @@ use anyhow::{Context, Result, anyhow}; use futures::{StreamExt, TryStreamExt}; use url::Url; -use crate::store::opendal_store::Store as OpendalStore; +use crate::store::opendal_store::{Store as OpendalStore, StoreConfig}; /// One discovery event while walking importable JSON objects. #[derive(Debug, Clone)] @@ -74,10 +74,15 @@ pub struct DatasetLocation { uri: String, kind: DatasetLocationKind, local_path: Option, + backend: Option, } impl DatasetLocation { pub fn parse(input: &str) -> Result { + Self::parse_with_backend(input, None) + } + + pub fn parse_with_backend(input: &str, backend: Option) -> Result { let input = input.trim(); anyhow::ensure!(!input.is_empty(), "Dataset URI must not be empty"); if !input.contains("://") { @@ -121,6 +126,7 @@ impl DatasetLocation { uri: trim_trailing_slashes(input), kind: DatasetLocationKind::ObjectStore, local_path: None, + backend, }) } "file" => { @@ -135,6 +141,7 @@ impl DatasetLocation { uri: trim_trailing_slashes(input), kind: DatasetLocationKind::Local, local_path: Some(path), + backend: None, }) } "local" => { @@ -146,6 +153,7 @@ impl DatasetLocation { uri: trim_trailing_slashes(input), kind: DatasetLocationKind::Local, local_path: Some(PathBuf::from(url.path())), + backend: None, }) } other => Err(anyhow!("unsupported Dataset URI scheme '{other}'")), @@ -157,6 +165,7 @@ impl DatasetLocation { uri, kind: DatasetLocationKind::Local, local_path: Some(path), + backend: None, } } @@ -217,7 +226,9 @@ impl DatasetLocation { if let Some(path) = &self.local_path { return Ok(path.exists()); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; store.exists().await } @@ -240,7 +251,9 @@ impl DatasetLocation { } return put_local_bytes(&path, bytes, true); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; store .write_overwrite(relative, bytes.to_vec()) .await @@ -258,7 +271,9 @@ impl DatasetLocation { let path = root.join(relative); return std::fs::read(&path).with_context(|| format!("read {}", path.display())); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let Some((bytes, _)) = store.read(relative).await? else { return Err(anyhow!("object not found: {relative} under {}", self.uri)); }; @@ -305,7 +320,9 @@ impl DatasetLocation { return Ok(None); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let join = |name: &str| { if relative.is_empty() { name.to_string() @@ -353,12 +370,35 @@ impl DatasetLocation { /// directories, leaf Datasets (with sidecar preview when present), and /// JSON / JSONL / NDJSON files. pub async fn list(&self, relative: &str) -> Result> { + self.list_impl(relative, true, true).await + } + + /// Browse one remote level without probing every child. A child is shown + /// as a directory until its own observation identifies it as a Dataset. + pub(crate) async fn list_for_browse(&self, relative: &str) -> Result> { + self.list_impl(relative, true, !self.is_object_store()) + .await + } + + /// List names for interactive browsing. Remote dataset identity and counts + /// come from the local manifest cache, never HEAD/GET probes on this path. + pub async fn list_directory(&self, relative: &str) -> Result> { + let local = !self.is_object_store(); + self.list_impl(relative, local, local).await + } + + async fn list_impl( + &self, + relative: &str, + probe_self: bool, + probe_children: bool, + ) -> Result> { let relative = relative.trim().trim_matches('/'); anyhow::ensure!( !relative.split('/').any(|part| part == ".."), "relative object path must not contain '..'" ); - if let Some(kind) = self.probe_nav_dataset_kind(relative).await? { + if probe_self && let Some(kind) = self.probe_nav_dataset_kind(relative).await? { let preview = self.list_dataset_preview(relative, kind).await?; let name = if relative.is_empty() { ".".to_string() @@ -379,7 +419,7 @@ impl DatasetLocation { }]); } - let nav = self.list_nav_children(relative).await?; + let nav = self.list_nav_children(relative, probe_children).await?; let mut out = Vec::with_capacity(nav.len()); for entry in nav { let path = if relative.is_empty() { @@ -449,7 +489,9 @@ impl DatasetLocation { } return Ok(preview); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let key = if relative.is_empty() { crate::store::CHRONICLE_MANIFEST_FILE.to_string() } else { @@ -491,11 +533,15 @@ impl DatasetLocation { if self.probe_nav_dataset_kind(relative).await?.is_some() { return Ok(Vec::new()); } - self.list_nav_children(relative).await + self.list_nav_children(relative, true).await } // Caller has validated the path and established that it is not a Dataset leaf. - async fn list_nav_children(&self, relative: &str) -> Result> { + async fn list_nav_children( + &self, + relative: &str, + probe_children: bool, + ) -> Result> { if let Some(root) = &self.local_path { let dir = if relative.is_empty() { root.clone() @@ -555,7 +601,9 @@ impl DatasetLocation { return Ok(out); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let prefix = if relative.is_empty() { String::new() } else { @@ -599,7 +647,11 @@ impl DatasetLocation { } else { format!("{relative}/{name}") }; - let kind = self.probe_nav_dataset_kind(&child_rel).await?; + let kind = if probe_children { + self.probe_nav_dataset_kind(&child_rel).await? + } else { + None + }; Ok::<_, anyhow::Error>(ShallowNavEntry { name, is_dir: kind.is_none(), @@ -703,7 +755,9 @@ impl DatasetLocation { return Ok(()); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let mut pending = vec![String::new()]; let mut found = 0usize; while let Some(prefix) = pending.pop() { @@ -819,7 +873,9 @@ impl DatasetLocation { if let Some(path) = &self.local_path { return put_local_bytes(path, bytes, overwrite); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; // DatasetLocation represents a prefix; use a stable marker inside it. let path = ".dataset-marker"; if overwrite { @@ -861,7 +917,9 @@ impl DatasetLocation { !url.path().trim_matches('/').is_empty(), "refusing to drop an entire object-store bucket; name a Dataset prefix" ); - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let entries = store .list("") .await diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index 0c1734365..86493d910 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -10,23 +10,42 @@ use std::time::Duration; use anyhow::Result; use serde::{Deserialize, Serialize}; -use tokio::sync::Mutex; use tokio::sync::RwLock; +const MAX_REFRESH_DIRECTORIES: usize = 10_000; +/// Observations per disk commit during a mount walk. Bounded so a long walk +/// still publishes progress, and so a crash forfeits at most this many reads. +const PERSIST_BATCH_DIRECTORIES: usize = 256; +/// How long a browse observation may wait for neighbours to join its commit. +const PERSIST_MAX_DELAY: Duration = Duration::from_secs(5); +const REFRESH_MOUNT_DEADLINE: Duration = Duration::from_secs(120); +const REFRESH_DIRECTORY_TIMEOUT: Duration = Duration::from_secs(20); + use crate::store::{DatasetLocation, PathListEntry, PersistentCache}; #[derive(Debug, Clone, Serialize, Deserialize)] pub struct ManifestListing { + #[serde(default)] + pub partial: bool, pub entries: Vec, pub observed_at: i64, } #[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct LocationSummary { + #[serde(default)] + pub partial: bool, pub datasets: u64, pub trajectories: u64, } +/// Outcome of one bounded mount walk. A partial walk is usable, but not complete. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct ManifestRefreshReport { + pub refreshed_directories: usize, + pub partial: bool, +} + #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ManifestReadMode { Cached, @@ -34,13 +53,27 @@ pub enum ManifestReadMode { Fresh, } +/// One directory observation that is live in memory but not yet on disk. +struct ObservedDirectory { + key: String, + listing: ManifestListing, + removed: Vec, +} + +/// Observations published in memory and still owed to disk. +#[derive(Default)] +struct PendingWrites { + observed: Vec, + since: Option, +} + /// Persistent manifest cache. Keys are caller-owned stable identities, e.g. /// `mount-uri\0relative-prefix`; this keeps the cache independent of UI types. #[derive(Clone)] pub struct ManifestCache { disk: Arc>, values: Arc>>, - refresh_gate: Arc>, + pending: Arc>, } impl ManifestCache { @@ -56,7 +89,7 @@ impl ManifestCache { Self { disk, values, - refresh_gate: Arc::new(Mutex::new(())), + pending: Arc::new(tokio::sync::Mutex::new(PendingWrites::default())), } } @@ -91,21 +124,153 @@ impl ManifestCache { location: &DatasetLocation, prefix: &str, ) -> Result { - let _guard = self.refresh_gate.lock().await; - let key = key.into(); + // A foreground miss is rare and its caller is waiting on the answer, so + // it writes through rather than joining the browse walk's batch. + let observed = self.observe(key.into(), location, prefix, false).await?; + let listing = observed.listing.clone(); + self.persist(std::slice::from_ref(&observed), prefix).await; + Ok(listing) + } + + /// Observe remote child names now; the browse worker resolves each child's + /// type and statistics when visiting that prefix, without delaying its parent. + /// + /// The observation is live in memory when this returns, but its disk write + /// rides with its neighbours: a browse walk visits thousands of prefixes, + /// and a Lance commit per prefix costs more than the listing it records. + pub async fn refresh_for_browse( + &self, + key: impl Into, + location: &DatasetLocation, + prefix: &str, + ) -> Result { + let observed = self.observe(key.into(), location, prefix, true).await?; + let listing = observed.listing.clone(); + self.enqueue(observed).await; + Ok(listing) + } + + /// Commit every observation that is only in memory. + /// + /// A browse walk publishes as it goes, so callers that need the disk copy + /// to agree with memory — a shutdown, a test, a read-after-write — must ask. + pub async fn flush(&self) { + let batch = std::mem::take(&mut *self.pending.lock().await); + self.persist(&batch.observed, "").await; + } + + async fn enqueue(&self, observed: ObservedDirectory) { + let due = { + let mut pending = self.pending.lock().await; + pending.since.get_or_insert_with(tokio::time::Instant::now); + pending.observed.push(observed); + let elapsed = pending + .since + .is_some_and(|since| since.elapsed() >= PERSIST_MAX_DELAY); + if pending.observed.len() >= PERSIST_BATCH_DIRECTORIES || elapsed { + Some(std::mem::take(&mut *pending)) + } else { + None + } + }; + if let Some(batch) = due { + self.persist(&batch.observed, "").await; + } + } + + /// Read one level and publish it in memory, returning what the disk cache + /// still owes. Split from persistence so a mount walk can commit its + /// observations together instead of once per directory. + async fn observe( + &self, + key: String, + location: &DatasetLocation, + prefix: &str, + browse: bool, + ) -> Result { + let refresh_gate = + crate::store::root_write_lock::for_root(&serde_json::to_string(&(self.path(), &key))?); + let _guard = refresh_gate.lock().await; let listing = ManifestListing { - entries: location.list(prefix).await?, + partial: false, + entries: if browse { + location.list_for_browse(prefix).await? + } else { + location.list(prefix).await? + }, observed_at: chrono::Utc::now().timestamp(), }; - self.values - .write() + let removed = { + let mut values = self.values.write().await; + // Only a successful observation can retire vanished descendants. + // Otherwise their old manifests would keep inflating UI counts. + let removed = values + .keys() + .filter(|other| { + other + .strip_prefix(&key) + .is_some_and(|suffix| suffix.starts_with('\0') || suffix.starts_with('/')) + && !listing.entries.iter().any(|entry| { + let relative = entry + .path + .strip_prefix(&format!("{prefix}/")) + .unwrap_or(&entry.path); + let child_key = format!( + "{key}{}{relative}", + if prefix.is_empty() { "\0" } else { "/" } + ); + (entry.kind == crate::store::PathListKind::Directory + && (other.as_str() == child_key + || other.starts_with(&format!("{child_key}/")))) + || (entry.kind == crate::store::PathListKind::Dataset + && other.as_str() == child_key) + }) + }) + .cloned() + .collect::>(); + for key in &removed { + values.remove(key); + } + values.insert(key.clone(), listing.clone()); + removed + }; + // `key` contains NUL separators and is an internal cache identity; + // logging it makes journald truncate the record at the dataset name. + tracing::debug!(target: "pchronicle.serve", prefix, entries = listing.entries.len(), browse, "manifest cache updated"); + Ok(ObservedDirectory { + key, + listing, + removed, + }) + } + + /// Commit observations to disk. The cache is rebuildable, so a failure or + /// a crash mid-walk only costs the unwritten observations. + async fn persist(&self, observed: &[ObservedDirectory], prefix: &str) { + let payloads = observed + .iter() + .map(|item| serde_json::to_value(&item.listing).map(|value| (item.key.clone(), value))) + .collect::, _>>(); + let payloads = match payloads { + Ok(payloads) => payloads, + Err(error) => { + tracing::warn!(target: "pchronicle.serve", prefix, %error, + "manifest cache encoding failed; using memory"); + return; + } + }; + let removed = observed + .iter() + .flat_map(|item| item.removed.iter().cloned()) + .collect::>(); + if let Err(error) = self + .disk + .upsert_many(payloads.iter().map(|(key, value)| (key, value)), &removed) .await - .insert(key.clone(), listing.clone()); - self.disk - .upsert(&key, &serde_json::to_value(&listing)?, &[]) - .await?; - tracing::info!(target: "pchronicle.serve", %key, prefix, entries = listing.entries.len(), "manifest cache updated"); - Ok(listing) + { + tracing::warn!(target: "pchronicle.serve", prefix, %error, + "manifest cache persistence failed; using memory"); + } } pub fn writable(&self) -> bool { @@ -163,6 +328,7 @@ impl ManifestCache { }) }) .fold(LocationSummary::default(), |mut total, (_, listing)| { + total.partial |= listing.partial; for entry in &listing.entries { let is_dataset = matches!(entry.kind, crate::store::PathListKind::Dataset) || (matches!(entry.kind, crate::store::PathListKind::File) @@ -178,42 +344,143 @@ impl ManifestCache { }) } - /// Breadth-first refresh of a mount. Only one refresh runs at a time; - /// shallow paths are published before deeper paths. - pub async fn refresh_mount(&self, key_prefix: &str, location: &DatasetLocation) -> Result<()> { - let _guard = self.refresh_gate.lock().await; - tracing::info!(target: "pchronicle.serve", %key_prefix, "manifest cache refresh started"); + /// Whether every directory below a mount-relative prefix has an observed + /// manifest listing. Uses the same keys as refresh_mount; no remote I/O. + pub async fn is_complete_under(&self, mount_key: &str, prefix: &str) -> bool { + let values = self.values.read().await; + let key = |path: &str| { + if path.is_empty() { + mount_key.to_owned() + } else { + format!("{mount_key}\0{path}") + } + }; + if !values.contains_key(&key(prefix)) { + return false; + } + let descendant_key = format!("{mount_key}\0"); + values + .iter() + .filter(|(key, _)| { + let path = if key.as_str() == mount_key { + Some("") + } else { + key.strip_prefix(&descendant_key) + }; + path.is_some_and(|path| { + prefix.is_empty() || path == prefix || path.starts_with(&format!("{prefix}/")) + }) + }) + .all(|(_, listing)| { + !listing.partial + && listing + .entries + .iter() + .filter(|entry| entry.kind == crate::store::PathListKind::Directory) + .all(|entry| values.contains_key(&key(&entry.path))) + }) + } + + /// Breadth-first refresh of a mount. Refreshes serialize per observation + /// key, allowing unrelated foreground directories to load concurrently. + pub async fn refresh_mount( + &self, + key_prefix: &str, + location: &DatasetLocation, + ) -> Result { + self.refresh_mount_bounded( + key_prefix, + location, + MAX_REFRESH_DIRECTORIES, + REFRESH_MOUNT_DEADLINE, + ) + .await + } + + async fn refresh_mount_bounded( + &self, + key_prefix: &str, + location: &DatasetLocation, + max_directories: usize, + budget: Duration, + ) -> Result { let mut queue = VecDeque::from([String::new()]); - let mut refreshed = 0usize; + let mut seen = HashSet::from([String::new()]); + let mut report = ManifestRefreshReport::default(); + let mut pending = Vec::::new(); + let deadline = tokio::time::Instant::now() + budget; while let Some(prefix) = queue.pop_front() { - let listing = ManifestListing { - entries: location.list(&prefix).await?, - observed_at: chrono::Utc::now().timestamp(), - }; + if report.refreshed_directories >= max_directories + || tokio::time::Instant::now() >= deadline + { + report.partial = true; + break; + } let key = if prefix.is_empty() { key_prefix.to_owned() } else { format!("{key_prefix}\0{prefix}") }; - self.values - .write() - .await - .insert(key.clone(), listing.clone()); - self.disk - .upsert(&key, &serde_json::to_value(&listing)?, &[]) - .await?; - refreshed += 1; + let directory_deadline = + deadline.min(tokio::time::Instant::now() + REFRESH_DIRECTORY_TIMEOUT); + let observed = match tokio::time::timeout_at( + directory_deadline, + self.observe(key, location, &prefix, false), + ) + .await + { + Ok(result) => result?, + Err(_) => { + report.partial = true; + break; + } + }; + let listing = observed.listing.clone(); + pending.push(observed); + if pending.len() >= PERSIST_BATCH_DIRECTORIES { + self.persist(&pending, &prefix).await; + pending.clear(); + } + report.refreshed_directories += 1; for child in listing .entries .iter() .filter(|e| matches!(e.kind, crate::store::PathListKind::Directory)) { + if seen.contains(&child.path) { + continue; + } + if seen.len() >= max_directories { + report.partial = true; + continue; + } + seen.insert(child.path.clone()); queue.push_back(child.path.clone()); } tokio::task::yield_now().await; } - tracing::info!(target: "pchronicle.serve", %key_prefix, refreshed, "manifest cache refresh finished"); - Ok(()) + self.persist(&pending, key_prefix).await; + // Publish completeness alongside the cached root, including across restarts. + // Do not hold the values lock over persistence. + let root = { + let mut values = self.values.write().await; + values.get_mut(key_prefix).map(|root| { + root.partial = report.partial; + root.clone() + }) + }; + if let Some(root) = root + && let Err(error) = self + .disk + .upsert(&key_prefix.to_owned(), &serde_json::to_value(root)?, &[]) + .await + { + tracing::warn!(target: "pchronicle.serve", %error, "mount completeness persistence failed"); + } + tracing::info!(target: "pchronicle.serve", %key_prefix, + refreshed = report.refreshed_directories, partial = report.partial, + "manifest cache refresh finished"); + Ok(report) } pub fn spawn_periodic_refresh( @@ -240,6 +507,208 @@ mod tests { use super::*; use crate::store::{PathListEntry, PathListKind}; + #[tokio::test] + async fn successful_parent_observation_retires_deleted_manifest_counts() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + let leaf = source.join("nested/leaf"); + std::fs::create_dir_all(&leaf).unwrap(); + crate::storage::write_compact_jsonl_manifest(&leaf, 1, 42).unwrap(); + let path = temp.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + cache.refresh_mount("mount", &location).await.unwrap(); + assert_eq!(cache.summary("mount").await.trajectories, 42); + assert!(cache.is_complete_under("mount", "").await); + std::fs::remove_dir_all(source.join("nested")).unwrap(); + cache + .refresh_for_browse("mount", &location, "") + .await + .unwrap(); + assert_eq!(cache.summary("mount").await.datasets, 0); + assert!(cache.get("mount\0nested").await.is_none()); + // A browse observation is published in memory first; the disk copy only + // has to agree once the pending writes are asked for. + cache.flush().await; + drop(cache); + assert_eq!( + ManifestCache::open(path) + .await + .summary("mount") + .await + .trajectories, + 0 + ); + } + + #[tokio::test] + async fn mount_walk_commits_its_observations_together() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + let directories = 40; + for index in 0..directories { + let leaf = source.join(format!("dir-{index:03}/leaf")); + std::fs::create_dir_all(&leaf).unwrap(); + crate::storage::write_compact_jsonl_manifest(&leaf, 1, 1).unwrap(); + } + let path = temp.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + let report = cache.refresh_mount("mount", &location).await.unwrap(); + // The root and every `dir-*`; a manifest makes `leaf` a dataset entry. + assert_eq!(report.refreshed_directories, directories + 1); + drop(cache); + + // Each Lance commit re-reads the cache to find matches, so a commit per + // directory makes a mount walk quadratic in the directory count. + let versions = lance::Dataset::open(path.to_str().unwrap()) + .await + .unwrap() + .version() + .version; + assert!( + versions <= 4, + "a {directories}-directory walk took {versions} commits" + ); + assert_eq!( + ManifestCache::open(path).await.summary_under("mount").await, + LocationSummary { + partial: false, + datasets: directories as u64, + trajectories: directories as u64, + } + ); + } + + #[tokio::test] + async fn browse_observations_share_one_commit_and_stay_readable() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + let prefixes = 40; + for index in 0..prefixes { + let leaf = source.join(format!("dir-{index:03}/leaf")); + std::fs::create_dir_all(&leaf).unwrap(); + crate::storage::write_compact_jsonl_manifest(&leaf, 1, 1).unwrap(); + } + let path = temp.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + // The browse coordinator visits one prefix per call, exactly like this. + cache + .refresh_for_browse("mount", &location, "") + .await + .unwrap(); + for index in 0..prefixes { + let prefix = format!("dir-{index:03}"); + cache + .refresh_for_browse(format!("mount\0{prefix}"), &location, &prefix) + .await + .unwrap(); + } + // Every observation must answer from memory before it reaches disk. + assert_eq!(cache.summary_under("mount").await.datasets, prefixes as u64); + assert!(cache.get("mount\0dir-017").await.is_some()); + + cache.flush().await; + drop(cache); + // Each commit builds a DataFusion context and re-reads the target, so a + // commit per visited prefix costs far more than the listings it records. + let versions = lance::Dataset::open(path.to_str().unwrap()) + .await + .unwrap() + .version() + .version; + assert!( + versions <= 3, + "{prefixes} browse observations took {versions} commits" + ); + assert_eq!( + ManifestCache::open(path).await.summary_under("mount").await, + LocationSummary { + partial: false, + datasets: prefixes as u64, + trajectories: prefixes as u64, + } + ); + } + + #[tokio::test] + async fn bounded_walk_reports_and_persists_partial_then_recovers() { + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().join("source"); + std::fs::create_dir_all(root.join("child")).unwrap(); + let location = DatasetLocation::parse(root.to_str().unwrap()).unwrap(); + let path = dir.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let partial = cache + .refresh_mount_bounded("mount", &location, 1, Duration::from_secs(20)) + .await + .unwrap(); + assert_eq!( + partial, + ManifestRefreshReport { + refreshed_directories: 1, + partial: true + } + ); + assert!(cache.summary("mount").await.partial); + drop(cache); + let cache = ManifestCache::open(path).await; + assert!(cache.summary("mount").await.partial); + let complete = cache.refresh_mount("mount", &location).await.unwrap(); + assert!(!complete.partial); + assert!(!cache.summary("mount").await.partial); + let timed_out = cache + .refresh_mount_bounded("mount", &location, 10, Duration::ZERO) + .await + .unwrap(); + assert!(timed_out.partial); + assert_eq!(timed_out.refreshed_directories, 0); + } + + #[tokio::test] + async fn refresh_survives_disk_failure_and_unrelated_refresh_lock() { + let dir = tempfile::tempdir().unwrap(); + let source = dir.path().join("source"); + std::fs::create_dir_all(source.join("nested")).unwrap(); + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + let path = dir.path().join("manifest.lance"); + let cache = ManifestCache::open(path.clone()).await; + // Inject a persistent disk failure after opening a writable cache. + std::fs::write(&path, "not a Lance directory").unwrap(); + assert!( + cache + .disk + .upsert(&"probe".into(), &serde_json::Value::Null, &[]) + .await + .is_err() + ); + let gate = crate::store::root_write_lock::for_root( + &serde_json::to_string(&(cache.path(), "blocked")).unwrap(), + ); + let _guard = gate.lock().await; + let listing = tokio::time::timeout( + Duration::from_secs(5), + cache.refresh("healthy", &location, ""), + ) + .await + .unwrap() + .unwrap(); + assert!(!listing.entries.is_empty()); + assert_eq!( + cache.get("healthy").await.unwrap().entries.len(), + listing.entries.len() + ); + cache.refresh_mount("mount", &location).await.unwrap(); + assert!(cache.get("mount").await.is_some()); + assert!( + cache + .refresh("missing", &location, "../escape") + .await + .is_err() + ); + } + #[tokio::test] async fn summary_counts_cached_manifest_entries() { let dir = tempfile::tempdir().unwrap(); @@ -247,6 +716,7 @@ mod tests { cache.values.write().await.insert( "mount".into(), ManifestListing { + partial: false, entries: vec![PathListEntry { name: "a".into(), path: "a".into(), @@ -261,6 +731,7 @@ mod tests { assert_eq!( cache.summary("mount").await, LocationSummary { + partial: false, datasets: 1, trajectories: 3 } @@ -268,6 +739,7 @@ mod tests { cache.values.write().await.insert( "mount\0nested".into(), ManifestListing { + partial: false, entries: vec![PathListEntry { name: "b".into(), path: "nested/b".into(), @@ -321,6 +793,7 @@ mod tests { assert_eq!( cache.summary_under(prefix).await, LocationSummary { + partial: false, datasets, trajectories }, @@ -345,6 +818,7 @@ mod tests { assert_eq!( cache.summary_under("rfs\0fingerprint\0nested").await, LocationSummary { + partial: false, datasets: 2, trajectories: 708 } diff --git a/crates/persisting-pchronicle/src/store/catalog/mod.rs b/crates/persisting-pchronicle/src/store/catalog/mod.rs index 61f101834..41842a3c3 100644 --- a/crates/persisting-pchronicle/src/store/catalog/mod.rs +++ b/crates/persisting-pchronicle/src/store/catalog/mod.rs @@ -14,6 +14,7 @@ mod namespace; mod provider; mod resolver; mod source; +mod status; pub use identity::{CatalogSourceRevision, DatasetMount, NamespacePath}; #[allow(unused_imports)] @@ -29,11 +30,14 @@ pub use manifest::{ load_manifest_at_uri, try_load_manifest, write_compact_jsonl_manifest, write_storyline_manifest, write_storyline_manifest_at_uri, }; -pub use manifest_cache::{LocationSummary, ManifestCache, ManifestListing, ManifestReadMode}; +pub use manifest_cache::{ + LocationSummary, ManifestCache, ManifestListing, ManifestReadMode, ManifestRefreshReport, +}; pub use namespace::{CatalogNamespace, CatalogPage, CatalogSourceDescription}; use provider::*; pub use resolver::{CachedDataset, Dataset, DatasetResolver, ResolveMode, ResolveTarget}; use source::*; +pub use status::{CatalogConsistency, CatalogState, CatalogStatus}; use discovery::{ bind_canonical_storyline_projections, discover_cached_candidates, discover_candidate_at, diff --git a/crates/persisting-pchronicle/src/store/catalog/status.rs b/crates/persisting-pchronicle/src/store/catalog/status.rs new file mode 100644 index 000000000..365ea84b8 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/catalog/status.rs @@ -0,0 +1,31 @@ +//! Shared catalog state vocabulary for UI, CLI, and query callers. + +use serde::Serialize; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum CatalogConsistency { + BestEffort, + #[serde(rename = "per_source_pinned")] + Pinned, + Exact, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum CatalogState { + Ready, + Partial, + Stale, + Refreshing, + Unavailable, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct CatalogStatus { + pub consistency: CatalogConsistency, + pub state: CatalogState, + pub generation: String, + pub observed_at: i64, + pub last_error: Option, +} diff --git a/crates/persisting-pchronicle/src/store/catalog/tests.rs b/crates/persisting-pchronicle/src/store/catalog/tests.rs index 340d0089c..754a36f10 100644 --- a/crates/persisting-pchronicle/src/store/catalog/tests.rs +++ b/crates/persisting-pchronicle/src/store/catalog/tests.rs @@ -1402,6 +1402,10 @@ async fn scoped_remote_discovery_includes_json_and_new_prefix_members() -> Resul store .write_overwrite("nested2/other.json", b"[]".to_vec()) .await?; + // A sibling that cannot be parsed must never be visited by scoped reads. + store + .write_overwrite("nested2/chronicle.manifest", b"invalid = [".to_vec()) + .await?; let discover = |file: &str| { DatasetCatalogSnapshot::discover_scoped( vec![DatasetMount::default(uri.clone()).unwrap()], @@ -1429,6 +1433,37 @@ async fn scoped_remote_discovery_includes_json_and_new_prefix_members() -> Resul Ok(()) } +#[tokio::test] +async fn scoped_remote_candidates_preserve_event_binding_and_opaque_ancestors() -> Result<()> { + let uri = format!( + "shared-memory://scoped-bundle-{}/root", + uuid::Uuid::new_v4().simple() + ); + let store = OpendalStore::from_uri(&uri).await?; + for path in [ + "agent/run/events.lance/_manifest.json", + "agent/run/storyline/CURRENT", + "opaque/CURRENT", + "opaque/interior.json", + ] { + store.write_overwrite(path, b"{}".to_vec()).await?; + } + let mount = DatasetMount::default(uri)?; + let options = CatalogSnapshotOptions::default().manifest; + for file in ["agent/run/events.lance", "agent/run/storyline"] { + let candidates = super::discovery::discover_candidate_at(&mount, file, options).await?; + let files: Vec<_> = candidates.iter().map(|c| c.source_stub().file).collect(); + assert_eq!(files, ["agent/run/events.lance", "agent/run/storyline"]); + } + let candidates = + super::discovery::discover_candidate_at(&mount, "opaque/interior.json", options).await?; + assert!( + candidates.is_empty(), + "opaque dataset contents are not separate sources" + ); + Ok(()) +} + #[tokio::test] async fn scoped_discovery_preserves_single_file_mount_identity() -> Result<()> { let temp = tempfile::tempdir()?; diff --git a/crates/persisting-pchronicle/src/store/compact_jsonl.rs b/crates/persisting-pchronicle/src/store/compact_jsonl.rs index d0ab8f109..ea0e10307 100644 --- a/crates/persisting-pchronicle/src/store/compact_jsonl.rs +++ b/crates/persisting-pchronicle/src/store/compact_jsonl.rs @@ -168,7 +168,7 @@ impl CompactJsonlStore { /// here so catalog/CLI paths cannot skip `chronicle.manifest`. pub async fn publish_manifest(root: impl AsRef) -> Result { let root = root.as_ref(); - let dataset = Dataset::open(root.to_string_lossy().as_ref()) + let dataset = crate::storage::open_lance_dataset(root.to_string_lossy().as_ref()) .await .with_context(|| { format!( @@ -188,10 +188,11 @@ impl CompactJsonlStore { /// manifests are rewritten; compatible old datasets are upgraded in place. pub async fn ensure_manifest(root: impl AsRef) -> Result> { let root = root.as_ref(); - let dataset = match Dataset::open(root.to_string_lossy().as_ref()).await { - Ok(dataset) => dataset, - Err(_) => return Ok(None), - }; + let dataset = + match crate::storage::open_lance_dataset(root.to_string_lossy().as_ref()).await { + Ok(dataset) => dataset, + Err(_) => return Ok(None), + }; if validate_dataset_schema(&dataset).is_err() { return Ok(None); } @@ -229,7 +230,7 @@ impl CompactJsonlStore { let input = input.as_ref(); let limit = limit.max(1); let manifest = Self::ensure_manifest(input).await?; - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; validate_dataset_schema(&dataset)?; let total = if let Some(count) = manifest .as_ref() @@ -283,7 +284,7 @@ impl CompactJsonlStore { pub async fn records(input: impl AsRef) -> Result> { let input = input.as_ref(); let _ = Self::ensure_manifest(input).await?; - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; validate_dataset_schema(&dataset)?; let stream = dataset.scan().scan_in_order(true).try_into_stream().await?; let mut stream = stream; @@ -320,7 +321,7 @@ impl CompactJsonlStore { pub async fn read_record(input: impl AsRef, id: &str) -> Result> { let input = input.as_ref(); let _ = Self::ensure_manifest(input).await?; - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; let (_, offload_idx) = validate_dataset_schema(&dataset)?; let stream = dataset.scan().scan_in_order(true).try_into_stream().await?; let mut stream = stream; @@ -670,7 +671,7 @@ impl CompactJsonlStore { output_root != input_root && !output_root.starts_with(&input_root), "compact JSONL export output must be outside the input dataset" ); - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; let (raw_idx, offload_idx) = validate_dataset_schema(&dataset)?; if output.exists() { fs::remove_dir_all(output) diff --git a/crates/persisting-pchronicle/src/store/events/mod.rs b/crates/persisting-pchronicle/src/store/events/mod.rs index 888b3b39b..fb388e05d 100644 --- a/crates/persisting-pchronicle/src/store/events/mod.rs +++ b/crates/persisting-pchronicle/src/store/events/mod.rs @@ -421,7 +421,7 @@ pub(crate) async fn compact_sealed_event_segment( ); let _guard = dataset_write_lock::acquire(&sealed.root_uri).await?; let segment_uri = raw_event_manifest::segment_uri(&sealed.root_uri, &sealed.segment.id); - let mut dataset = Dataset::open(&segment_uri) + let mut dataset = crate::storage::open_lance_dataset(&segment_uri) .await .with_context(|| format!("open sealed event segment {segment_uri}"))?; let metrics = compact_files( @@ -514,7 +514,7 @@ pub(super) fn validate_event_schema(dataset: &Dataset, uri: &str) -> Result<()> async fn open_visible_segment(root_uri: &str, segment: &EventSegment) -> Result { let uri = raw_event_manifest::segment_uri(root_uri, &segment.id); - let latest = Dataset::open(&uri) + let latest = crate::storage::open_lance_dataset(&uri) .await .with_context(|| format!("open event segment {uri}"))?; let dataset = if latest.version_id() == segment.version { diff --git a/crates/persisting-pchronicle/src/store/inspect.rs b/crates/persisting-pchronicle/src/store/inspect.rs index 426464508..453a9e809 100644 --- a/crates/persisting-pchronicle/src/store/inspect.rs +++ b/crates/persisting-pchronicle/src/store/inspect.rs @@ -245,7 +245,7 @@ pub async fn inspect_physical_file( }) .collect::>(); if let Ok(uri) = table_uri(snapshot, dataset, file, table).await - && let Ok(lance) = Dataset::open(&uri).await + && let Ok(lance) = crate::storage::open_lance_dataset(&uri).await { enrich_column_stats(&lance, fragment_id, &mut columns).await; } @@ -304,7 +304,7 @@ pub async fn inspect_physical_page( anyhow::ensure!(!columns.is_empty(), "physical data file has no columns"); let limit = query.limit.clamp(1, DEFAULT_PHYSICAL_PAGE_LIMIT); let uri = table_uri(snapshot, query.dataset, query.file, query.table).await?; - let lance = Dataset::open(&uri) + let lance = crate::storage::open_lance_dataset(&uri) .await .with_context(|| format!("open Lance table {uri}"))?; preview_rows(&lance, &columns, query.offset, limit).await @@ -402,7 +402,7 @@ fn storyline_table_path<'a>(paths: &'a StorylineTablePaths, table: &str) -> Opti async fn open_table_version(path: &Path, version: u64) -> Result { let uri = path.to_string_lossy(); - let dataset = Dataset::open(uri.as_ref()) + let dataset = crate::storage::open_lance_dataset(uri.as_ref()) .await .with_context(|| format!("open Lance table {uri}"))?; dataset diff --git a/crates/persisting-pchronicle/src/store/mod.rs b/crates/persisting-pchronicle/src/store/mod.rs index 8e4b293a8..6e7c77d1e 100644 --- a/crates/persisting-pchronicle/src/store/mod.rs +++ b/crates/persisting-pchronicle/src/store/mod.rs @@ -12,6 +12,8 @@ mod agenticmd_datafusion; #[cfg(feature = "lance-store")] mod attempt_registry; #[cfg(feature = "lance-store")] +pub(crate) mod blockcache; +#[cfg(feature = "lance-store")] mod cas_store; #[cfg(feature = "lance-store")] mod catalog; @@ -84,14 +86,15 @@ pub use catalog::manifest::{ }; #[cfg(feature = "lance-store")] pub use catalog::{ - CATALOG_SOURCES_TABLE, CATALOG_TRAJECTORIES_TABLE, CachedDataset, CatalogDataset, - CatalogErrorPolicy, CatalogEventProvenance, CatalogEventView, CatalogNamespace, CatalogPage, - CatalogProjectionStatus, CatalogSnapshotOptions, CatalogSourceDescription, CatalogSourceKind, - CatalogSourceRevision, CatalogSourceStatus, CatalogStorylineKey, CatalogTrajectoryBundle, - DEFAULT_DATASET_NAME, DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, - Dataset, DatasetCatalogSnapshot, DatasetMount, DatasetResolver, DiscoveredSource, - LocationSummary, ManifestCache, ManifestListing, ManifestReadMode, NamespacePath, QueryScope, - ResolveMode, ResolveTarget, + CATALOG_SOURCES_TABLE, CATALOG_TRAJECTORIES_TABLE, CachedDataset, CatalogConsistency, + CatalogDataset, CatalogErrorPolicy, CatalogEventProvenance, CatalogEventView, CatalogNamespace, + CatalogPage, CatalogProjectionStatus, CatalogSnapshotOptions, CatalogSourceDescription, + CatalogSourceKind, CatalogSourceRevision, CatalogSourceStatus, CatalogState, CatalogStatus, + CatalogStorylineKey, CatalogTrajectoryBundle, DEFAULT_DATASET_NAME, + DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, Dataset, + DatasetCatalogSnapshot, DatasetMount, DatasetResolver, DiscoveredSource, LocationSummary, + ManifestCache, ManifestListing, ManifestReadMode, ManifestRefreshReport, NamespacePath, + QueryScope, ResolveMode, ResolveTarget, }; #[cfg(feature = "lance-store")] pub use compact_jsonl::{ diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 15c1a0c84..94b53aec7 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -3,21 +3,29 @@ //! Lance opens and table writes against flaky S3-compatible gateways amplify //! timeouts when several datasets race (list `_versions/`, retries, AIMD inside //! object_store). This gate: -//! 1. caps concurrent remote Lance ops (default 1); +//! 1. caps concurrent remote ops per endpoint + bucket (default 4); //! 2. after a transient failure, forces a shared cooldown + growing delay; -//! 3. decays the delay after a streak of successes. +//! 3. decays the delay after a streak of successes; +//! 4. keeps interactive (foreground) work ahead of browse/maintenance +//! (background): background acquires yield while the same scope has +//! foreground demand, so shared S3 bandwidth is not split evenly. //! //! Local `file://` paths bypass the gate entirely. +use std::collections::HashMap; use std::sync::{Arc, Mutex, OnceLock}; use std::time::{Duration, Instant}; use tokio::sync::{OwnedSemaphorePermit, Semaphore}; -const DEFAULT_REMOTE_CONCURRENCY: usize = 1; -const MAX_REMOTE_CONCURRENCY: usize = 2; +const DEFAULT_REMOTE_CONCURRENCY: usize = 4; +const DEFAULT_BACKGROUND_CONCURRENCY: usize = 1; +const MAX_REMOTE_CONCURRENCY: usize = 8; +const MAX_RETAINED_SCOPES: usize = 1024; +const SCOPE_IDLE_TTL: Duration = Duration::from_secs(300); const MAX_DELAY_MS: u64 = 30_000; const SUCCESS_STREAK_TO_DECAY: u32 = 4; +const BACKGROUND_YIELD_POLL: Duration = Duration::from_millis(25); /// Whether the gated op is primarily reading metadata/objects or writing them. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -117,8 +125,10 @@ fn emit_throttle(event: ObjectStoreThrottleEvent) { } } -#[derive(Debug)] +#[derive(Debug, Clone)] struct AimdState { + semaphore: Arc, + last_used: Instant, /// Extra sleep applied before each remote acquire while degraded. delay_ms: u64, /// No new remote op starts until this instant. @@ -134,6 +144,8 @@ struct AimdState { impl Default for AimdState { fn default() -> Self { Self { + semaphore: Arc::new(Semaphore::new(DEFAULT_REMOTE_CONCURRENCY)), + last_used: Instant::now(), delay_ms: 0, cooldown_until: None, successes_since_backoff: 0, @@ -145,27 +157,228 @@ impl Default for AimdState { } struct Gate { - semaphore: Arc, + /// `foreground` or `background`, so a log line says which lane queued. + lane: &'static str, + concurrency: usize, + states: Mutex>, +} + +/// Admission waits shorter than this are noise next to a remote round trip. +const WAIT_LOG_THRESHOLD: Duration = Duration::from_millis(100); + +fn state_for<'a>( + states: &'a mut HashMap, + key: &str, concurrency: usize, - state: Mutex, +) -> &'a mut AimdState { + let now = Instant::now(); + if !states.contains_key(key) { + states.retain(|_, state| { + Arc::strong_count(&state.semaphore) > 1 + || state.active_waiters > 0 + || state.cooldown_until.is_some_and(|until| until > now) + || now.duration_since(state.last_used) < SCOPE_IDLE_TTL + }); + while states.len() >= MAX_RETAINED_SCOPES { + let oldest = states + .iter() + .filter(|(_, state)| { + Arc::strong_count(&state.semaphore) == 1 + && state.active_waiters == 0 + && state.cooldown_until.is_none_or(|until| until <= now) + }) + .min_by_key(|(_, state)| state.last_used) + .map(|(key, _)| key.clone()); + if let Some(oldest) = oldest { + states.remove(&oldest); + } else { + break; + } + } + } + // Live waits/cooldowns may temporarily exceed the retention limit. Evicting + // them would let backend overload bypass AIMD; reclaim after they finish. + let state = states.entry(key.to_owned()).or_insert_with(|| AimdState { + semaphore: Arc::new(Semaphore::new(concurrency)), + ..Default::default() + }); + state.last_used = now; + state +} + +tokio::task_local! { + static BACKGROUND_IO: (); +} + +/// Run manifest maintenance with independent admission and AIMD state. Call +/// inside the spawned task: Tokio task-local state is not inherited by spawn. +pub async fn with_background_object_store_io(work: F) -> F::Output { + BACKGROUND_IO.scope((), work).await +} + +/// Process-wide interactive demand per admission scope (holders + in-flight +/// acquires). Background work polls this so browse refresh does not share the +/// pipe with turns/run while a user request is active. +fn foreground_demand_map() -> &'static Mutex> { + static FOREGROUND_DEMAND: OnceLock>> = OnceLock::new(); + FOREGROUND_DEMAND.get_or_init(|| Mutex::new(HashMap::new())) +} + +fn enter_foreground_demand(key: &str) { + let Ok(mut map) = foreground_demand_map().lock() else { + return; + }; + *map.entry(key.to_owned()).or_insert(0) += 1; +} + +fn leave_foreground_demand(key: &str) { + let Ok(mut map) = foreground_demand_map().lock() else { + return; + }; + let Some(count) = map.get_mut(key) else { + return; + }; + *count = count.saturating_sub(1); + if *count == 0 { + map.remove(key); + } +} + +fn scope_foreground_demand(key: &str) -> u32 { + foreground_demand_map() + .lock() + .ok() + .and_then(|map| map.get(key).copied()) + .unwrap_or(0) +} + +/// Total interactive object-store acquires currently in flight (any scope). +pub fn foreground_object_store_demand() -> u32 { + foreground_demand_map() + .lock() + .ok() + .map(|map| map.values().copied().sum()) + .unwrap_or(0) +} + +/// Block until no interactive object-store work is admitted. Browse refresh +/// calls this before starting a background walk so user requests go first. +pub async fn wait_for_foreground_object_store_idle() { + let mut logged = false; + let started = Instant::now(); + loop { + if foreground_object_store_demand() == 0 { + if logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + yielded_ms = started.elapsed().as_millis() as u64, + "background resumed after interactive object-store idle" + ); + } + return; + } + if !logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + demand = foreground_object_store_demand(), + "background yielding to interactive object-store demand" + ); + logged = true; + } + tokio::time::sleep(BACKGROUND_YIELD_POLL).await; + } +} + +struct ForegroundDemandGuard { + key: String, +} + +impl ForegroundDemandGuard { + fn enter(key: &str) -> Self { + enter_foreground_demand(key); + Self { + key: key.to_owned(), + } + } +} + +impl Drop for ForegroundDemandGuard { + fn drop(&mut self) { + leave_foreground_demand(&self.key); + } } fn gate() -> &'static Gate { - static GATE: OnceLock = OnceLock::new(); - GATE.get_or_init(|| { - let concurrency = std::env::var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") + static FOREGROUND_GATE: OnceLock = OnceLock::new(); + static BACKGROUND_GATE: OnceLock = OnceLock::new(); + let background = BACKGROUND_IO.try_with(|_| ()).is_ok(); + let slot = if background { + &BACKGROUND_GATE + } else { + &FOREGROUND_GATE + }; + slot.get_or_init(|| { + let configured = std::env::var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") .ok() .and_then(|value| value.parse::().ok()) .unwrap_or(DEFAULT_REMOTE_CONCURRENCY) .clamp(1, MAX_REMOTE_CONCURRENCY); + // Background stays single-flight: even when interactive is idle, browse + // should not open a second S3 pipeline beside itself. + let concurrency = if background { + DEFAULT_BACKGROUND_CONCURRENCY.min(configured) + } else { + configured + }; Gate { - semaphore: Arc::new(Semaphore::new(concurrency)), + lane: if background { + "background" + } else { + "foreground" + }, concurrency, - state: Mutex::new(AimdState::default()), + states: Mutex::new(HashMap::new()), } }) } +/// Canonical admission scope, independent of object path and cache credentials. +/// Adapters capture this at construction so later environment changes cannot +/// move feedback for an existing store into another endpoint's state. +pub(crate) fn scope_key(uri: &str) -> String { + if uri + .split_once('#') + .is_some_and(|(_, fragment)| fragment.starts_with("endpoint=")) + { + return uri.to_owned(); + } + let endpoint_vars: &[&str] = if uri.starts_with("s3") { + &["AWS_ENDPOINT_URL_S3", "AWS_ENDPOINT", "AWS_ENDPOINT_URL"] + } else if uri.starts_with("az") { + &["AZURE_STORAGE_ENDPOINT"] + } else { + &["GOOGLE_STORAGE_BASE_URL"] + }; + let endpoint = endpoint_vars + .iter() + .find_map(|name| std::env::var(name).ok().filter(|v| !v.is_empty())) + .unwrap_or_default(); + scope_for_endpoint(uri, &endpoint) +} + +pub(crate) fn scope_for_endpoint(uri: &str, endpoint: &str) -> String { + let uri = uri.split('#').next().unwrap_or(uri); + let Some((scheme, rest)) = uri.split_once("://") else { + return uri.to_owned(); + }; + let bucket = rest.split('/').next().unwrap_or(rest); + let endpoint = url::Url::parse(endpoint) + .map(|url| url.to_string()) + .unwrap_or_else(|_| endpoint.to_owned()); + let identity = blake3::hash(endpoint.trim_end_matches('/').as_bytes()); + format!("{scheme}://{bucket}#endpoint={}", identity.to_hex()) +} + /// True for s3/gs/az (and similar) URIs; false for local paths / file://. pub(crate) fn is_remote_uri(uri: &str) -> bool { let Some((scheme, _)) = uri.split_once("://") else { @@ -174,12 +387,59 @@ pub(crate) fn is_remote_uri(uri: &str) -> bool { !matches!(scheme, "file" | "file+uring" | "memory" | "shared-memory") } +/// Classify errors for shared AIMD feedback. Permanent object identity, +/// authorization, and precondition failures must never increase cooldown. +pub(crate) fn is_transient_error(error: &object_store::Error) -> bool { + if matches!( + error, + object_store::Error::NotFound { .. } + | object_store::Error::InvalidPath { .. } + | object_store::Error::NotSupported { .. } + | object_store::Error::AlreadyExists { .. } + | object_store::Error::Precondition { .. } + | object_store::Error::NotModified { .. } + | object_store::Error::PermissionDenied { .. } + | object_store::Error::Unauthenticated { .. } + ) { + return false; + } + let text = error.to_string().to_ascii_lowercase(); + [ + "timeout", + "timed out", + "connection", + "broken pipe", + "temporarily", + "slowdown", + "throttl", + "503", + "429", + "reset", + ] + .iter() + .any(|needle| text.contains(needle)) +} + /// Snapshot AIMD / cooldown state for progress UI. pub fn snapshot() -> ObjectStoreGateSnapshot { let g = gate(); - let available_permits = g.semaphore.available_permits(); + let available_permits = g.concurrency; let max_permits = g.concurrency; - let Ok(state) = g.state.lock() else { + let Ok(states) = g.states.lock() else { + return ObjectStoreGateSnapshot { + kind: IoKind::Read, + delay_ms: 0, + cooldown_remaining_ms: 0, + failures: 0, + success_streak: 0, + success_streak_target: SUCCESS_STREAK_TO_DECAY, + active_waiters: 0, + available_permits, + max_permits, + }; + }; + let state = states.values().max_by_key(|state| state.failures); + let Some(state) = state else { return ObjectStoreGateSnapshot { kind: IoKind::Read, delay_ms: 0, @@ -205,7 +465,7 @@ pub fn snapshot() -> ObjectStoreGateSnapshot { success_streak: state.successes_since_backoff, success_streak_target: SUCCESS_STREAK_TO_DECAY, active_waiters: state.active_waiters, - available_permits, + available_permits: state.semaphore.available_permits(), max_permits, } } @@ -243,44 +503,158 @@ pub fn format_aimd_flow_label(snap: &ObjectStoreGateSnapshot, event: Option<&str pub(crate) struct Permit { _permit: Option, + /// Held for the full foreground acquire+hold window so background yields. + _demand: Option, } /// Acquire admission for a Lance/object-store operation on `uri`. pub(crate) async fn acquire(uri: &str, kind: IoKind) -> Permit { if !is_remote_uri(uri) { - return Permit { _permit: None }; + return Permit { + _permit: None, + _demand: None, + }; } - if let Ok(mut state) = gate().state.lock() { - state.last_kind = kind; + acquire_scoped(gate(), &scope_key(uri), kind).await +} + +async fn yield_to_foreground(key: &str, kind: IoKind) { + let mut logged = false; + let started = Instant::now(); + loop { + if scope_foreground_demand(key) == 0 { + if logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + scope = key, + kind = kind.as_str(), + yielded_ms = started.elapsed().as_millis() as u64, + "background admission resumed after interactive demand cleared" + ); + } + return; + } + if !logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + scope = key, + kind = kind.as_str(), + demand = scope_foreground_demand(key), + "background admission yielding to interactive object-store demand" + ); + logged = true; + } + tokio::time::sleep(BACKGROUND_YIELD_POLL).await; } - wait_out_degradation(kind).await; - let permit = match gate().semaphore.clone().try_acquire_owned() { - Ok(permit) => permit, - Err(_) => { - enter_wait(kind, "admit", 0); - let permit = match gate().semaphore.clone().acquire_owned().await { - Ok(permit) => permit, - Err(error) => { - leave_wait(kind); - tracing::error!(?error, "object-store I/O semaphore closed unexpectedly"); - return Permit { _permit: None }; +} + +async fn acquire_scoped(g: &Gate, key: &str, kind: IoKind) -> Permit { + let background = g.lane == "background"; + // Interactive demand covers the whole wait+hold window so browse cannot + // race into the same S3 endpoint between admit and first byte. + let demand = if background { + None + } else { + Some(ForegroundDemandGuard::enter(key)) + }; + let semaphore = { + let mut states = g + .states + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let state = state_for(&mut states, key, g.concurrency); + state.last_kind = kind; + Arc::clone(&state.semaphore) + }; + loop { + if background { + yield_to_foreground(key, kind).await; + } + wait_out_degradation(g, key, kind).await; + let permit = match semaphore.clone().try_acquire_owned() { + Ok(permit) => permit, + Err(_) => { + // A stall here is admission queueing, not the backend. Without + // a log line the two are indistinguishable in a trace, and + // every remote read looks like a slow round trip. + let queued = Instant::now(); + let _wait = WaitGuard::new(g, key, kind, "admit", 0); + let permit = match semaphore.clone().acquire_owned().await { + Ok(permit) => permit, + Err(error) => { + tracing::error!(?error, "object-store I/O semaphore closed unexpectedly"); + return Permit { + _permit: None, + _demand: demand, + }; + } + }; + let waited = queued.elapsed(); + if waited >= WAIT_LOG_THRESHOLD { + tracing::debug!( + lane = g.lane, + scope = key, + kind = kind.as_str(), + waited_ms = waited.as_millis() as u64, + concurrency = g.concurrency, + "object-store admission queued" + ); } - }; - leave_wait(kind); - permit + permit + } + }; + // A failure may have started a new cooldown while admission was + // queued. Release capacity before sleeping, then compete again. + if !cooldown_remaining(g, key).is_zero() { + drop(permit); + continue; } - }; - wait_out_degradation(kind).await; - Permit { - _permit: Some(permit), + // Interactive arrived while we waited for a background slot: give the + // permit back and yield instead of holding endpoint bandwidth. + if background && scope_foreground_demand(key) > 0 { + drop(permit); + continue; + } + return Permit { + _permit: Some(permit), + _demand: demand, + }; } } -fn enter_wait(kind: IoKind, reason: &'static str, wait_ms: u64) { +fn cooldown_remaining(g: &Gate, key: &str) -> Duration { + g.states + .lock() + .ok() + .and_then(|states| states.get(key).and_then(|state| state.cooldown_until)) + .and_then(|until| until.checked_duration_since(Instant::now())) + .unwrap_or_default() +} + +// Cancellation of an HTTP request must also release progress wait accounting. +struct WaitGuard<'a> { + gate: &'a Gate, + key: &'a str, + kind: IoKind, +} +impl<'a> WaitGuard<'a> { + fn new(gate: &'a Gate, key: &'a str, kind: IoKind, reason: &'static str, wait_ms: u64) -> Self { + enter_wait(gate, key, kind, reason, wait_ms); + Self { gate, key, kind } + } +} +impl Drop for WaitGuard<'_> { + fn drop(&mut self) { + leave_wait(self.gate, self.key, self.kind); + } +} + +fn enter_wait(g: &Gate, key: &str, kind: IoKind, reason: &'static str, wait_ms: u64) { let (delay_ms, failures) = { - let Ok(mut state) = gate().state.lock() else { + let Ok(mut states) = g.states.lock() else { return; }; + let state = state_for(&mut states, key, g.concurrency); state.last_kind = kind; state.active_waiters = state.active_waiters.saturating_add(1); (state.delay_ms, state.failures) @@ -294,18 +668,20 @@ fn enter_wait(kind: IoKind, reason: &'static str, wait_ms: u64) { }); } -fn leave_wait(kind: IoKind) { - if let Ok(mut state) = gate().state.lock() { +fn leave_wait(g: &Gate, key: &str, kind: IoKind) { + if let Ok(mut states) = g.states.lock() { + let state = state_for(&mut states, key, g.concurrency); state.active_waiters = state.active_waiters.saturating_sub(1); } emit_throttle(ObjectStoreThrottleEvent::Leave { kind }); } -fn emit_update(kind: IoKind, reason: &'static str, wait_ms: u64) { +fn emit_update(g: &Gate, key: &str, kind: IoKind, reason: &'static str, wait_ms: u64) { let (delay_ms, failures) = { - let Ok(state) = gate().state.lock() else { + let Ok(states) = g.states.lock() else { return; }; + let state = states.get(key).cloned().unwrap_or_default(); (state.delay_ms, state.failures) }; emit_throttle(ObjectStoreThrottleEvent::Update { @@ -317,9 +693,12 @@ fn emit_update(kind: IoKind, reason: &'static str, wait_ms: u64) { }); } -async fn wait_out_degradation(kind: IoKind) { +async fn wait_out_degradation(g: &Gate, key: &str, kind: IoKind) { let (sleep_for, delay_ms, failures) = { - let Ok(state) = gate().state.lock() else { + let Ok(states) = g.states.lock() else { + return; + }; + let Some(state) = states.get(key) else { return; }; let cooldown = state @@ -332,14 +711,17 @@ async fn wait_out_degradation(kind: IoKind) { return; } let wait_ms = sleep_for.as_millis() as u64; - enter_wait(kind, "throttle", wait_ms); + let _wait = WaitGuard::new(g, key, kind, "throttle", wait_ms); crate::store::index_build_progress::note(format!( "s3 {} throttle wait {:.1}s (failures={failures}, delay={delay_ms}ms)", kind.as_str(), sleep_for.as_secs_f32() )); - tracing::warn!( + // A cooldown can have several queued callers; the failure itself is + // already logged by `note_failure`, so each waiter need not be WARN noise. + tracing::debug!( target: "pchronicle.object_store_gate", + lane = g.lane, kind = kind.as_str(), wait_ms, delay_ms, @@ -347,24 +729,25 @@ async fn wait_out_degradation(kind: IoKind) { "object-store I/O gate cooling down before next remote op" ); // Tick the progress UI while cooling down so `cd=` counts down live. - let deadline = Instant::now() + sleep_for; const TICK: Duration = Duration::from_millis(250); loop { - let now = Instant::now(); - if now >= deadline { + let remaining = cooldown_remaining(g, key); + if remaining.is_zero() { break; } - let remaining = deadline - now; - emit_update(kind, "throttle", remaining.as_millis() as u64); + emit_update(g, key, kind, "throttle", remaining.as_millis() as u64); tokio::time::sleep(remaining.min(TICK)).await; } - leave_wait(kind); } /// Publish the current I/O phase for progress UI without taking a permit. /// Used around Lance writes that do not go through [`acquire`]. -pub(crate) fn mark_kind(kind: IoKind) { - if let Ok(mut state) = gate().state.lock() { +pub(crate) fn mark_kind(uri: &str, kind: IoKind) { + if !is_remote_uri(uri) { + return; + } + if let Ok(mut states) = gate().states.lock() { + let state = state_for(&mut states, &scope_key(uri), gate().concurrency); state.last_kind = kind; } } @@ -375,9 +758,11 @@ pub(crate) fn note_success(uri: &str) { return; } let (kind, changed, delay_ms, failures) = { - let Ok(mut state) = gate().state.lock() else { + let key = scope_key(uri); + let Ok(mut states) = gate().states.lock() else { return; }; + let state = state_for(&mut states, &key, gate().concurrency); let kind = state.last_kind; state.successes_since_backoff = state.successes_since_backoff.saturating_add(1); if state.delay_ms == 0 { @@ -421,9 +806,11 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { return; } let (delay_ms, failures, wait_ms) = { - let Ok(mut state) = gate().state.lock() else { + let key = scope_key(uri); + let Ok(mut states) = gate().states.lock() else { return; }; + let state = state_for(&mut states, &key, gate().concurrency); state.last_kind = kind; state.failures = state.failures.saturating_add(1); state.successes_since_backoff = 0; @@ -435,6 +822,7 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { state.cooldown_until = Some(Instant::now() + Duration::from_millis(state.delay_ms)); tracing::warn!( target: "pchronicle.object_store_gate", + scope = %key, kind = kind.as_str(), delay_ms = state.delay_ms, failures = state.failures, @@ -460,6 +848,226 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { mod tests { use super::*; + #[tokio::test] + async fn background_yields_while_foreground_holds_admission() { + let uri = scope_for_endpoint("s3://workload-isolation/path", "http://localhost:18060"); + let mut foreground_permits = Vec::new(); + for _ in 0..gate().concurrency { + foreground_permits.push(acquire(&uri, IoKind::Read).await); + } + assert!(foreground_object_store_demand() > 0); + with_background_object_store_io(async { + let raced = + tokio::time::timeout(Duration::from_millis(80), acquire(&uri, IoKind::Read)).await; + assert!( + raced.is_err(), + "background must yield while interactive holds the endpoint" + ); + }) + .await; + drop(foreground_permits); + assert_eq!(foreground_object_store_demand(), 0); + with_background_object_store_io(async { + let _permit = + tokio::time::timeout(Duration::from_millis(200), acquire(&uri, IoKind::Read)) + .await + .expect("background proceeds once interactive is idle"); + note_failure(&uri, IoKind::Read); + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + // Foreground AIMD is independent of the background failure above. + assert_eq!( + gate() + .states + .lock() + .unwrap() + .get(&uri) + .map(|s| s.failures) + .unwrap_or(0), + 0 + ); + let _permit = tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) + .await + .expect("background cooldown must not delay foreground requests"); + note_success(&uri); + with_background_object_store_io(async { + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + } + + #[tokio::test] + async fn foreground_and_background_have_independent_admission_and_feedback() { + let uri = scope_for_endpoint("s3://aimd-isolation/path", "http://localhost:18061"); + // Idle interactive: background may admit on its own lane immediately. + with_background_object_store_io(async { + let _permit = + tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) + .await + .expect("background must not wait when interactive is idle"); + note_failure(&uri, IoKind::Read); + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + assert_eq!( + gate().states.lock().unwrap().get(&uri).map(|s| s.failures), + None + ); + let _permit = tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) + .await + .expect("background cooldown must not delay foreground requests"); + note_success(&uri); + with_background_object_store_io(async { + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + } + + #[test] + fn registry_reclaims_idle_scopes_but_preserves_waits_and_cooldowns() { + let mut states = HashMap::new(); + state_for(&mut states, "waiting", 1).active_waiters = 1; + state_for(&mut states, "cooling", 1).cooldown_until = + Some(Instant::now() + Duration::from_secs(60)); + for n in 0..MAX_RETAINED_SCOPES * 2 { + state_for(&mut states, &n.to_string(), 1); + } + assert_eq!(states.len(), MAX_RETAINED_SCOPES); + assert!(states.contains_key("waiting")); + assert!(states.contains_key("cooling")); + assert!(!states.contains_key("0")); + for state in states.values_mut() { + state.last_used = Instant::now() - SCOPE_IDLE_TTL; + } + state_for(&mut states, "new", 1); + assert_eq!(states.len(), 3); + } + + #[test] + fn scopes_share_bucket_paths_and_isolate_endpoints_and_buckets() { + let scope = scope_for_endpoint( + "s3://bucket/a/table#old-cache-credentials", + "HTTP://LOCALHOST:9000/", + ); + assert_eq!( + scope, + scope_for_endpoint("s3://bucket/b", "http://localhost:9000") + ); + assert_eq!(scope_key(&scope), scope); + assert_ne!( + scope, + scope_for_endpoint("s3://bucket/a", "http://localhost:9001") + ); + assert_ne!( + scope, + scope_for_endpoint("s3://other/a", "http://localhost:9000") + ); + } + + #[tokio::test] + async fn busy_backend_does_not_block_other_endpoints_or_buckets() { + let g = Gate { + lane: "foreground", + concurrency: 1, + states: Mutex::new(HashMap::new()), + }; + let busy = scope_for_endpoint("s3://bucket/a", "http://slow"); + let held = acquire_scoped(&g, &busy, IoKind::Read).await; + for independent in [ + scope_for_endpoint("s3://bucket/a", "http://healthy"), + scope_for_endpoint("s3://other/a", "http://slow"), + ] { + let permit = tokio::time::timeout( + Duration::from_millis(200), + acquire_scoped(&g, &independent, IoKind::Read), + ) + .await + .unwrap(); + drop(permit); + } + // Registry reclamation must not replace a semaphore with a live permit. + { + let mut states = g.states.lock().unwrap(); + states.get_mut(&busy).unwrap().last_used = Instant::now() - SCOPE_IDLE_TTL; + for n in 0..MAX_RETAINED_SCOPES + 1 { + state_for(&mut states, &n.to_string(), g.concurrency); + } + assert!(states.contains_key(&busy)); + } + assert!( + tokio::time::timeout( + Duration::from_millis(50), + acquire_scoped(&g, &busy, IoKind::Read) + ) + .await + .is_err() + ); + assert_eq!(g.states.lock().unwrap()[&busy].active_waiters, 0); + drop(held); + let _permit = tokio::time::timeout( + Duration::from_millis(200), + acquire_scoped(&g, &busy, IoKind::Read), + ) + .await + .unwrap(); + } + + #[tokio::test] + async fn cooldown_after_queued_admission_releases_capacity_and_cancellation_clears_waiters() { + let g = Arc::new(Gate { + lane: "foreground", + concurrency: 1, + states: Mutex::new(HashMap::new()), + }); + let bad = scope_for_endpoint("s3://bucket/a", "http://slow"); + let healthy = scope_for_endpoint("s3://bucket/a", "http://healthy"); + let held = acquire_scoped(&g, &bad, IoKind::Read).await; + let waiter = tokio::spawn({ + let g = g.clone(); + let bad = bad.clone(); + async move { acquire_scoped(&g, &bad, IoKind::Read).await } + }); + tokio::time::timeout(Duration::from_secs(2), async { + while g + .states + .lock() + .unwrap() + .get(&bad) + .is_none_or(|s| s.active_waiters == 0) + { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + g.states + .lock() + .unwrap() + .get_mut(&bad) + .unwrap() + .cooldown_until = Some(Instant::now() + Duration::from_secs(10)); + drop(held); + // Cooling on one endpoint must not block another endpoint. + let permit = tokio::time::timeout( + Duration::from_millis(500), + acquire_scoped(&g, &healthy, IoKind::Read), + ) + .await + .unwrap(); + assert!(!waiter.is_finished()); + waiter.abort(); + assert!(waiter.await.is_err()); + assert_eq!(g.states.lock().unwrap()[&bad].active_waiters, 0); + drop(permit); + assert_eq!( + g.states.lock().unwrap()[&healthy] + .semaphore + .available_permits(), + 1 + ); + } + #[test] fn aimd_flow_label_healthy_and_degraded() { let healthy = ObjectStoreGateSnapshot { diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 265381c9f..a351521a1 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -4,20 +4,83 @@ //! module keeps pChronicle's own reads, listings and conditional writes on //! OpenDAL so backend differences are handled in one place. +use crate::store::object_store_io_gate::{self as io_gate, IoKind}; use anyhow::{Context, Result, anyhow}; use futures::TryStreamExt; use opendal::layers::RetryLayer; use opendal::{EntryMode, ErrorKind, Metadata, Operator}; -use std::collections::HashMap; +use std::collections::{BTreeMap, HashMap}; use std::sync::Arc; use std::sync::{Mutex, OnceLock}; -use std::time::Duration; +use std::time::{Duration, Instant}; use url::Url; +#[derive(Clone, Default, PartialEq, Eq)] +pub struct StoreConfig { + pub endpoint: Option, + pub region: Option, + pub access_key: Option, + pub secret_key: Option, +} + +impl std::fmt::Debug for StoreConfig { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("StoreConfig") + .field("endpoint", &self.endpoint) + .field("region", &self.region) + .finish_non_exhaustive() + } +} + +impl StoreConfig { + pub(crate) fn fingerprint(&self) -> String { + blake3::hash( + &serde_json::to_vec(&( + &self.endpoint, + &self.region, + &self.access_key, + &self.secret_key, + )) + .unwrap_or_default(), + ) + .to_hex() + .to_string() + } +} + +/// How long object-store retries may run before the caller gives up. +/// +/// A long import can wait out a multi-minute outage and should. A request +/// serving a browser cannot: retries that outlast its deadline turn every +/// unreachable endpoint into a timeout, so the operator sees "too slow" +/// where the truth is "cannot connect". The two workloads run in separate +/// processes, so each declares its own patience at startup. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RetryPatience { + /// Bounded well inside an interactive request budget. + Interactive, + /// Waits out long outages; the default. + Batch, +} + +static PATIENCE: OnceLock = OnceLock::new(); + +/// Declare this process's retry patience. Only the first call takes effect, +/// and it must happen before any object store is opened. +pub fn set_retry_patience(patience: RetryPatience) { + let _ = PATIENCE.set(patience); +} + /// Retries for transient object-store failures (DNS blips, connect resets, -/// 5xx, rate limits). Tuned for long imports over flaky endpoints: up to 8 -/// retries with exponential backoff + jitter, capped at 30s. +/// 5xx, rate limits): exponential backoff with jitter, bounded by the +/// process's declared patience. fn with_object_store_retries(operator: Operator) -> Operator { + let (max_delay, max_times) = match PATIENCE.get().copied().unwrap_or(RetryPatience::Batch) { + // 0.5s + 1s + 2s + 4s, so a failing operation still reports its own + // error well inside the request budget that wraps it. + RetryPatience::Interactive => (Duration::from_secs(4), 4), + RetryPatience::Batch => (Duration::from_secs(30), 8), + }; operator.layer( RetryLayer::new() .with_notify(|event: opendal::layers::RetryEvent<'_>| { @@ -33,8 +96,8 @@ fn with_object_store_retries(operator: Operator) -> Operator { .with_jitter() .with_factor(2.0) .with_min_delay(Duration::from_millis(500)) - .with_max_delay(Duration::from_secs(30)) - .with_max_times(8), + .with_max_delay(max_delay) + .with_max_times(max_times), ) } @@ -54,6 +117,7 @@ impl Version { pub(crate) struct Store { operator: Operator, fallback_lock: Option>>, + io_scope: String, } #[derive(Clone, Debug)] @@ -73,34 +137,133 @@ static SHARED_MEMORY: OnceLock>> = OnceLock::new static SHARED_LOCKS: OnceLock>>>> = OnceLock::new(); +const MAX_CACHED_OPERATORS: usize = 128; +const OPERATOR_IDLE_TTL: Duration = Duration::from_secs(300); + +#[derive(Default)] +struct OperatorRegistry { + entries: HashMap, +} + +impl OperatorRegistry { + fn get(&mut self, key: &str, now: Instant) -> Option { + self.entries + .retain(|_, (_, used)| now.duration_since(*used) < OPERATOR_IDLE_TTL); + self.entries.get_mut(key).map(|(operator, used)| { + *used = now; + operator.clone() + }) + } + + fn insert(&mut self, key: String, operator: Operator, now: Instant) { + if !self.entries.contains_key(&key) + && self.entries.len() >= MAX_CACHED_OPERATORS + && let Some(oldest) = self + .entries + .iter() + .min_by_key(|(_, (_, used))| *used) + .map(|(key, _)| key.clone()) + { + self.entries.remove(&oldest); + } + self.entries.insert(key, (operator, now)); + } +} + +static OPERATORS: OnceLock> = OnceLock::new(); + impl Store { + pub(crate) async fn from_uri_with_config(uri: &str, config: StoreConfig) -> Result { + if !uri.starts_with("s3://") || config == StoreConfig::default() { + return Self::from_uri(uri).await; + } + let parsed = Url::parse(uri).context("parse object-store URI")?; + let bucket = parsed + .host_str() + .ok_or_else(|| anyhow!("S3 URI must name a bucket"))?; + let root = parsed.path().trim_matches('/'); + let io_scope = io_gate::scope_for_endpoint(uri, config.endpoint.as_deref().unwrap_or("")); + let cache_key = format!( + "{}\0{}", + operator_cache_key(uri, parsed.as_str()), + config.fingerprint() + ); + let registry = OPERATORS.get_or_init(|| Mutex::new(OperatorRegistry::default())); + let cached = registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&cache_key, Instant::now()); + let operator = if let Some(operator) = cached { + operator + } else { + let mut builder = opendal::services::S3::default() + .bucket(bucket) + .root(root) + .region(config.region.as_deref().unwrap_or("us-east-1")); + if let Some(v) = config.endpoint.as_deref() { + builder = builder.endpoint(v); + } + if let Some(v) = config.access_key.as_deref() { + builder = builder.access_key_id(v); + } + if let Some(v) = config.secret_key.as_deref() { + builder = builder.secret_access_key(v); + } + let operator = with_object_store_retries(Operator::new(builder)?.finish()); + registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(cache_key, operator.clone(), Instant::now()); + operator + }; + Ok(Self { + operator, + fallback_lock: None, + io_scope, + }) + } + pub(crate) async fn from_uri(uri: &str) -> Result { let uri = uri.trim(); let normalized = normalize_uri(uri)?; + let cache_key = operator_cache_key(uri, &normalized); let shared_memory = uri.contains("://") && Url::parse(uri) .map(|parsed| parsed.scheme() == "shared-memory") .unwrap_or(false); - let operator = if shared_memory { - let map = SHARED_MEMORY.get_or_init(|| Mutex::new(HashMap::new())); - let mut map = map + // Memory operators own the data itself and must not be evicted like clients. + let operator = if normalized.starts_with("memory://") { + let mut map = SHARED_MEMORY + .get_or_init(|| Mutex::new(HashMap::new())) .lock() - .map_err(|_| anyhow!("shared-memory operator registry poisoned"))?; + .unwrap_or_else(std::sync::PoisonError::into_inner); if let Some(operator) = map.get(uri) { operator.clone() } else { + let operator = with_object_store_retries(Operator::from_uri(normalized.as_str())?); + map.insert(uri.to_owned(), operator.clone()); + operator + } + } else { + let registry = OPERATORS.get_or_init(|| Mutex::new(OperatorRegistry::default())); + let cached = registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&cache_key, Instant::now()); + if let Some(operator) = cached { + operator + } else { + // Construct outside the registry lock so one backend cannot block all others. let operator = with_object_store_retries( Operator::from_uri(normalized.as_str()) .with_context(|| format!("open OpenDAL store {uri}"))?, ); - map.insert(uri.to_string(), operator.clone()); + registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(cache_key, operator.clone(), Instant::now()); operator } - } else { - with_object_store_retries( - Operator::from_uri(normalized.as_str()) - .with_context(|| format!("open OpenDAL store {uri}"))?, - ) }; let fallback_lock = if shared_memory { let locks = SHARED_LOCKS.get_or_init(|| Mutex::new(HashMap::new())); @@ -119,26 +282,68 @@ impl Store { Ok(Self { operator, fallback_lock, + io_scope: io_gate::scope_key(uri), }) } + async fn remote(&self, kind: IoKind, request: F) -> Result + where + F: FnOnce(Operator) -> Fut, + Fut: std::future::IntoFuture>, + { + let _permit = io_gate::acquire(&self.io_scope, kind).await; + match request(self.operator.clone()).into_future().await { + Ok(value) => { + io_gate::note_success(&self.io_scope); + Ok(value) + } + Err(error) => { + if is_transient_error(&error) { + tracing::warn!(target: "pchronicle.opendal", + scope = %self.io_scope, kind = kind.as_str(), error = %error, + "remote operation failed; applying shared cooldown"); + io_gate::note_failure(&self.io_scope, kind); + } + Err(error.into()) + } + } + } + pub(crate) async fn read(&self, path: &str) -> Result, Version)>> { - let metadata = match self.operator.stat(path).await { + let path_owned = path.to_owned(); + let metadata = match self + .remote(IoKind::Read, |operator| async move { + operator.stat(&path_owned).await + }) + .await + { Ok(metadata) => metadata, - Err(error) if error.kind() == ErrorKind::NotFound => return Ok(None), - Err(error) => return Err(error.into()), + Err(error) + if error + .downcast_ref::() + .is_some_and(|error| error.kind() == ErrorKind::NotFound) => + { + return Ok(None); + } + Err(error) => return Err(error), }; - let bytes = self.operator.read(path).await?.to_vec(); + let path_owned = path.to_owned(); + let bytes = self + .remote(IoKind::Read, |operator| async move { + operator.read(&path_owned).await + }) + .await? + .to_vec(); Ok(Some((bytes, version(&metadata)))) } pub(crate) async fn write_create(&self, path: &str, bytes: Vec) -> Result<()> { - self.operator - .write_with(path, bytes) - .if_not_exists(true) - .await - .map(|_| ()) - .map_err(Into::into) + let path = path.to_owned(); + self.remote(IoKind::Write, |operator| async move { + operator.write_with(&path, bytes).if_not_exists(true).await + }) + .await + .map(|_| ()) } pub(crate) async fn write_match( @@ -155,127 +360,152 @@ impl Store { let condition = expected.condition().ok_or_else(|| { anyhow!("OpenDAL backend did not return an ETag/version for conditional write") })?; + let path = path.to_owned(); + let condition = condition.to_owned(); + let log_path = path.clone(); + let log_condition = condition.clone(); let result = self - .operator - .write_with(path, bytes.clone()) - .if_match(condition) - .await; - match result { - Ok(_) => Ok(()), - // Some S3-compatible gateways compare the If-Match header against - // their unquoted ETag, so a correctly quoted condition always - // fails with 412. One retry with the unquoted form still proves - // the stored ETag matched; real contention fails both attempts. - Err(error) - if error.kind() == ErrorKind::ConditionNotMatch - && let Some(unquoted) = unquoted_etag(condition) => - { - let retry = self - .operator - .write_with(path, bytes) - .if_match(unquoted) + .remote(IoKind::Write, |operator| async move { + let result = operator + .write_with(&path, bytes.clone()) + .if_match(&condition) .await; - match retry { + match result { Ok(_) => Ok(()), - // Preserve the original conditional conflict when the - // gateway rejects the compatibility form itself. - Err(retry_error) if retry_error.kind() != ErrorKind::ConditionNotMatch => { - Err(error) + // Some S3-compatible gateways compare the If-Match header against + // their unquoted ETag, so retry once with the compatibility form. + Err(error) + if error.kind() == ErrorKind::ConditionNotMatch + && let Some(unquoted) = unquoted_etag(&condition) => + { + match operator.write_with(&path, bytes).if_match(unquoted).await { + Ok(_) => Ok(()), + Err(retry_error) + if retry_error.kind() != ErrorKind::ConditionNotMatch => + { + Err(error) + } + Err(retry_error) => Err(retry_error), + } } - Err(retry_error) => Err(retry_error), + Err(error) => Err(error), } - } - Err(error) => Err(error), - } - .map_err(|error| { - if is_conflict(&error) { + }) + .await; + result.inspect_err(|error| { + if let Some(error) = error.downcast_ref::() + && is_conflict(error) + { tracing::debug!( target: "pchronicle.opendal", - path, - if_match = condition, + path = log_path, + if_match = log_condition, error = %error, kind = ?error.kind(), "conditional object write conflict (If-Match)" ); } - error.into() }) } pub(crate) async fn write_overwrite(&self, path: &str, bytes: Vec) -> Result<()> { - self.operator - .write(path, bytes) - .await - .map(|_| ()) - .map_err(Into::into) + let path = path.to_owned(); + self.remote(IoKind::Write, |operator| async move { + operator.write(&path, bytes).await + }) + .await + .map(|_| ()) } pub(crate) async fn list(&self, prefix: &str) -> Result> { - let mut lister = self.operator.lister_with(prefix).recursive(true).await?; - let mut entries = Vec::new(); - while let Some(entry) = lister.try_next().await? { - if entry.metadata().mode() == EntryMode::FILE { - entries.push(Entry { - path: entry.path().to_string(), - metadata: entry.metadata().clone(), - }); + let prefix = prefix.to_owned(); + self.remote(IoKind::Read, |operator| async move { + let mut lister = operator.lister_with(&prefix).recursive(true).await?; + let mut entries = Vec::new(); + while let Some(entry) = lister.try_next().await? { + if entry.metadata().mode() == EntryMode::FILE { + entries.push(Entry { + path: entry.path().to_string(), + metadata: entry.metadata().clone(), + }); + } } - } - Ok(entries) + Ok(entries) + }) + .await } /// Non-recursive listing of the immediate children under `prefix`. /// Returns both files and directories so callers can navigate lazily. pub(crate) async fn list_shallow(&self, prefix: &str) -> Result> { - let mut lister = self.operator.lister_with(prefix).recursive(false).await?; - let mut entries = Vec::new(); - while let Some(entry) = lister.try_next().await? { - entries.push(ShallowEntry { - path: entry.path().to_string(), - mode: entry.metadata().mode(), - metadata: entry.metadata().clone(), - }); - } - Ok(entries) + let prefix = prefix.to_owned(); + self.remote(IoKind::Read, |operator| async move { + let mut lister = operator.lister_with(&prefix).recursive(false).await?; + let mut entries = Vec::new(); + while let Some(entry) = lister.try_next().await? { + entries.push(ShallowEntry { + path: entry.path().to_string(), + mode: entry.metadata().mode(), + metadata: entry.metadata().clone(), + }); + } + Ok(entries) + }) + .await } pub(crate) async fn stat_file(&self, path: &str) -> Result> { - match self.operator.stat(path).await { + let path_owned = path.to_owned(); + match self + .remote(IoKind::Read, |operator| async move { + operator.stat(&path_owned).await + }) + .await + { Ok(metadata) if metadata.mode() == EntryMode::FILE => Ok(Some(Entry { path: path.to_string(), metadata, })), Ok(_) => Ok(None), - Err(error) if error.kind() == ErrorKind::NotFound => Ok(None), - Err(error) => Err(error.into()), + Err(error) + if error + .downcast_ref::() + .is_some_and(|error| error.kind() == ErrorKind::NotFound) => + { + Ok(None) + } + Err(error) => Err(error), } } pub(crate) async fn exists(&self) -> Result { - Ok(self - .operator - .lister_with("") - .recursive(true) - .await? - .try_next() - .await? - .is_some()) + self.remote(IoKind::Read, |operator| async move { + Ok(operator + .lister_with("") + .recursive(true) + .await? + .try_next() + .await? + .is_some()) + }) + .await } pub(crate) async fn remove_all(&self) -> Result<()> { - self.operator - .delete_with("") - .recursive(true) - .await - .map_err(Into::into) + self.remote(IoKind::Write, |operator| async move { + operator.delete_with("").recursive(true).await + }) + .await + .map(|_| ()) } pub(crate) async fn remove(&self, path: &str) -> Result<()> { - self.operator - .delete_with(path) - .recursive(true) - .await - .map_err(Into::into) + let path = path.to_owned(); + self.remote(IoKind::Write, |operator| async move { + operator.delete_with(&path).recursive(true).await + }) + .await + .map(|_| ()) } pub(crate) fn fallback_lock(&self) -> Option>> { @@ -283,6 +513,22 @@ impl Store { } } +fn operator_cache_key(uri: &str, normalized: &str) -> String { + let config: BTreeMap<_, _> = std::env::vars() + .filter(|(key, _)| { + key.starts_with("AWS_") + || key.starts_with("AZURE_") + || key.starts_with("GOOGLE_") + || matches!( + key.as_str(), + "HTTP_PROXY" | "HTTPS_PROXY" | "NO_PROXY" | "ALL_PROXY" + ) + }) + .collect(); + let fingerprint = blake3::hash(&serde_json::to_vec(&config).unwrap_or_default()).to_hex(); + format!("{uri}\0{normalized}\0{fingerprint}") +} + pub(crate) fn is_conflict(error: &opendal::Error) -> bool { matches!( error.kind(), @@ -304,6 +550,13 @@ pub(crate) fn version(metadata: &Metadata) -> Version { } } +fn is_transient_error(error: &opendal::Error) -> bool { + // RetryLayer marks all returned errors persistent, even 404/403/412. + // Never classify by response headers or request IDs in the display text. + matches!(error.kind(), ErrorKind::Unexpected | ErrorKind::RateLimited) + && (error.is_temporary() || error.is_persistent()) +} + fn normalize_uri(uri: &str) -> Result { if !uri.contains("://") { let path = std::path::Path::new(uri); @@ -334,6 +587,118 @@ fn normalize_uri(uri: &str) -> Result { mod tests { use super::*; + #[tokio::test] + async fn explicit_backends_scope_gate_by_endpoint_and_bucket_not_credentials() { + let config = StoreConfig { + endpoint: Some("http://127.0.0.1:18060".into()), + region: Some("us-east-1".into()), + access_key: Some("scope-test-key".into()), + secret_key: Some("scope-test-secret".into()), + }; + let a = Store::from_uri_with_config("s3://scope-test/a", config.clone()) + .await + .unwrap(); + let mut other = config.clone(); + other.endpoint = Some("http://127.0.0.1:18061".into()); + let b = Store::from_uri_with_config("s3://scope-test/a", other.clone()) + .await + .unwrap(); + assert_ne!(a.io_scope, b.io_scope); + assert_ne!(config.fingerprint(), other.fingerprint()); + other = config.clone(); + other.secret_key = Some("rotated-secret".into()); + assert_ne!(config.fingerprint(), other.fingerprint()); + let c = Store::from_uri_with_config("s3://scope-test/b", other) + .await + .unwrap(); + assert_eq!(a.io_scope, c.io_scope); + assert!(!format!("{config:?}").contains("scope-test-secret")); + assert!(!format!("{config:?}").contains("scope-test-key")); + // Saturate A without making a network request: B must retain admission. + let _a = io_gate::acquire(&a.io_scope, IoKind::Read).await; + let _b = tokio::time::timeout( + Duration::from_millis(100), + io_gate::acquire(&b.io_scope, IoKind::Read), + ) + .await + .unwrap(); + } + + #[test] + fn temporary_and_exhausted_transport_errors_still_trigger_backoff() { + for kind in [ErrorKind::Unexpected, ErrorKind::RateLimited] { + assert!(is_transient_error( + &opendal::Error::new(kind, "backend failure").set_temporary() + )); + assert!(is_transient_error( + &opendal::Error::new(kind, "backend failure").set_persistent() + )); + } + assert!(!is_transient_error( + &opendal::Error::new(ErrorKind::Unexpected, "invalid response").set_permanent() + )); + assert!(!is_transient_error( + &opendal::Error::new(ErrorKind::ConfigInvalid, "connection").set_persistent() + )); + } + + #[tokio::test] + async fn missing_markers_with_connection_headers_do_not_trigger_cooldown() -> Result<()> { + // RetryLayer marks even non-retryable errors persistent. S3 includes + // response headers in the error context, including `connection`. + let mut store = Store::from_uri("shared-memory://missing-marker-gate-test").await?; + store.io_scope = io_gate::scope_key("s3://missing-marker-gate-test"); + for kind in [ + ErrorKind::NotFound, + ErrorKind::PermissionDenied, + ErrorKind::ConditionNotMatch, + ] { + let error = store + .remote::<(), _, _>(IoKind::Read, |_| async move { + Err(opendal::Error::new(kind, "S3 response") + .with_context("response", "connection: keep-alive; request-id: 503429") + .set_persistent()) + }) + .await + .unwrap_err(); + assert_eq!(error.downcast_ref::().unwrap().kind(), kind); + tokio::time::timeout( + Duration::from_millis(200), + store.remote(IoKind::Read, |_| async { Ok(()) }), + ) + .await + .context("non-transient response started AIMD cooldown")??; + } + Ok(()) + } + + #[tokio::test] + async fn operator_registry_bounds_clients_without_invalidating_live_handles() -> Result<()> { + let store = Store::from_uri("shared-memory://registry-test").await?; + let mut registry = OperatorRegistry::default(); + let now = Instant::now(); + registry.insert("first".into(), store.operator.clone(), now); + let live = registry.get("first", now).unwrap(); + for n in 0..MAX_CACHED_OPERATORS { + registry.insert( + n.to_string(), + store.operator.clone(), + now + Duration::from_millis(1), + ); + } + assert_eq!(registry.entries.len(), MAX_CACHED_OPERATORS); + assert!(!registry.entries.contains_key("first")); + live.write("probe", "still alive").await?; + assert_eq!(live.read("probe").await?.to_vec(), b"still alive"); + assert!( + registry + .get("0", now + OPERATOR_IDLE_TTL + Duration::from_secs(1)) + .is_none() + ); + assert!(registry.entries.is_empty()); + Ok(()) + } + #[test] fn unquoted_etag_strips_one_quote_pair() { assert_eq!(unquoted_etag("\"abc\""), Some("abc")); diff --git a/crates/persisting-pchronicle/src/store/persistent_cache.rs b/crates/persisting-pchronicle/src/store/persistent_cache.rs index e2a43ae38..9c2721baf 100644 --- a/crates/persisting-pchronicle/src/store/persistent_cache.rs +++ b/crates/persisting-pchronicle/src/store/persistent_cache.rs @@ -2,6 +2,7 @@ use anyhow::{Context, Result}; use futures::TryStreamExt; use lance::Dataset; +use lance::dataset::optimize::{CompactionOptions, compact_files}; use lance::dataset::{InsertBuilder, MergeInsertBuilder, WhenMatched, WhenNotMatched}; use lance::deps::arrow_array::{Array, RecordBatch, RecordBatchIterator, StringArray}; use lance::deps::arrow_schema::{DataType, Field, Schema}; @@ -11,10 +12,14 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use tokio::sync::RwLock; const SCHEMA_VERSION: &str = "persistent-cache-v1"; +/// Fragments tolerated before a write folds them back together. +const MAX_FRAGMENTS: usize = 32; +const COMPACTION_TARGET_ROWS: usize = 16 * 1024; pub struct PersistentCache { path: PathBuf, values: RwLock>, disk_lock: Option, + write_gate: tokio::sync::Mutex<()>, } impl PersistentCache where @@ -37,20 +42,46 @@ where path, values: RwLock::new(HashMap::new()), disk_lock: lock.ok(), + write_gate: tokio::sync::Mutex::new(()), }; - if cache.disk_lock.is_some() { + let writable = cache.writable(); + let attempts = if writable { 1 } else { 5 }; + let mut loaded = None; + let mut last_error = None; + for attempt in 0..attempts { match cache.load().await { - Ok(values) => *cache.values.get_mut() = values, + Ok(values) => { + loaded = Some(values); + break; + } Err(error) => { - tracing::warn!(target: "pchronicle.serve", error = %error, "persistent cache unreadable; rebuilding"); - let _ = if cache.path.is_dir() { - std::fs::remove_dir_all(&cache.path) - } else { - std::fs::remove_file(&cache.path) - }; + last_error = Some(error); + if attempt + 1 < attempts { + tokio::time::sleep(std::time::Duration::from_millis( + 50 * (1u64 << attempt), + )) + .await; + } } } } + if let Some(values) = loaded { + *cache.values.get_mut() = values; + if let Err(error) = cache.rewrite_if_fragmented().await { + tracing::warn!(target: "pchronicle.serve", error = %error, + "persistent cache compaction failed"); + } + } else if let Some(error) = last_error { + tracing::warn!(target: "pchronicle.serve", error = %error, writable, attempts, + "persistent cache unreadable"); + if writable { + let _ = if cache.path.is_dir() { + std::fs::remove_dir_all(&cache.path) + } else { + std::fs::remove_file(&cache.path) + }; + } + } cache } pub fn path(&self) -> &Path { @@ -67,48 +98,141 @@ where self.values.read().await.clone() } pub async fn upsert(&self, key: &K, value: &V, removed: &[K]) -> Result<()> { + self.upsert_many([(key, value)], removed).await + } + + /// Persist many entries in one merge-insert. + /// + /// Every merge-insert is a Lance commit that re-reads the target to find + /// matches, so writing a bulk observation one entry at a time costs + /// `entries × cache size`. Callers that produce a whole observation should + /// hand it over together. + pub async fn upsert_many<'a, I>(&self, entries: I, removed: &[K]) -> Result<()> + where + I: IntoIterator, + K: 'a, + V: 'a, + { if !self.writable() { return Ok(()); } - let schema = Arc::new(Schema::new(vec![ - Field::new("key", DataType::Utf8, false), - Field::new("payload", DataType::Utf8, false), - Field::new("schema_version", DataType::Utf8, false), - ])); - let batch = RecordBatch::try_new( - schema.clone(), - vec![ - Arc::new(StringArray::from(vec![serde_json::to_string(key)?])) as _, - Arc::new(StringArray::from(vec![serde_json::to_string(value)?])) as _, - Arc::new(StringArray::from(vec![SCHEMA_VERSION])) as _, - ], - )?; + let mut keys = Vec::new(); + let mut payloads = Vec::new(); + for (key, value) in entries { + keys.push(serde_json::to_string(key)?); + payloads.push(serde_json::to_string(value)?); + } + // Merge-insert rejects a source that repeats a key, and a later + // observation supersedes an earlier one in the same batch. + let mut seen = std::collections::HashSet::with_capacity(keys.len()); + for index in (0..keys.len()).rev() { + if !seen.insert(keys[index].clone()) { + keys.remove(index); + payloads.remove(index); + } + } + // An entry present in this batch was just observed, so a retirement + // recorded earlier in the same batch must not delete it. + let removed = removed + .iter() + .map(|key| serde_json::to_string(key)) + .collect::, _>>()? + .into_iter() + .filter(|key| !seen.contains(key)) + .collect::>(); + if keys.is_empty() && removed.is_empty() { + return Ok(()); + } + let _write = self.write_gate.lock().await; + let batch = cache_batch(keys, payloads)?; + let schema = batch.schema(); if self.path.exists() { let mut dataset = Dataset::open(self.path.to_string_lossy().as_ref()).await?; for chunk in removed.chunks(128) { let keys = chunk .iter() - .map(|key| { - Ok(format!( - "'{}'", - serde_json::to_string(key)?.replace("'", "''") - )) - }) - .collect::>>()? + .map(|key| format!("'{}'", key.replace("'", "''"))) + .collect::>() .join(","); dataset.delete(&format!("key IN ({keys})")).await?; } - MergeInsertBuilder::try_new(Arc::new(dataset), vec!["key".into()])? - .when_matched(WhenMatched::UpdateAll) - .when_not_matched(WhenNotMatched::InsertAll) - .try_build()? - .execute_reader(Box::new(RecordBatchIterator::new(vec![Ok(batch)], schema))) + if batch.num_rows() > 0 { + MergeInsertBuilder::try_new(Arc::new(dataset), vec!["key".into()])? + .when_matched(WhenMatched::UpdateAll) + .when_not_matched(WhenNotMatched::InsertAll) + .try_build()? + .execute_reader(Box::new(RecordBatchIterator::new(vec![Ok(batch)], schema))) + .await?; + } + self.compact_if_fragmented().await?; + } else if batch.num_rows() > 0 { + InsertBuilder::new(self.path.to_string_lossy().as_ref()) + .execute(vec![batch]) .await?; - } else { + } + Ok(()) + } + + /// Fold accumulated fragments back together. + /// + /// A merge-insert reads every fragment of the target to find matches, so a + /// cache grown one write at a time makes each later write scan every row + /// ever written. Compaction keeps that drift bounded between restarts. + async fn compact_if_fragmented(&self) -> Result<()> { + let mut dataset = Dataset::open(self.path.to_string_lossy().as_ref()).await?; + if dataset.get_fragments().len() <= MAX_FRAGMENTS { + return Ok(()); + } + compact_files( + &mut dataset, + CompactionOptions { + target_rows_per_fragment: COMPACTION_TARGET_ROWS, + max_rows_per_group: 1024, + num_threads: Some(1), + ..Default::default() + }, + None, + ) + .await?; + Ok(()) + } + + /// Replace a badly fragmented cache with one file holding what we just read. + /// + /// Caches written one entry at a time before batching landed hold a + /// fragment and a version per entry — tens of thousands of them, gigabytes + /// for a directory listing. Compacting that many fragments is far slower + /// than rewriting the contents we already have in memory, and only a + /// rewrite reclaims the superseded files. The cache is rebuildable, so + /// losing it to a crash mid-rewrite costs a refresh, nothing more. + async fn rewrite_if_fragmented(&self) -> Result<()> { + if !self.writable() || !self.path.exists() { + return Ok(()); + } + let fragments = Dataset::open(self.path.to_string_lossy().as_ref()) + .await? + .get_fragments() + .len(); + if fragments <= MAX_FRAGMENTS { + return Ok(()); + } + let values = self.values.read().await; + let mut keys = Vec::with_capacity(values.len()); + let mut payloads = Vec::with_capacity(values.len()); + for (key, value) in values.iter() { + keys.push(serde_json::to_string(key)?); + payloads.push(serde_json::to_string(value)?); + } + let entries = keys.len(); + let batch = cache_batch(keys, payloads)?; + std::fs::remove_dir_all(&self.path)?; + if batch.num_rows() > 0 { InsertBuilder::new(self.path.to_string_lossy().as_ref()) .execute(vec![batch]) .await?; } + tracing::info!(target: "pchronicle.serve", fragments, entries, + path = %self.path.display(), "rewrote fragmented persistent cache"); Ok(()) } async fn load(&self) -> Result> { @@ -141,6 +265,24 @@ where Ok(values) } } + +fn cache_batch(keys: Vec, payloads: Vec) -> Result { + let schema = Arc::new(Schema::new(vec![ + Field::new("key", DataType::Utf8, false), + Field::new("payload", DataType::Utf8, false), + Field::new("schema_version", DataType::Utf8, false), + ])); + let versions = vec![SCHEMA_VERSION; keys.len()]; + Ok(RecordBatch::try_new( + schema, + vec![ + Arc::new(StringArray::from(keys)) as _, + Arc::new(StringArray::from(payloads)) as _, + Arc::new(StringArray::from(versions)) as _, + ], + )?) +} + fn text_column(batch: &RecordBatch, name: &str) -> Result> { let array = batch .column(batch.schema().index_of(name)?) @@ -152,3 +294,127 @@ fn text_column(batch: &RecordBatch, name: &str) -> Result> { .map(|index| array.value(index).to_owned()) .collect()) } + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn repeated_single_writes_stay_compact_and_readable() -> Result<()> { + let temp = tempfile::tempdir()?; + let path = temp.path().join("cache.lance"); + let cache = PersistentCache::::open(path.clone()).await; + assert!(cache.writable()); + for index in 0..80 { + cache + .upsert(&format!("key-{index:03}"), &format!("value-{index}"), &[]) + .await?; + } + let fragments = Dataset::open(path.to_str().context("cache path")?) + .await? + .get_fragments() + .len(); + assert!(fragments <= MAX_FRAGMENTS, "{fragments} fragments"); + + // Compaction rewrites fragments, so every entry must survive it. + cache + .upsert(&"key-007".into(), &"rewritten".into(), &[]) + .await?; + cache.upsert(&"gone".into(), &String::new(), &[]).await?; + cache + .upsert(&"key-000".into(), &"kept".into(), &["gone".into()]) + .await?; + drop(cache); + let values = PersistentCache::::open(path) + .await + .values() + .await; + assert_eq!(values.len(), 80); + assert_eq!(values["key-007"], "rewritten"); + assert_eq!(values["key-000"], "kept"); + assert!(!values.contains_key("gone")); + Ok(()) + } + + #[tokio::test] + async fn opening_rewrites_a_cache_fragmented_by_earlier_writes() -> Result<()> { + let temp = tempfile::tempdir()?; + let path = temp.path().join("cache.lance"); + // Reproduce the pre-batching shape: one fragment and one version per + // entry, written without ever folding them back together. + let cache = PersistentCache::::open(path.clone()).await; + let schema = cache_batch(vec!["seed".into()], vec!["\"seed\"".into()])?.schema(); + for index in 0..(MAX_FRAGMENTS * 2) { + let batch = cache_batch( + vec![format!("\"key-{index:03}\"")], + vec![format!("\"value-{index}\"")], + )?; + if index == 0 { + InsertBuilder::new(path.to_str().context("cache path")?) + .execute(vec![batch]) + .await?; + } else { + let dataset = Dataset::open(path.to_str().context("cache path")?).await?; + MergeInsertBuilder::try_new(Arc::new(dataset), vec!["key".into()])? + .when_matched(WhenMatched::UpdateAll) + .when_not_matched(WhenNotMatched::InsertAll) + .try_build()? + .execute_reader(Box::new(RecordBatchIterator::new( + vec![Ok(batch)], + schema.clone(), + ))) + .await?; + } + } + drop(cache); + let before = Dataset::open(path.to_str().context("cache path")?) + .await? + .get_fragments() + .len(); + assert!(before > MAX_FRAGMENTS, "{before} fragments"); + + let cache = PersistentCache::::open(path.clone()).await; + let after = Dataset::open(path.to_str().context("cache path")?) + .await? + .get_fragments() + .len(); + assert!(after < before, "{before} fragments became {after}"); + assert!(after <= MAX_FRAGMENTS, "{after} fragments"); + let values = cache.values().await; + assert_eq!(values.len(), MAX_FRAGMENTS * 2); + assert_eq!(values["key-005"], "value-5"); + // Superseded files must go with the rewrite, not linger as versions. + let versions = std::fs::read_dir(path.join("_versions"))?.count(); + assert!(versions <= 2, "{versions} versions"); + Ok(()) + } + + #[tokio::test] + async fn a_batch_supersedes_its_own_repeats_and_keeps_what_it_writes() -> Result<()> { + let temp = tempfile::tempdir()?; + let path = temp.path().join("cache.lance"); + let cache = PersistentCache::::open(path.clone()).await; + cache.upsert(&"stale".into(), &"old".into(), &[]).await?; + let entries = [ + ("a".to_owned(), "first".to_owned()), + ("a".to_owned(), "last".to_owned()), + ("stale".to_owned(), "revived".to_owned()), + ]; + cache + .upsert_many( + entries.iter().map(|(key, value)| (key, value)), + &["stale".to_owned(), "absent".to_owned()], + ) + .await?; + drop(cache); + let values = PersistentCache::::open(path) + .await + .values() + .await; + assert_eq!(values["a"], "last"); + // A key this batch writes was just observed; its earlier retirement + // inside the same batch must not delete it. + assert_eq!(values["stale"], "revived"); + Ok(()) + } +} diff --git a/crates/persisting-pchronicle/src/store/storyline/datafusion.rs b/crates/persisting-pchronicle/src/store/storyline/datafusion.rs index 42ef80d07..97e9e6b21 100644 --- a/crates/persisting-pchronicle/src/store/storyline/datafusion.rs +++ b/crates/persisting-pchronicle/src/store/storyline/datafusion.rs @@ -27,7 +27,7 @@ use lance::Dataset; use lance::deps::arrow_schema::{Schema as ArrowSchema, SchemaRef}; use super::content::{ - content_columns, hydrate_selected_batches, open_objects, preview_selected_batches, + content_columns, hydrate_batches, native_json_columns, open_objects, preview_selected_batches, }; use super::{StorylineLanceStore, StorylineTablePaths}; use crate::store::datafusion_bridge::{from_datafusion, into_datafusion}; @@ -184,7 +184,12 @@ impl TableProvider for StorylineTableProvider { .create_plan() .await .map_err(|error| into_datafusion(error.into()))?; - let selected = selected_content_columns(self.kind, projection, &self.schema); + let selected = selected_content_columns( + self.kind, + projection, + &self.schema, + self.options.content_read_mode == StorylineContentReadMode::Full, + ); if selected.is_empty() { Ok(plan) } else { @@ -193,7 +198,7 @@ impl TableProvider for StorylineTableProvider { selected, match self.options.content_read_mode { StorylineContentReadMode::Full => { - ContentMaterializationMode::Full(self.objects.clone()) + ContentMaterializationMode::Full(self.objects.clone(), self.kind) } StorylineContentReadMode::Preview => ContentMaterializationMode::Preview, }, @@ -239,6 +244,7 @@ fn selected_content_columns( kind: StorylineTableKind, projection: Option<&Vec>, schema: &SchemaRef, + include_native_json: bool, ) -> HashSet<&'static str> { let projected = projection.map(|projection| { projection @@ -248,11 +254,17 @@ fn selected_content_columns( }); content_columns(kind) .iter() - .filter_map(|(name, _)| { + .map(|(name, _)| *name) + .chain( + native_json_columns(kind) + .iter() + .copied() + .filter(|_| include_native_json), + ) + .filter(|name| { projected .as_ref() .is_none_or(|projected| projected.contains(name)) - .then_some(*name) }) .collect() } @@ -267,7 +279,7 @@ struct ContentHydrationExec { #[derive(Debug, Clone)] enum ContentMaterializationMode { - Full(Arc), + Full(Arc, StorylineTableKind), Preview, } @@ -296,7 +308,7 @@ impl DisplayAs for ContentHydrationExec { let mut selected = self.selected.iter().copied().collect::>(); selected.sort_unstable(); let mode = match self.mode { - ContentMaterializationMode::Full(_) => "full", + ContentMaterializationMode::Full(..) => "full", ContentMaterializationMode::Preview => "preview", }; write!( @@ -351,8 +363,10 @@ impl ExecutionPlan for ContentHydrationExec { async move { let batch = batch?; let mut batches = match mode { - ContentMaterializationMode::Full(objects) => { - hydrate_selected_batches(&objects, vec![batch], &selected).await + ContentMaterializationMode::Full(objects, kind) => { + // Match point reads: native JSON envelopes can contain + // offloaded values too, including unknown_fields. + hydrate_batches(&objects, vec![batch], kind).await } ContentMaterializationMode::Preview => { preview_selected_batches(vec![batch], &selected) @@ -435,24 +449,15 @@ impl StorylineDataSource { paths: StorylineTablePaths, options: StorylineDataSourceOptions, ) -> Result { - let remote = paths.runs.to_string_lossy().contains("://") - && !paths.runs.to_string_lossy().starts_with("file:"); - let (runs, steps, tool_calls, objects) = if remote { - // Avoid four concurrent Lance opens against flaky S3 gateways. - ( - open_dataset(&paths.runs, paths.runs_version).await?, - open_dataset(&paths.steps, paths.steps_version).await?, - open_dataset(&paths.tool_calls, paths.tool_calls_version).await?, - open_objects(&paths.objects, paths.objects_version).await?, - ) - } else { - tokio::try_join!( - open_dataset(&paths.runs, paths.runs_version), - open_dataset(&paths.steps, paths.steps_version), - open_dataset(&paths.tool_calls, paths.tool_calls_version), - open_objects(&paths.objects, paths.objects_version), - )? - }; + // Each open is independently gated by endpoint/bucket AIMD. Running + // them together lets metadata HEADs overlap without bypassing the S3 + // safety limits. + let (runs, steps, tool_calls, objects) = tokio::try_join!( + open_dataset(&paths.runs, paths.runs_version), + open_dataset(&paths.steps, paths.steps_version), + open_dataset(&paths.tool_calls, paths.tool_calls_version), + open_objects(&paths.objects, paths.objects_version), + )?; let objects = Arc::new(objects); Ok(Self { paths, diff --git a/crates/persisting-pchronicle/src/store/storyline/mod.rs b/crates/persisting-pchronicle/src/store/storyline/mod.rs index d9a5dac79..cb984ede1 100644 --- a/crates/persisting-pchronicle/src/store/storyline/mod.rs +++ b/crates/persisting-pchronicle/src/store/storyline/mod.rs @@ -49,7 +49,7 @@ use std::io::Write; use std::path::{Path, PathBuf}; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; -use std::time::{Duration, SystemTime, UNIX_EPOCH}; +use std::time::{SystemTime, UNIX_EPOCH}; use anyhow::{Context, Result}; use fs2::FileExt; @@ -711,6 +711,33 @@ impl StorylineLanceStore { Ok(Some((paths.generation, ids))) } + /// Report whether one document is present, without materializing every + /// identity in the source. + /// + /// A source can hold tens of thousands of runs, so reading the whole + /// `document_id` column to answer "does this run live here" dominated + /// Explorer navigation. `document_id` carries a scalar index, so the + /// filtered read is a point lookup. + pub async fn contains_document(&self, document_id: &str) -> Result> { + let Some(paths) = self.current_table_paths().await? else { + return Ok(None); + }; + let predicate = format!("document_id = '{}'", document_id.replace('\'', "''")); + let batches = read_projected_batches( + &paths.runs, + paths.runs_version, + &["document_id"], + Some(&predicate), + ) + .await?; + let matched = batches.iter().map(RecordBatch::num_rows).sum::(); + anyhow::ensure!( + matched <= 1, + "duplicate document_id in committed Storyline snapshot" + ); + Ok(Some((paths.generation, matched == 1))) + } + pub(crate) async fn resolve_current_table_paths(&self) -> Result> { let current = self.read_current_control().await?; let Some(pointer) = current.control.committed else { @@ -1906,25 +1933,13 @@ pub(super) async fn open_dataset_uri(uri: &str) -> Result { let mut attempt = 0u32; loop { attempt += 1; - let _permit = crate::store::object_store_io_gate::acquire( - uri, - crate::store::object_store_io_gate::IoKind::Read, - ) - .await; - match Dataset::open(uri).await { - Ok(dataset) => { - crate::store::object_store_io_gate::note_success(uri); - return Ok(dataset); - } + match crate::storage::open_lance_dataset(uri).await { + Ok(dataset) => return Ok(dataset), Err(error) => { let error = anyhow::Error::from(error); if !is_transient_storage_error(&error) { return Err(error).with_context(|| format!("open Lance dataset {uri}")); } - crate::store::object_store_io_gate::note_failure( - uri, - crate::store::object_store_io_gate::IoKind::Read, - ); if attempt >= DATASET_OPEN_MAX_ATTEMPTS { return Err(error).with_context(|| format!("open Lance dataset {uri}")); } @@ -1939,9 +1954,8 @@ pub(super) async fn open_dataset_uri(uri: &str) -> Result { error = %error, "transient object-store error opening Lance dataset; retrying under I/O gate" ); - // Shared AIMD delay is applied on the next acquire(); keep a - // small per-attempt floor so we never spin. - tokio::time::sleep(Duration::from_millis(100)).await; + // Individual HEAD/range requests already fed the shared AIMD + // gate; do not add a second fixed backoff here. } } } @@ -2123,6 +2137,7 @@ async fn maintain_table_layout( } if options.optimize_indices { crate::store::object_store_io_gate::mark_kind( + path.to_string_lossy().as_ref(), crate::store::object_store_io_gate::IoKind::Write, ); crate::store::index_build_progress::note(format!( diff --git a/crates/persisting-pchronicle/src/store/storyline/mutation.rs b/crates/persisting-pchronicle/src/store/storyline/mutation.rs index 6d9f3cfc6..bf5da626a 100644 --- a/crates/persisting-pchronicle/src/store/storyline/mutation.rs +++ b/crates/persisting-pchronicle/src/store/storyline/mutation.rs @@ -359,6 +359,7 @@ async fn write_record_batch_reader( ) -> Result { let uri = path.to_string_lossy().into_owned(); crate::store::object_store_io_gate::mark_kind( + &uri, crate::store::object_store_io_gate::IoKind::Write, ); let mut dataset = InsertBuilder::new(&uri) diff --git a/crates/persisting-pchronicle/src/store/storyline/tests.rs b/crates/persisting-pchronicle/src/store/storyline/tests.rs index da70d62aa..fadb01538 100644 --- a/crates/persisting-pchronicle/src/store/storyline/tests.rs +++ b/crates/persisting-pchronicle/src/store/storyline/tests.rs @@ -379,6 +379,86 @@ async fn unknown_content_ref_magic_string_round_trips_as_literal() { ); } +#[tokio::test] +async fn datafusion_hydrates_offloaded_native_json_columns() { + let temporary = tempfile::tempdir().unwrap(); + let store = StorylineLanceStore::open(temporary.path()).await.unwrap(); + let large = serde_json::json!({"payload": "x".repeat(128 * 1024)}); + let mut expected = story("native-json-content"); + expected + .unknown_fields + .insert("codex", "source", "/events/10", large.clone()) + .unwrap(); + expected.refresh_unknown_key_counts().unwrap(); + expected.turns[0].extra = Some(large.clone()); + let mut envelope = story("native-json-envelope"); + for index in 0..16 { + envelope + .unknown_fields + .insert( + "codex", + "source", + format!("/events/{index}"), + serde_json::json!({"text": "y".repeat(8192)}), + ) + .unwrap(); + } + envelope.refresh_unknown_key_counts().unwrap(); + store + .replace_storylines(&[expected.clone(), envelope.clone()]) + .await + .unwrap(); + + let source = super::datafusion::StorylineDataSource::from_store(&store) + .await + .unwrap(); + let context = source.session_context().unwrap(); + // Project only native JSON: hydration must not depend on also selecting + // an ordinary content column such as notes or message_value. + let batches = context + .sql("SELECT unknown_fields FROM runs WHERE session_id = 'native-json-content'") + .await + .unwrap() + .collect() + .await + .unwrap(); + let actual = lance_arrow::json::decode_json( + batches[0] + .column(0) + .as_any() + .downcast_ref::() + .unwrap() + .value(0), + ); + let actual: crate::formats::unknown_fields::StorylineUnknownFields = + serde_json::from_str(&actual).unwrap(); + assert!( + actual == expected.unknown_fields, + "unknown field payload was not hydrated" + ); + let batches = context + .sql("SELECT * FROM runs WHERE session_id = 'native-json-envelope'") + .await + .unwrap() + .collect() + .await + .unwrap(); + let runs = story_runs_from_batch(&batches[0]).unwrap(); + assert!( + runs[0].unknown_fields == envelope.unknown_fields, + "unknown field envelope was not hydrated" + ); + let batches = context + .sql("SELECT * FROM steps WHERE session_id = 'native-json-content' ORDER BY turn_ordinal") + .await + .unwrap() + .collect() + .await + .unwrap(); + let steps = story_steps_from_batch(&batches[0]).unwrap(); + assert!(steps[0].extra == Some(large), "step extra was not hydrated"); +} + #[tokio::test] async fn default_store_accepts_large_compressible_unknown_value() { let mut expected = story("large-logical-unknown"); diff --git a/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs b/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs index 7bf10be58..f47ec891d 100644 --- a/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs +++ b/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs @@ -175,6 +175,46 @@ async fn atif_singleton_array_shape_is_lossless_through_storyline_lance() -> Res Ok(()) } +#[tokio::test] +async fn single_document_lookup_answers_without_listing_the_source() -> Result<()> { + let temporary = tempfile::tempdir()?; + let store = StorylineLanceStore::open(temporary.path()).await?; + let documents = (0..64) + .map(|index| { + let id = format!("run-{index:03}"); + let mut document = StorylineDocument::new(&id, "agent"); + document.trajectory_id = Some(id); + document + }) + .collect::>(); + store.replace_storylines(&documents).await?; + + let (generation, ids) = store + .document_ids_snapshot() + .await? + .expect("committed snapshot"); + assert_eq!(ids.len(), documents.len()); + + // Resolving one run must not depend on the source's document count: the + // Explorer path used to read every identity to answer this question. + let (lookup_generation, present) = store + .contains_document("run-017") + .await? + .expect("committed snapshot"); + assert_eq!(lookup_generation, generation); + assert!(present); + assert_eq!( + store.contains_document("run-999").await?, + Some((generation.clone(), false)) + ); + // A quoted identity must stay a value, never become SQL. + assert_eq!( + store.contains_document("run-017' OR '1'='1").await?, + Some((generation, false)) + ); + Ok(()) +} + #[tokio::test] async fn incremental_storyline_replace_preserves_global_collection_order() -> Result<()> { let temporary = tempfile::tempdir()?; diff --git a/docs/src/en/pchronicle/reference/cli.md b/docs/src/en/pchronicle/reference/cli.md index 693249542..6ace554d3 100644 --- a/docs/src/en/pchronicle/reference/cli.md +++ b/docs/src/en/pchronicle/reference/cli.md @@ -341,9 +341,16 @@ an isolated exec worker containing only authorized mounts. It also enables `pchronicle serve catalog dataset add|remove|list` and `issue|grant|revoke` rewrite that file and do not start HTTP; `issue` prints the user secret once. Restart serve after changing libraries, users, or grants. -The pool allows at most 8 workers and 32 admitted requests, with serial execution -per worker, a 60-second execution/queue timeout and a 10-second body-read timeout. -Overload returns 503; timeout or IPC failure discards the worker. Worker and disk +The pool reuses idle workers in the same authenticated scope and grows on demand +when they are busy, up to 4 workers per scope and 8 workers server-wide. At capacity, +requests wait for available capacity rather than a particular busy worker; at most +32 requests are admitted. Each worker still executes serially to keep IPC isolated. +A cleanup task runs every 30 seconds and reclaims workers idle for 120 seconds; +idle workers from other scopes may be reclaimed sooner when global capacity is full. +Queueing, startup and execution share a 60-second timeout; body reads have a +10-second timeout. Overload returns 503. Cancellation during execution or IPC +failure discards that worker; cancelling a queued request does not interrupt +other requests. Worker and disk cache identity includes the user, grants and backend credential version. Caches live under `PCHRONICLE_CACHE_DIR/workers/` or the system pchronicle cache directory. Workers receive backend keys over private IPC before starting runtime threads; @@ -400,3 +407,32 @@ pchronicle serve catalog dataset list --catalog-config FILE registers the URI and optional backend storage credentials without creating or deleting object-store data. `grant` / `revoke` add or remove library names on that user (v1 grants are library membership, not fine-grained permission flags). + + +### Request execution diagnostics + +The **Requests** tab shows the current browser tab session's recent API requests, their request IDs, +HTTP outcomes, execution phases and elapsed time. The indicator above Local / profile +opens the same tab. Select a request or enter its ID to inspect it while it is running. +Worker admission, worker startup and worker execution are separate stages; directory +browsing and scoped Runs queries report their own stages. `pending` stages have not +started; `skipped` stages were not used. Stages describe application operations, not +individual DNS, TCP or S3 requests. A successful partial directory response can still +report a background refresh in its note. + +`GET /api/requests/{request_id}` (also under `/api/v1`) returns the live snapshot: +`request_id`, `method`, `path`, `state`, `elapsed_ms`, `status`, `error`, `note`, +`phases` (`name`, `state`, `elapsed_ms`) and an optional nested `worker` snapshot. +This endpoint runs in the front server, independently of busy query workers. + +Tracking is opt-in: send a cryptographically random 32-character hexadecimal +`x-pchronicle-observer` token with the original request and the diagnostic lookup. +Set `x-request-id` on the original request to look it up before it completes. +The UI supplies both headers automatically. The observer token is a private capability, +not a replacement for catalog credentials; a different token cannot read the record. +The server retains at most 512 records for up to 10 minutes, in memory only; restarting +the server clears them. Unknown, expired or inaccessible IDs return 404. The UI keeps +40 recent entries and polls unfinished requests once per second. Diagnostic connection +failures are shown separately and can be retried. Query strings, bodies and credentials +are not included in the diagnostic record. Internal error details remain in server logs, +correlated by request ID. diff --git a/docs/src/en/pchronicle/reference/query-model.md b/docs/src/en/pchronicle/reference/query-model.md index 2cae7c447..e0c7dc169 100644 --- a/docs/src/en/pchronicle/reference/query-model.md +++ b/docs/src/en/pchronicle/reference/query-model.md @@ -79,7 +79,7 @@ uses `#field(term)`: | Selector | Meaning | | --- | --- | -| `#content` | `message_value`, `observation`, and `prompt` | +| `#content` | `message_value` (message body; the default for plain text) | | `#message` | `message_value` | | `#user` | `message_value` where `source = 'user'` | | `#assistant` | `message_value` where `source = 'agent'` (`#agent` is an alias) | diff --git a/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md b/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md index 4cde65186..0d2f481cf 100644 --- a/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md +++ b/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md @@ -3,334 +3,328 @@ | Field | Value | |---|---| | **Status** | Proposed | -| **Date** | 2026-08-30 | -| **Component** | pChronicle CLI、`pchronicle serve`、pChronicle Web | -| **Related** | [RFC-0003 Ownership](0003-pchronicle-ownership.md) · [Warehouse 指南](../pchronicle/guides/serve.md) · [CLI 参考](../pchronicle/reference/cli.md) · [架构](../pchronicle/design/architecture.md) | +| **Date** | 2026-09-18 | +| **Component** | pChronicle CLI, `pchronicle serve`, pChronicle Web | +| **Related** | [RFC-0003 Ownership](0003-pchronicle-ownership.md) · [Warehouse guide](../pchronicle/guides/serve.md) · [CLI reference](../pchronicle/reference/cli.md) · [Architecture](../pchronicle/design/architecture.md) | --- -## 摘要 +## Summary -本 RFC 定义 pChronicle 平台部署时打开 **path** 的一种方式:**Directory**(名字 → path + ACL + 换票)。 +This RFC defines one way to open a **path** in a pChronicle platform deployment: a **Directory** (name → path + ACL + ticket exchange). -Dataset 身份始终是 path(本机路径或 `s3://` / `az://` / `gs://` URI)。Directory 不是第三种 Dataset,也不替代 Snapshot。它只决定调用方可以解析到哪些 path;换票后的 `uri` 才是引擎打开的 Dataset。 +Dataset identity is always a path (local path or `s3://` / `az://` / `gs://` URI). Directory is not a third Dataset kind and does not replace Snapshot. It only decides which paths a caller may resolve; the ticket `uri` is what the engine opens. -CLI 标志、配置文件和 HTTP 路径为兼容性仍使用 `catalog` 一词(`--catalog-config`、`catalog.toml`、`catalog://`、`/api/v1/catalog/datasets`)。产品与 RFC 口径称 Directory。 +CLI flags, config files, and HTTP paths keep the word `catalog` for compatibility (`--catalog-config`, `catalog.toml`, `catalog://`, `/api/v1/catalog/datasets`). Product and RFC language use Directory. The config is TOML (extension may be `.toml` / `.yml`, etc.; content is still parsed as TOML). -规范实现挂在现有 `pchronicle serve --catalog-config` 上,不引入独立 `catalog serve` 进程。 -Listener 默认可为 loopback;也允许绑定非环回地址,但部署方 MUST 自行保证网络边界。 +The normative implementation hangs off `pchronicle serve --catalog-config`. There is no separate `catalog serve` process. +The listener defaults to loopback and MAY bind non-loopback addresses; deployers MUST own the network boundary. -- **Serve 挂载**:`pchronicle serve --catalog-config FILE` MUST 把 `catalog.toml` 中的 **全部** - `[datasets.*]` 挂进 Warehouse(与位置参数挂载等价)。本机 Web / 无用户钥的数据面请求在 - **父进程内**打开这些 mount,不再 front-only。 -- **CLI 配置**:`pchronicle serve catalog dataset add|remove|list` 改写 libraries; - `issue|grant|revoke` 改写用户与授权。 -- **Directory 换票**:`@team` 解析为 `catalog://…`;`@team/prod` 换票后客户端打开票里的 path。 - `/api/v1/catalog/datasets` 仍按用户钥过滤可见 library。 +- **Serve**: the parent only authenticates, serves the directory/tickets, and spawns workers (front-only). It MUST NOT open `[datasets.*]` in-process. Authorized data-plane requests open mounts in a one-shot `--catalog-query-worker`. +- **Public browse**: datasets granted with `user = "*"` are visible anonymously; the parent MAY cache browse projections for them without backend keys. +- **CLI config**: `pchronicle serve catalog dataset add|remove|list` rewrites datasets; `issue|grant|revoke` rewrites users and grants. Users/grants hot-reload; Dataset URI and backend credential changes REQUIRE a restart. +- **Directory tickets**: `@team` resolves as `catalog://…`; `@team/prod` opens the ticket path after exchange. `/api/v1/catalog/datasets` filters by user keys (no headers → public libraries only). ```text pchronicle serve catalog dataset add --catalog-config catalog.toml prod --uri s3://bucket/prod \ --access-key BACKEND_AK --secret-key BACKEND_SK pchronicle serve catalog issue --catalog-config catalog.toml alice -pchronicle serve catalog grant --catalog-config catalog.toml alice prod evals +pchronicle serve catalog grant --catalog-config catalog.toml alice prod pchronicle serve --catalog-config catalog.toml --listen 127.0.0.1:8081 pchronicle dataset pin team catalog://127.0.0.1:8081 --ak USER_AK --sk USER_SK pchronicle query @team/prod 'SELECT 1' ``` -## 动机 +## Motivation -本机路径和静态 Warehouse mount 假设操作者已经能看见全部 Dataset。把对象存储上的多个评测库交给一组人使用时,出现三个缺口: +Local paths and static Warehouse mounts assume the operator can already see every Dataset. Sharing several object-store evaluation libraries with a group leaves three gaps: -1. **发现与授权混在一起**。用户需要一份目录,列出自己可以打开的 library 名,而不是把所有 bucket URI 写进每人的 `config.toml`。 -2. **后端密钥不能进用户配置**。对象存储 ak/sk 属于存储账户;用户钥只用于 Directory 鉴权。把后端钥写入本机 dataset pin 会扩散到每台笔记本,也无法按人裁剪可见库。 -3. **Web 与 CLI 的数据面不同**。CLI 可以在换票后自己打开 `s3://`。Web 的查询跑在 serve 进程里;若父进程加载全部 library 的后端密钥并执行 SQL,一次鉴权绕过就会看到未授权库。 +1. **Discovery and authorization are mixed**. Users need a directory of library names they may open, not every bucket URI in each laptop `config.toml`. +2. **Backend keys must not live in user config**. Object-store ak/sk belong to the storage account; user keys only authenticate to Directory. Writing backend keys into local dataset pins spreads them and cannot trim visibility per person. +3. **Web and CLI data planes differ**. After a ticket exchange, the CLI can open `s3://` itself. Web queries run inside serve; if the parent loads every library's backend keys and runs SQL, one auth bypass sees unauthorized libraries. -本 RFC 把 Directory 定义为 **目录 + ACL + 换票**,把存储访问留给已有 `open(path)`,并把 Web 数据面隔离到一次性 worker。 +This RFC defines Directory as **directory + ACL + ticket exchange**, leaves storage access to existing `open(path)`, and isolates the Web data plane in a one-shot worker. -## 目标与非目标 +## Goals and non-goals -### 目标 +### Goals -- 用一份 `catalog.toml` 同时描述 libraries 和 users。 -- 用 CLI 签发用户钥并改写 ACL:`pchronicle serve catalog issue|grant|revoke` 不启动 HTTP。 -- 让 `@name/library` 解析为一条 path(换票后的 `uri`);引擎随后只打开该 path。 -- 换票后 CLI 自己访问存储;后端密钥只出现在票和 worker stdin 中,不写入用户 `config.toml`。 -- Web 用用户钥换授权范围,查询只看到该用户的 mounts。 -- 允许 Warehouse 绑定任意 listen 地址;默认示例仍用 loopback。Catalog 头不是公网认证边界,不可信网络上的暴露由部署方负责。 +- Describe `meta`, `users`, `datasets`, and `[[grants]]` in one Directory config. +- Issue user keys and rewrite ACL via CLI: `pchronicle serve catalog issue|grant|revoke` without starting HTTP. +- Resolve `@name/library` to one path (the ticket `uri`); the engine then opens only that path. +- After exchange, the CLI talks to storage itself; backend keys appear only in tickets and worker stdin, never in the user's `config.toml`. +- Web exchanges user keys for an authorized mount set; datasets with `user = "*"` MAY be listed/browsed anonymously. +- Hot-reload users and grants about every 3 seconds; reject hot-reload of Dataset definitions and backend credentials (restart required). +- Allow Warehouse to bind any listen address; examples stay on loopback. Catalog headers are not a public auth boundary; exposure on untrusted networks is the deployer's responsibility. -### 非目标 +### Non-goals -- STS、临时凭证轮换、或把用户钥映射成短时 AWS session。 -- 热加载 `catalog.toml`;改配置 MUST 重启 serve。 -- 在运行中的 Warehouse 上提供 HTTP 签发接口。 -- 提供独立 `catalog serve` 二进制。 -- 在已运行的 Tokio runtime 上 `fork(2)`(未定义行为)。 -- 把后端对象存储密钥写入本机 dataset pin 配置。 -- 改变 Snapshot 协议、SQL schema 或 Gateway/Control 协议。 +- STS, short-lived credential rotation, or mapping user keys to AWS sessions. +- Hot-reloading Dataset URI / endpoint / region / backend ak/sk (restart serve). +- HTTP mint APIs on a running Warehouse. +- A separate `catalog serve` binary. +- `fork(2)` of a running Tokio runtime (undefined behavior). +- Writing backend object-store keys into local dataset pin config. +- Enforcing fine-grained `permissions` in v1 (the field is writable; semantics remain library membership). +- Changing Snapshot protocol, SQL schema, or Gateway/Control protocols. -本 RFC 的 Directory 与打开 path 之后的 **Snapshot**(见 [Snapshot 设计](../pchronicle/design/catalog.md))不是同一对象。Directory 列出授权 path;Snapshot 钉住一条已打开 path 上的 Source 成员与版本。 +Directory in this RFC is not the same object as **Snapshot** after a path is opened (see [Snapshot design](../pchronicle/design/catalog.md)). Directory lists authorized paths; Snapshot pins Source membership and versions on an opened path. -## 角色与信任边界 +## Roles and trust boundary -| 角色 | 持有 | 用途 | +| Role | Holds | Use | |---|---|---| -| 存储账户 | 后端 `access_key` / `secret_key`,以及可选 endpoint、region | 打开 `s3://` library | -| Directory 用户 | 用户 `access_key` / `secret_key` | 列出/领取被授权 library 的票 | -| 本机 CLI | 用户钥(存在 dataset pin 配置) | 换票后把后端钥注入进程环境并打开票中的 path | -| 浏览器 | 用户钥(`localStorage`) | 作为请求头发给 loopback serve | -| serve 父进程 | 完整 `catalog.toml` | 鉴权、返回票、spawn worker;不把后端钥写入 AWS 环境 | -| query worker | 该用户被授权 library 的票 | 一次性执行 Warehouse 数据面请求 | +| Storage account | Backend `access_key` / `secret_key`, optional endpoint, region | Open `s3://` libraries | +| Directory user | User `access_key` / `secret_key` | List/fetch tickets for granted libraries | +| Local CLI | User keys (in dataset pin config) | After exchange, inject backend keys into process env and open the ticket path | +| Browser | User keys (`localStorage`) | Send as request headers to loopback serve | +| serve parent | Full Directory config | Authenticate, return tickets, spawn worker; do not write backend keys into AWS env | +| query worker | Tickets for that user's libraries | One-shot Warehouse data-plane request | -ACL 是 **发现与授权** 边界,不是对象存储的强制隔离。持有后端密钥或能猜测 URI 的调用方,仍可能绕过 Directory 直接访问存储。Directory 不替代 bucket policy。 +ACL is a **discovery and authorization** boundary, not mandatory object-store isolation. Callers who hold backend keys or can guess URIs may still bypass Directory. Directory does not replace bucket policy. -## 进程模型 +## Process model -Directory 挂在现有 Warehouse listener 上。未传 `--catalog-config` 时,`pchronicle serve` 行为不变:静态 mount、无用户鉴权。 +Directory hangs on the existing Warehouse listener. Without `--catalog-config`, `pchronicle serve` is unchanged: static mounts, no user auth. ```text -浏览器 / CLI +browser / CLI → Warehouse listener ├─ GET /health - ├─ GET /api/v1/catalog/datasets[/{name}] 父进程:鉴权 + 目录/票 - ├─ 静态 UI - └─ 其余 /api/* 父进程内挂载 / 或 spawn worker + ├─ GET /api/v1/catalog/datasets[/{name}] parent: auth + directory/ticket + ├─ static UI + └─ other /api/* parent mounts / or spawn worker → pchronicle serve --catalog-query-worker - stdin: mounts + HTTP 请求 + stdin: mounts + HTTP request stdout: status / content-type / body - 退出 + exit ``` -约束: +Constraints: -1. Listener MAY 绑定非 loopback 地址。本 RFC 不把 catalog 头当作公网认证边界;部署方 MUST 在不可信网络上自行加边界。 -2. 父进程 MUST NOT 打开 `catalog.toml` 中的 libraries。父进程使用空 mount 的 front-only Warehouse。 -3. Worker MUST 由 `Command` 启动新进程,MUST NOT `fork(2)` 已运行的 Tokio runtime。 -4. Worker MUST NOT 监听端口、MUST NOT 读取 `catalog.toml`、MUST NOT 读取用户钥。它只消费 stdin 中过滤后的 mounts 和原始请求。 -5. Worker 继承父进程环境(证书、`PATH` 等),但父进程 MUST NOT 预先把 catalog 后端密钥写入 `AWS_*`。Worker 在打开存储前为自己设置该用户票中的后端环境。 -6. 同一 `catalog.toml` 内所有 `s3://` library MUST 共用同一组 endpoint、region 和后端 ak/sk。进程级 AWS 环境一次只能持有一套凭据。 -7. 隐藏 flag `--catalog-query-worker` MUST NOT 出现在用户可见的 `serve --help` 中。 +1. The listener MAY bind non-loopback. This RFC does not treat catalog headers as a public auth boundary; deployers MUST add a boundary on untrusted networks. +2. The parent MUST NOT open datasets from the config. It uses a front-only Warehouse with empty mounts; public browse caches hold paths only and MUST NOT write backend keys into parent `AWS_*`. +3. Workers MUST be started with `Command`, MUST NOT `fork(2)` a running Tokio runtime. +4. Workers MUST NOT listen, MUST NOT read the Directory config, MUST NOT read user keys. They only consume filtered mounts and the raw request from stdin. +5. Workers inherit parent env (certs, `PATH`, …), but the parent MUST NOT pre-write catalog backend keys into `AWS_*`. The worker sets backend env from the user's tickets before opening storage. +6. Each `[datasets.*]` MAY carry its own endpoint, region, and backend ak/sk. One worker process can hold only one process-global AWS env; if a user is granted incompatible `s3://` backends, the request MUST include `dataset=` to select one, or MUST fail. +7. The hidden flag `--catalog-query-worker` MUST NOT appear in user-facing `serve --help`. -Worker 超时后父进程 MUST 返回 `unavailable`,不得把 stdin 中的密钥写进日志。 +On worker timeout the parent MUST return `unavailable` and MUST NOT log keys from stdin. -## 配置 +## Configuration -`catalog.toml` 是唯一配置面: +The Directory config only manages users, datasets, and grants. It is the single source of truth; runtime serve options still come from `pchronicle serve` flags. When the file is missing, catalog management commands create an empty catalog (with `[meta]`). + +Authoritative schema (matches current implementation / deployment samples): ```toml -[libraries.prod] -uri = "s3://bucket/prod" -endpoint = "http://127.0.0.1:9000" -region = "us-west-2" -access_key = "BACKEND_AK" -secret_key = "BACKEND_SK" - -[libraries.evals] -uri = "s3://bucket/evals" -endpoint = "http://127.0.0.1:9000" -region = "us-west-2" -access_key = "BACKEND_AK" -secret_key = "BACKEND_SK" +[meta] +version = 1 +revision = 1 +name = "default" [users.alice] -access_key = "USER_AK" -secret_key = "USER_SK" -datasets = ["prod", "evals"] - -[users.bob] -access_key = "BOB_AK" -secret_key = "BOB_SK" -datasets = ["evals"] +access_key = "pcak_…" +secret_key = "…" + +[datasets.default] +uri = "/data/warehouse" + +[datasets.prod] +uri = "s3://prod/" +endpoint = "http://s3-a.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_A" +secret_key = "BACKEND_SK_A" + +[datasets.prod2] +uri = "s3://prod" +endpoint = "http://s3-b.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_B" +secret_key = "BACKEND_SK_B" + +[[grants]] +user = "*" +dataset = "prod" + +[[grants]] +user = "*" +dataset = "prod2" + +[[grants]] +user = "alice" +dataset = "default" +# permissions optional; v1 ignores fine-grained semantics (membership only) +# permissions = ["read", "query", "analyze"] ``` -规则: - -- 启动 `pchronicle serve --catalog-config` MUST 至少有一个 library 和一个 user。 -- `pchronicle serve catalog issue` MAY 在只有 `[libraries.*]`、尚无 `[users]` 的文件上签发第一个用户。 -- library 名与用户段名 MUST 是合法 Dataset mount 名(小写 `[A-Za-z_][A-Za-z0-9_]*`)。 -- `s3://` library MUST 同时设置后端 `access_key` 和 `secret_key`。 -- 非 `s3://` library MUST NOT 设置后端密钥。 -- 所有 `s3://` library 的 endpoint、region、后端密钥 MUST 完全一致。 -- `users.*.datasets` 引用的名字 MUST 存在于 `libraries`。 -- 用户 `access_key` MUST 全局唯一。 -- 配置文件 MUST 是普通文件,大小有上界;解析失败则 serve 拒绝启动。 +Rules: -本地路径 library 允许不设后端密钥,便于同机目录通过 catalog 做授权发现。客户端换票后仍按票中的 URI 打开。 +- `meta.version` MUST be a supported config version; successful CLI writes SHOULD maintain `meta.revision` / `meta.name` (optional). +- User and Dataset names MUST be lowercase `[A-Za-z_][A-Za-z0-9_]*`. +- `[users.*]` contains only `access_key` / `secret_key`; `access_key` MUST be globally unique; v1 allows plaintext `secret_key`. +- `[datasets.*]` MUST include `uri`; local paths MUST NOT set backend keys; `s3://` MUST set both `access_key` and `secret_key`, and MAY set `endpoint` / `region`. +- Different datasets MAY use different endpoint / region / backend keys (see process model item 6). +- `[[grants]]` MUST include `user` and `dataset`; `permissions` is optional and not enforced in v1. +- `grants.user = "*"` marks the dataset public (anonymous list/browse) and expands to **all current** users at parse time; newly issued users inherit it after hot-reload. +- Named `grants.user` MUST reference an existing user; `grants.dataset` MUST reference an existing dataset. +- Duplicate grants for the same user and dataset MUST be rejected (including after `*` expansion). +- Config size MUST be bounded; parse/validation failure refuses serve start; hot-reload failure MUST keep the last valid ACL. +- TOML is authoritative; future SQLite/Postgres may only be indexes or derived projections. -## CLI 签发与授权 +## CLI management -签发和改授权是 **写 `catalog.toml` 的 CLI**,不是运行中 Warehouse 的 HTTP API。出现 `catalog` 子命令时 MUST NOT 启动 listener。运行中的 serve 每 3 秒检查配置,用户和授权无需重启即可生效。 +Catalog management commands only edit the config file and do not start an HTTP listener. Missing files create the parent directory and an empty config. ```text -pchronicle serve catalog dataset add --catalog-config FILE NAME --uri URI [--endpoint URL] [--region REGION] [--access-key KEY] [--secret-key KEY] +pchronicle serve catalog dataset add --catalog-config FILE NAME --uri URI [OPTIONS] pchronicle serve catalog dataset remove --catalog-config FILE NAME... pchronicle serve catalog dataset list --catalog-config FILE + pchronicle serve catalog issue --catalog-config FILE NAME pchronicle serve catalog grant --catalog-config FILE NAME DATASET... pchronicle serve catalog revoke --catalog-config FILE NAME DATASET... -pchronicle serve --catalog-config FILE --listen 127.0.0.1:8081 ``` -`catalog` 是 `serve` 的保留子命令。要挂载名为 `catalog` 的路径,使用 `./catalog` 或 `NAME=./catalog`。 -`--catalog-config` MUST NOT 与位置参数 Dataset 同时使用。 - -### `dataset add` / `remove` / `list` - -- MUST NOT 启动 Warehouse。只改 `FILE` 后退出。 -- `add` 写入 `[datasets.NAME]`。已存在的名字 MUST 拒绝。`s3://` MUST 设置后端钥,且 MUST 与文件中已有 s3 library 的 endpoint/region/ak/sk 完全一致;非 `s3://` MUST NOT 设置后端钥。 -- `remove` 删除列出的 library。若仍有 grant 引用该 library,MUST 失败且 MUST NOT 改文件。 -- `list` 打印 `name` / `uri`(及可选 endpoint/region),MUST NOT 打印后端密钥。 - -### `issue` - -- MUST NOT 启动 Warehouse。只改 `FILE` 后退出。 -- 已存在的用户名 MUST 拒绝,MUST NOT 覆盖或轮换密钥。本 RFC 不引入 `issue --rotate`。 -- 生成的用户钥: - - `access_key`:`pcak_` 前缀 + 24 字节小写 hex(48 个 hex 字符) - - `secret_key`:32 字节小写 hex(无前缀) -- 写入 `[users.NAME]`:`access_key`、`secret_key`、`datasets = []`。签发 MUST NOT 授予任何 library。 -- stdout 打印该用户的 `name` / `access_key` / `secret_key`(表或 JSON)。secret MUST 只在这次 stdout 出现;stderr 只报 `config= updated=true`,MUST NOT 打印 sk。`dataset list` 等其它命令 MUST NOT 回显 catalog 用户 sk。 -- `access_key` 碰撞时 MUST 重试生成,MUST NOT 写入半截配置。 - -### `grant` / `revoke` - -- `grant` 是累加:已授权的 library 保持不变,新名字追加。未知用户或未知 library MUST 失败,且 MUST NOT 改文件。 -- `revoke` 从该用户的 `datasets` 里去掉列出的名字。未知用户、或该用户当前并未持有的 library 名 MUST 失败。 -- 两个命令的 stdout 只报 `name` 与更新后的 `datasets`,MUST NOT 打印密钥。 - -改写配置可以整表重写,不要求保留注释。新用户通常在 3 秒内生效。 +`dataset add` only registers a Dataset; it does not create or delete backend data. When appending `s3://` via CLI, if other `s3://` entries already exist, the new endpoint / region / backend keys MUST match them exactly (hand-written multi-backend configs remain valid, but workers must select per item 6). `issue` generates user AK/SK (secret printed once on stdout) and MUST NOT write any grant. `grant` / `revoke` edit `[[grants]]`; NAME `*` writes/removes **named** grants for every **current** user (it does not write a `user = "*"` public row). All writes MUST replace the file atomically and keep the previous file on failure. ## HTTP -Directory 路由与 Warehouse 共用 `/api` 与 `/api/v1` 前缀。鉴权头: +Directory routes share Warehouse `/api` and `/api/v1` prefixes. Auth headers: -| Header | 含义 | +| Header | Meaning | |---|---| -| `x-pchronicle-access-key` | 用户 access key | -| `x-pchronicle-secret-key` | 用户 secret key | +| `x-pchronicle-access-key` | User access key | +| `x-pchronicle-secret-key` | User secret key | -缺失、空白或密钥不匹配 MUST 返回 `401`,且 MUST NOT 区分“用户不存在”与“密钥错误”。 +Missing, blank, or mismatched keys MUST return `401`, and MUST NOT distinguish “unknown user” from “bad secret”. +With no catalog headers at all, list MAY return only datasets granted to `user = "*"`; partial headers still MUST `401`. -未授权的 library 名与不存在的 library 名 MUST 都返回 `404`。 +Unauthorized and unknown library names MUST both return `404`. -| 路由 | 父进程 | 响应 | +| Route | Parent | Response | |---|---|---| -| `GET /api/v1/catalog/datasets` | 是 | 该用户可见 library 的 `name`、`uri`、可选 `endpoint`/`region`;**不含**后端密钥 | -| `GET /api/v1/catalog/datasets/{name}` | 是 | 授权时返回完整票,含后端 `access_key` / `secret_key` | -| `GET /api/health` | 是 | 无鉴权 | -| 静态 UI | 是 | 无鉴权 | -| 其余 `/api/*`(含 `GET /api/catalog`,返回当前 Snapshot) | 否,转发 worker | 先鉴权,再按用户 mounts 执行 | +| `GET /api/v1/catalog/datasets` | yes | Libraries visible to the auth user, or public libraries when anonymous: `name`, `uri`, optional `endpoint`/`region`; **no** backend keys | +| `GET /api/v1/catalog/datasets/{name}` | yes | Full ticket with backend `access_key` / `secret_key` when authenticated and authorized; anonymous MUST `401` (public libraries allow key-free list/browse only) | +| `GET /api/health` | yes | No auth | +| Static UI | yes | No auth | +| Other `/api/*` (including `GET /api/catalog` Snapshot) | no, forward to worker | Authenticate, then run with the user's mounts; multi-backend requires `dataset=` | -错误 JSON 沿用 Warehouse 的 `code`、`message`、`request_id`。日志可以包含用户段名、library 名和 `request_id`,MUST NOT 打印用户钥或后端钥。 +Error JSON keeps Warehouse `code`, `message`, `request_id`. Logs MAY include user segment, library name, and `request_id`; MUST NOT print user or backend keys. -`GET /api/v1/catalog/datasets/{name}` 是 CLI 换票接口。拿到票的客户端随后直接打开 `uri`(Dataset path),不再把查询代理回 Directory。 +`GET /api/v1/catalog/datasets/{name}` is the CLI ticket exchange. Clients then open `uri` (the Dataset path) directly and do not proxy queries back through Directory. ## CLI dataset pin -`catalog://` 是 pin **类型**,不是 DatasetLocation 可解析的存储 URI。换票成功后 Dataset 身份是票里的 path,不是 `catalog://…` 本身。 +`catalog://` is a pin **type**, not a storage URI `DatasetLocation` can open. After a successful exchange, Dataset identity is the ticket path, not `catalog://…` itself. ```bash pchronicle dataset pin team catalog://127.0.0.1:8081 --ak USER_AK --sk USER_SK ``` -规范化规则: +Normalization: -- scheme MUST 为 `catalog`; -- host MUST 是环回 IP(如 `127.0.0.1`),MUST 带端口; -- MUST NOT 包含 userinfo、path、query 或 fragment; -- MUST NOT 接受 `--endpoint` / `--region`(那是对象存储参数,来自票而不是 pin)。 +- scheme MUST be `catalog`; +- host MUST be a loopback IP (e.g. `127.0.0.1`) with a port; +- MUST NOT include userinfo, path, query, or fragment; +- MUST NOT accept `--endpoint` / `--region` (those come from the ticket, not the pin). -解析按 pin **类型** 分派,而不是把所有 `@name/suffix` 都做路径拼接: +Resolution dispatches on pin **type**, not path join for every `@name/suffix`: -| 引用 | catalog pin | 普通 URI pin | +| Reference | catalog pin | ordinary URI pin | |---|---|---| -| `@team` / `@team/` | `ls` 列出该用户可访问的 Datasets | 解析为 pin 根 URI | -| `@team/prod` | 向 Directory 领取 library `prod` 的票,打开票中 path | 根 URI 再拼接路径 `prod` | -| `@team/prod/more` | 先领 `prod`,再把 `more` 拼到票的 path 上 | 根 URI 拼接 `prod/more` | +| `@team` / `@team/` | `ls` libraries the user may access | resolve to pin root URI | +| `@team/prod` | fetch ticket for library `prod`, open ticket path | join `prod` onto root URI | +| `@team/prod/more` | fetch `prod`, then join `more` onto ticket path | join `prod/more` onto root URI | -用户 `--ak/--sk` 存入本机 dataset pin 凭据表,与 S3 pin 相同的隔离方式:不出现在 `dataset list` / `dataset show` 的 URI 里。后端密钥 MUST NOT 写入该文件。 +User `--ak/--sk` live in the local dataset pin credential table, isolated like S3 pins: not shown in `dataset list` / `dataset show` URIs. Backend keys MUST NOT be written there. -换到的票缓存在 CLI 进程内(`thread_local`),按 catalog URL、用户 access key 和 library 名索引。长生命周期的 `serve` 进程不使用这份 CLI 缓存;Web 每次请求重新鉴权。进程退出即丢弃缓存。 +Tickets cache in the CLI process (`thread_local`), keyed by catalog URL, user access key, and library name. Long-lived `serve` does not use this CLI cache; Web re-authenticates each request. Cache is dropped on process exit. ## Web -Settings(左侧 **Keys**)保存 catalog 用户钥到 `localStorage`: +Settings (left **Keys**) store catalog user keys in `localStorage`: - `pchronicle.catalog.access_key` - `pchronicle.catalog.secret_key` -浏览器把这两项作为上述 HTTP 头附加到 **发往当前 pChronicle serve 的** `/api/` 请求。这与 Assistant 的 Browser BYOK 相反:Assistant 钥只发给模型端点,catalog 钥必须到达 serve 才能鉴权。 +The browser attaches these as the HTTP headers above on `/api/` requests **to the current pChronicle serve**. This is the opposite of Assistant Browser BYOK: Assistant keys go only to model endpoints; catalog keys must reach serve for auth. -未配置用户钥时,Web MUST NOT 假装本地 Warehouse 已授权;catalog 模式下无头请求在数据面得到 `401`。无 `--catalog-config` 的普通 serve 不要求这些头。 +Without user keys, Web MAY still browse `user = "*"` public libraries; authenticated data-plane requests MUST `401`. Ordinary serve without `--catalog-config` does not require these headers. -查询在 worker 中执行。浏览器不直接持有后端对象存储密钥。 +Queries run in the worker. The browser does not hold backend object-store keys. -## 数据面隔离 +## Data-plane isolation -父进程在数据面中间件中: +In the data-plane middleware the parent: -1. 校验用户钥; -2. 过滤该用户的 library 票; -3. 把 HTTP method、path、query、body 和 mounts 写成 JSON job; -4. spawn 同源二进制 `serve --catalog-query-worker`; -5. 把 stdout 信封还原为 HTTP 响应。 +1. Validates user keys; +2. Filters that user's library tickets; +3. Writes HTTP method, path, query, body, and mounts as a JSON job; +4. Spawns the same binary as `serve --catalog-query-worker`; +5. Reconstructs the HTTP response from the stdout envelope. -Worker 用票构造 `ChronicleServerConfig` mounts,执行与普通 Warehouse 相同的只读路由,然后退出。 +The worker builds `ChronicleServerConfig` mounts from tickets, runs the same read-only Warehouse routes, then exits. -不得把未授权 library 的票放进 job。空授权集合 MUST 表现为 `404`,而不是启动一个空 Warehouse。 +Unauthorized library tickets MUST NOT enter the job. An empty grant set MUST surface as `404`, not an empty Warehouse. -## 被拒绝的方案 +## Rejected alternatives -### 把签发做成 Warehouse HTTP mint +### Warehouse HTTP mint -拒绝。Catalog 头不是公网认证边界;loopback 上无认证的 mint 会把用户钥发给任何能打到端口的本机进程。签发入口是改写 `catalog.toml` 的 CLI。 +Rejected. Catalog headers are not a public auth boundary; unauthenticated mint on loopback would hand user keys to any local process that can hit the port. Issuance is CLI rewriting of the Directory config. -### 独立 `catalog serve` 进程 +### Separate `catalog serve` process -拒绝。第二套 listener、端口和生命周期会与 Warehouse 文档分叉。Catalog 目录流量很小,适合挂在现有 `pchronicle serve` 上。 +Rejected. A second listener, port, and lifecycle would fork Warehouse docs. Directory traffic is small and belongs on existing `pchronicle serve`. -### 父进程打开全部 libraries 再按用户过滤 SQL +### Parent opens every dataset then filters SQL per user -拒绝。DataFusion 与对象存储客户端一旦持有全量后端密钥和 mount,过滤错误就会越权。Web 查询必须在只含授权 mounts 的进程里执行。 +Rejected. Once DataFusion and object-store clients hold all backend keys and mounts, a filter bug is a privilege escalation. Web queries MUST run in a process that only has authorized mounts. -### `fork(2)` 已运行的 Tokio 以“降权” +### `fork(2)` a running Tokio “to drop privilege” -拒绝。在多线程 runtime 上 fork 是未定义行为。使用 `Command` 新进程。 +Rejected. Fork on a multi-threaded runtime is undefined behavior. Use a new `Command` process. -### STS / 短时会话券 +### STS / short-lived session tickets -拒绝。当前目标是本机协作目录,不是云上身份联邦。透传后端密钥给已授权客户端,配置更简单,也与现有 S3 pin 注入 `AWS_*` 的方式一致。 +Rejected. The target is a local collaboration directory, not cloud identity federation. Passing backend keys to authorized clients is simpler and matches existing S3 pin injection into `AWS_*`. -### 把 catalog 做成普通路径拼接 pin +### Treat catalog as ordinary path-join pins -拒绝。`@prod/evals` 对 `s3://bucket` 是路径拼接;对 Directory locator 则是“名字 + library 名”,换票后打开票中 path。混用会让 `@team/prod` 被拼成非法 URI `catalog://127.0.0.1:8081/prod`。 +Rejected. `@prod/evals` on `s3://bucket` is path join; on a Directory locator it is “name + library name”, then open the ticket path. Mixing would produce the illegal URI `catalog://127.0.0.1:8081/prod`. -### 非环回 bind + 把 catalog 头当公网认证 +### Non-loopback bind + catalog headers as public auth -拒绝。Warehouse 仍是本机检查面。打开 `0.0.0.0` 需要独立的认证、TLS 与多租户威胁模型,超出本 RFC。 +Rejected. Warehouse remains a local inspection surface. Binding `0.0.0.0` needs separate auth, TLS, and multi-tenant threat modeling beyond this RFC. -## 兼容性与演进 +## Compatibility and evolution -- 无 `--catalog-config` 时,现有 Dataset 引用、普通 pin 的 `@name/suffix` 路径拼接、以及无鉴权 loopback Warehouse MUST 保持不变。 -- `catalog://` MUST NOT 成为 `DatasetLocation` 可打开的存储 scheme;只有 dataset pin 解析器认识它。 -- 新增 library 字段、鉴权头或 worker 协议属于破坏性变更,需要修订本 RFC。 -- 未来的 STS 或热加载可以作为后续 RFC,不得 silently 改变“透传后端密钥 / 重启生效”的语义。 +- Without `--catalog-config`, existing Dataset references, ordinary pin `@name/suffix` path joins, and unauthenticated loopback Warehouse MUST stay unchanged. +- `catalog://` MUST NOT become an openable `DatasetLocation` storage scheme; only the dataset pin resolver understands it. +- Authoritative config keys are `meta` / `users` / `datasets` / `grants`; legacy `[libraries.*]` or grants embedded in `users.*.datasets` MUST NOT remain normative. +- New Dataset fields, auth headers, or worker protocols are breaking and require revising this RFC. +- Future STS or Dataset hot-reload may be follow-on RFCs and MUST NOT silently change “pass through backend keys / Dataset changes require restart”. -本 RFC 修正架构文档中“loopback Warehouse 完全没有 authentication”的表述:在 `--catalog-config` 下,数据面和 Directory 路由使用用户钥请求头;它仍不是公网多租户服务。 +This RFC corrects architecture language that said loopback Warehouse had no authentication at all: with `--catalog-config`, the data plane and Directory routes use user-key headers, except for public libraries. It is still not a public multi-tenant service. -## 实施状态 +## Implementation status -当前实现覆盖本 RFC 的核心范围: +Current implementation covers the core of this RFC: -- `catalog.toml` 解析与启动期校验; -- `pchronicle serve catalog issue|grant|revoke` 改写 ACL(签发不授权,sk 只打一次 stdout); -- `GET /api/v1/catalog/datasets` 与 `/{name}`; -- `--catalog-config` front-only 父进程与 `--catalog-query-worker`; -- `catalog://` pin、`@team/prod` 换票与进程内票缓存; -- Web `localStorage` 用户钥与数据面请求头。 +- TOML Directory parse and startup validation (`meta` / `users` / `datasets` / `[[grants]]`, including `user = "*"`); +- `pchronicle serve catalog issue|grant|revoke|dataset …` config editors (issue grants nothing; sk printed once on stdout); +- ~3s hot-reload of users/grants; Dataset / backend credential changes rejected with the previous ACL kept; +- `GET /api/v1/catalog/datasets` and `/{name}` (including anonymous public list); +- `--catalog-config` front-only parent and `--catalog-query-worker`; multi-backend narrowed by `dataset=`; +- `catalog://` pins, `@team/prod` ticket exchange, in-process ticket cache; +- Web `localStorage` user keys and data-plane headers. -后续工作: +Follow-ups: -1. 覆盖真实 worker 子进程的集成测试(环境中不得出现未授权 library 的密钥); -2. 评估是否为本地路径 library 提供与 S3 相同的显式审计日志字段; -3. `issue --rotate`:轮换已有用户密钥(当前重名签发直接拒绝)。 +1. Integration tests that cover real worker subprocesses (unauthorized library keys must not appear in the environment); +2. Evaluate explicit audit fields for local-path libraries comparable to S3; +3. `issue --rotate` for existing user keys (duplicate-name issue currently rejects); +4. Whether CLI `dataset add` should formally write multiple S3 backends (hand-written multi-backend is valid today; CLI append still requires matching existing s3 backend identity). diff --git a/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md b/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md index 43f6797ea..de04ab1bb 100644 --- a/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md +++ b/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md @@ -71,20 +71,30 @@ unreadable cache MUST rebuild it and continue with an empty cache. Cache errors MUST NOT make `serve` fail when the authoritative location is still available. -The cache exposes three refresh paths: - -1. A foreground tree request refreshes only the requested level when its - cached view is missing or explicitly stale. -2. A periodic catalog refresh runs every 30 seconds and performs a breadth - first walk of each mount. This publishes shallow levels before deeper - levels and bounds the frontier. -3. A successful refresh updates memory and persistent storage atomically from - the caller's point of view. A failed refresh leaves the previous view - available and records the error for the UI. - -At most one manifest refresh is active per cache instance. This prevents -duplicate timers and concurrent scans from multiplying local or object-store -I/O. The UI browse worker also has a bounded process-level I/O gate. +The Warehouse separates interactive directory browsing from manifest maintenance: + +| Work | Remote I/O | Local result | Admission and backoff | +|---|---|---|---| +| `GET /api/explorer/tree` | One-level LIST on a directory cache miss or refresh; no remote manifest probes | Names enriched from `ManifestCache` with Dataset types and trajectory counts | Foreground AIMD, scoped by endpoint + bucket | +| Background manifest worker | Probe the current prefix and list immediate children, then walk breadth first | Update persistent manifest observations and the browse projection | Independent background AIMD, also scoped by endpoint + bucket | + +Foreground requests use an existing directory projection or manifest observation +immediately. A cold request starts a shallow LIST and waits up to 250 ms before +returning an explicit loading view. Metadata that has not been observed is +partial, not a certified zero count. Background work is bounded to 32 prefixes +per 30-second round and a 10,000-directory frontier; unfinished observations +remain partial. Each refresh job has a 10-second deadline. + +The two workloads do not share in-flight deduplication or failure cooldowns. +A foreground LIST of a prefix can run while that same prefix's background +manifest request is stalled. They may reuse an OpenDAL client, but their AIMD +semaphores, failure counters and cooldowns are independent. Each workload +continues to share admission across paths on the same endpoint and bucket. + +A successful observation updates memory and attempts persistent storage. A +failed observation retains the previous cache. The directory projection stores +navigation data; Dataset identity and aggregate counts come only from the +local manifest observations, scoped to the selected mount and prefix. ## Aggregation semantics diff --git a/docs/src/zh/pchronicle/reference/cli.md b/docs/src/zh/pchronicle/reference/cli.md index f9cc6e8f0..050627f31 100644 --- a/docs/src/zh/pchronicle/reference/cli.md +++ b/docs/src/zh/pchronicle/reference/cli.md @@ -443,8 +443,12 @@ pchronicle serve \ 独立 exec worker 读取;不能与位置参数 Dataset、Gateway 或 Control 同时使用。配合 `dataset pin NAME catalog://127.0.0.1:PORT --ak --sk`。 `pchronicle serve catalog dataset add|remove|list` 与 `issue|grant|revoke` 只改该文件、 不启动 HTTP;`issue` 把用户 sk 只打印一次。改 library、用户或授权后必须重启 serve。 -worker 池最多 8 个进程、32 个正在处理或排队的请求;每个 worker 串行处理请求, -计算等待上限为 60 秒,请求体读取上限为 10 秒。过载返回 503,超时或 IPC 失败会淘汰进程。 +worker 池按并发压力创建进程:优先复用同一权限范围的空闲 worker,全部忙碌时按需扩容, +每个范围最多 4 个、整个服务最多 8 个进程。达到上限后等待任意可用容量, +最多接纳 32 个正在处理或排队的请求。每个 worker 内部仍串行处理,IPC 不交叉。 +空闲超过 120 秒的进程由每 30 秒执行一次的清理任务回收;全局容量不足时可提前回收 +其他范围的空闲进程。排队、启动和执行合计上限为 60 秒,请求体读取上限为 10 秒。 +过载返回 503;执行期间超时或 IPC 失败会淘汰对应进程,排队取消不会中断其他请求。 同一用户、授权范围和后端凭证版本复用 worker 及独立磁盘缓存;修改授权后重启生效。 缓存位于 `PCHRONICLE_CACHE_DIR/workers/`,未设置时使用系统 pchronicle 缓存目录。 子进程不继承父进程的 AWS 环境、profile 或用户主目录配置;登录 AK/SK 用于认证, @@ -576,3 +580,25 @@ pchronicle \ 定位后再写 SQL 见 [发现并查询](../guides/discover-and-query.md),交换见 [导入与导出](../guides/exchange.md),只读服务见 [本地服务 Dataset](../guides/serve.md)。 Snapshot 构造见 [Snapshot 设计](../design/catalog.md)。 + + +### 请求执行诊断 + +**Requests** 标签页展示当前浏览器标签页会话中的近期 API 请求的 request ID、HTTP 结果、执行阶段和耗时。 +左下角 Local / profile 上方的状态提示可打开该页。选择请求或输入 request ID,即可在请求 +尚未结束时查看进度。Worker 排队、启动和执行分别计时;目录浏览与 Runs 查询报告各自的 +执行阶段。`pending` 表示尚未开始,`skipped` 表示未使用该阶段。阶段描述应用操作,并非 +逐条 DNS、TCP 或 S3 请求。目录返回 200 但仍在后台刷新时,会显示相应提示。 + +`GET /api/requests/{request_id}`(也支持 `/api/v1`)返回实时快照: +`request_id`、`method`、`path`、`state`、`elapsed_ms`、`status`、`error`、`note`、 +`phases`(`name`、`state`、`elapsed_ms`),以及可选的嵌套 `worker` 快照。 +该接口由前置服务直接处理,不等待业务 worker。 + +追踪需要显式开启:原请求和诊断查询必须携带同一个随机生成的 32 位十六进制 +`x-pchronicle-observer` 请求头;原请求可通过 `x-request-id` 指定 ID,以便完成前查询。 +UI 自动生成这两个请求头。Observer token 是私有查询凭证,不替代 catalog 身份认证, +其他 token 无法访问该记录。服务端最多保留 512 条记录、最长 10 分钟,仅保存在内存中, +重启即清空。记录不存在、过期或无权访问时均返回 404。UI 保留最近 40 条,每秒查询未完成 +请求;诊断连接失败会单独提示,可手动重试。记录不包含查询参数、请求体或凭据;内部错误 +详情仍需通过 request ID 在服务端日志中定位。 diff --git a/docs/src/zh/pchronicle/reference/query-model.md b/docs/src/zh/pchronicle/reference/query-model.md index a49fe5547..c80fd475c 100644 --- a/docs/src/zh/pchronicle/reference/query-model.md +++ b/docs/src/zh/pchronicle/reference/query-model.md @@ -70,7 +70,7 @@ JOIN dataset.steps s | 选择器 | 含义 | | --- | --- | -| `#content` | `message_value`、`observation` 和 `prompt` | +| `#content` | `message_value`(消息正文;普通文本默认范围) | | `#message` | `message_value` | | `#user` | `source = 'user'` 的 `message_value` | | `#assistant` | `source = 'agent'` 的 `message_value`(`#agent` 是别名) | diff --git a/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md b/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md index f7807c4ad..1d7ec1c25 100644 --- a/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md +++ b/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md @@ -3,7 +3,7 @@ | Field | Value | |---|---| | **Status** | Proposed | -| **Date** | 2026-08-30 | +| **Date** | 2026-09-18 | | **Component** | pChronicle CLI、`pchronicle serve`、pChronicle Web | | **Related** | [RFC-0003 Ownership](0003-pchronicle-ownership.md) · [Warehouse 指南](../pchronicle/guides/serve.md) · [CLI 参考](../pchronicle/reference/cli.md) · [架构](../pchronicle/design/architecture.md) | @@ -15,24 +15,25 @@ Dataset 身份始终是 path(本机路径或 `s3://` / `az://` / `gs://` URI)。Directory 不是第三种 Dataset,也不替代 Snapshot。它只决定调用方可以解析到哪些 path;换票后的 `uri` 才是引擎打开的 Dataset。 -CLI 标志、配置文件和 HTTP 路径为兼容性仍使用 `catalog` 一词(`--catalog-config`、`catalog.toml`、`catalog://`、`/api/v1/catalog/datasets`)。产品与 RFC 口径称 Directory。 +CLI 标志、配置文件和 HTTP 路径为兼容性仍使用 `catalog` 一词(`--catalog-config`、`catalog.toml`、`catalog://`、`/api/v1/catalog/datasets`)。产品与 RFC 口径称 Directory。配置文件是 TOML(扩展名可为 `.toml` / `.yml` 等,内容仍按 TOML 解析)。 规范实现挂在现有 `pchronicle serve --catalog-config` 上,不引入独立 `catalog serve` 进程。 Listener 默认可为 loopback;也允许绑定非环回地址,但部署方 MUST 自行保证网络边界。 -- **Serve 挂载**:`pchronicle serve --catalog-config FILE` MUST 把 `catalog.toml` 中的 **全部** - `[datasets.*]` 挂进 Warehouse(与位置参数挂载等价)。本机 Web / 无用户钥的数据面请求在 - **父进程内**打开这些 mount,不再 front-only。 -- **CLI 配置**:`pchronicle serve catalog dataset add|remove|list` 改写 libraries; - `issue|grant|revoke` 改写用户与授权。 +- **Serve**:父进程只做鉴权、目录/换票与 spawn worker(front-only),MUST NOT 在父进程打开 + `[datasets.*]`。授权范围内的数据面请求由一次性 `--catalog-query-worker` 打开对应 mount。 +- **公开浏览**:`[[grants]]` 中 `user = "*"` 的 Dataset 对匿名调用可见;父进程可用其做浏览缓存, + 不含后端密钥。 +- **CLI 配置**:`pchronicle serve catalog dataset add|remove|list` 改写 datasets; + `issue|grant|revoke` 改写用户与授权。用户/授权热加载;Dataset URI 与后端凭证变更 MUST 重启。 - **Directory 换票**:`@team` 解析为 `catalog://…`;`@team/prod` 换票后客户端打开票里的 path。 - `/api/v1/catalog/datasets` 仍按用户钥过滤可见 library。 + `/api/v1/catalog/datasets` 按用户钥过滤可见 library(无头时仅返回公开库)。 ```text pchronicle serve catalog dataset add --catalog-config catalog.toml prod --uri s3://bucket/prod \ --access-key BACKEND_AK --secret-key BACKEND_SK pchronicle serve catalog issue --catalog-config catalog.toml alice -pchronicle serve catalog grant --catalog-config catalog.toml alice prod evals +pchronicle serve catalog grant --catalog-config catalog.toml alice prod pchronicle serve --catalog-config catalog.toml --listen 127.0.0.1:8081 pchronicle dataset pin team catalog://127.0.0.1:8081 --ak USER_AK --sk USER_SK pchronicle query @team/prod 'SELECT 1' @@ -52,21 +53,23 @@ pchronicle query @team/prod 'SELECT 1' ### 目标 -- 用一份 `catalog.toml` 描述 users、datasets 和 grants。 +- 用一份 Directory 配置描述 `meta`、`users`、`datasets` 和 `[[grants]]`。 - 用 CLI 签发用户钥并改写 ACL:`pchronicle serve catalog issue|grant|revoke` 不启动 HTTP。 - 让 `@name/library` 解析为一条 path(换票后的 `uri`);引擎随后只打开该 path。 - 换票后 CLI 自己访问存储;后端密钥只出现在票和 worker stdin 中,不写入用户 `config.toml`。 -- Web 用用户钥换授权范围,查询只看到该用户的 mounts。 +- Web 用用户钥换授权范围,查询只看到该用户的 mounts;`user = "*"` 的公开库可匿名列出/浏览。 +- 运行中每约 3 秒热加载用户与授权;Dataset 定义与后端凭证变更拒绝热加载,须重启。 - 允许 Warehouse 绑定任意 listen 地址;默认示例仍用 loopback。Catalog 头不是公网认证边界,不可信网络上的暴露由部署方负责。 ### 非目标 - STS、临时凭证轮换、或把用户钥映射成短时 AWS session。 -- 热加载 Dataset 定义和 S3 后端凭证;这类变更仍需重启 serve。 +- 热加载 Dataset URI / endpoint / region / 后端 ak/sk(须重启 serve)。 - 在运行中的 Warehouse 上提供 HTTP 签发接口。 - 提供独立 `catalog serve` 二进制。 - 在已运行的 Tokio runtime 上 `fork(2)`(未定义行为)。 - 把后端对象存储密钥写入本机 dataset pin 配置。 +- v1 细粒度 `permissions` 强制执行(字段可写,语义仍是库成员关系)。 - 改变 Snapshot 协议、SQL schema 或 Gateway/Control 协议。 本 RFC 的 Directory 与打开 path 之后的 **Snapshot**(见 [Snapshot 设计](../pchronicle/design/catalog.md))不是同一对象。Directory 列出授权 path;Snapshot 钉住一条已打开 path 上的 Source 成员与版本。 @@ -104,57 +107,75 @@ Directory 挂在现有 Warehouse listener 上。未传 `--catalog-config` 时, 约束: 1. Listener MAY 绑定非 loopback 地址。本 RFC 不把 catalog 头当作公网认证边界;部署方 MUST 在不可信网络上自行加边界。 -2. 父进程 MUST NOT 打开 `catalog.toml` 中的 datasets。父进程使用空 mount 的 front-only Warehouse。 +2. 父进程 MUST NOT 打开配置中的 datasets。父进程使用空 mount 的 front-only Warehouse;公开库的浏览缓存只持有 path,不把后端钥写入父进程 `AWS_*`。 3. Worker MUST 由 `Command` 启动新进程,MUST NOT `fork(2)` 已运行的 Tokio runtime。 -4. Worker MUST NOT 监听端口、MUST NOT 读取 `catalog.toml`、MUST NOT 读取用户钥。它只消费 stdin 中过滤后的 mounts 和原始请求。 +4. Worker MUST NOT 监听端口、MUST NOT 读取 Directory 配置、MUST NOT 读取用户钥。它只消费 stdin 中过滤后的 mounts 和原始请求。 5. Worker 继承父进程环境(证书、`PATH` 等),但父进程 MUST NOT 预先把 catalog 后端密钥写入 `AWS_*`。Worker 在打开存储前为自己设置该用户票中的后端环境。 -6. 每个 Dataset 可以使用自己的 endpoint、region 和后端 ak/sk;worker 必须按 Dataset ticket 设置对应存储环境。 +6. 每个 `[datasets.*]` MAY 自带 endpoint、region 和后端 ak/sk。同一 worker 进程的进程级 AWS 环境一次只能持有一套凭据;若用户同时授权了不兼容的多个 `s3://` 后端,请求 MUST 带 `dataset=` 显式选中其中一个,否则 MUST 失败。 7. 隐藏 flag `--catalog-query-worker` MUST NOT 出现在用户可见的 `serve --help` 中。 Worker 超时后父进程 MUST 返回 `unavailable`,不得把 stdin 中的密钥写进日志。 ## 配置 -`catalog.toml` 只管理用户、Dataset 和授权关系。它是唯一事实来源;运行时服务配置仍由 `pchronicle serve` 参数提供。配置文件不存在时,Catalog 管理命令会创建一个空 Catalog。 +Directory 配置只管理用户、Dataset 和授权关系。它是唯一事实来源;运行时服务配置仍由 `pchronicle serve` 参数提供。配置文件不存在时,Catalog 管理命令会创建一个空 Catalog(带 `[meta]`)。 + +权威 schema(与当前实现 / 部署样例一致): ```toml [meta] version = 1 revision = 1 -name = "team-catalog" +name = "default" [users.alice] -display_name = "Alice" -status = "active" -access_key = "USER_AK" -secret_key = "USER_SK" +access_key = "pcak_…" +secret_key = "…" + +[datasets.default] +uri = "/data/warehouse" [datasets.prod] -display_name = "Production trajectories" -description = "Production agent trajectories" -status = "active" -uri = "s3://bucket/prod" -endpoint = "http://127.0.0.1:9000" -region = "us-west-2" -access_key = "BACKEND_AK" -secret_key = "BACKEND_SK" +uri = "s3://prod/" +endpoint = "http://s3-a.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_A" +secret_key = "BACKEND_SK_A" + +[datasets.prod2] +uri = "s3://prod" +endpoint = "http://s3-b.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_B" +secret_key = "BACKEND_SK_B" [[grants]] -user = "alice" +user = "*" dataset = "prod" -permissions = ["read", "query", "analyze"] + +[[grants]] +user = "*" +dataset = "prod2" + +[[grants]] +user = "alice" +dataset = "default" +# permissions 可选;v1 忽略细粒度语义,仅表示库成员关系 +# permissions = ["read", "query", "analyze"] ``` 规则: -- `meta.version` 必须为支持的配置版本;每次成功写入 MUST 递增 `meta.revision`。 +- `meta.version` 必须为支持的配置版本;CLI 成功写入后 SHOULD 维护 `meta.revision` / `meta.name`(可缺省)。 - 用户名和 Dataset 名必须是小写 `[A-Za-z_][A-Za-z0-9_]*`。 -- `users.*.access_key` 必须全局唯一;第一版允许明文 `secret_key`。 -- Dataset 的 `uri` 必须是有效的本地、`s3://`、`az://`、`gs://` 或测试存储 URI。 -- 对象存储 Dataset 可以设置 `endpoint`、`region`、`access_key` 和 `secret_key`;本地 Dataset 不需要这些字段。 -- `grants.user` 和 `grants.dataset` 必须分别引用已存在的用户和 Dataset。 -- 同一用户和 Dataset 的 grant 不得重复;权限只能来自 `read`、`query`、`analyze`、`write`、`admin`。 -- 配置文件大小必须有上界;解析或校验失败时服务拒绝启动。 +- `[users.*]` 只含 `access_key` / `secret_key`;`access_key` 必须全局唯一;第一版允许明文 `secret_key`。 +- `[datasets.*]` 必含 `uri`;本地 path 不得设置后端密钥;`s3://` MUST 同时设置 `access_key` 与 `secret_key`,并可设 `endpoint` / `region`。 +- 不同 Dataset MAY 使用不同的 endpoint / region / 后端密钥(见进程模型第 6 条)。 +- `[[grants]]` 必含 `user` 与 `dataset`;`permissions` 可选,v1 不强制执行。 +- `grants.user = "*"`:将该 Dataset 标为公开(匿名可列出/浏览),并在解析时展开给配置中**当前全部**用户;新签发用户在热加载后自动继承。 +- `grants.user` 为具名用户时必须引用已存在用户;`grants.dataset` 必须引用已存在 Dataset。 +- 同一用户与 Dataset 的 grant 不得重复(含 `*` 展开后的冲突)。 +- 配置文件大小必须有上界;解析或校验失败时服务拒绝启动;热加载失败 MUST 保留上一份有效 ACL。 - TOML 是权威配置,后续 SQLite/Postgres 只能作为索引和派生投影。 ## CLI 管理 @@ -171,7 +192,7 @@ pchronicle serve catalog grant --catalog-config FILE NAME DATASET... pchronicle serve catalog revoke --catalog-config FILE NAME DATASET... ``` -`dataset add` 只登记 Dataset,不创建或删除后端数据。`issue` 生成用户 AK/SK(secret 只在本次 stdout 输出)。`grant` / `revoke` 修改 `[[grants]]`。所有写操作 MUST 原子替换文件,失败时保留原文件。`serve --catalog-config` MUST 挂载文件中全部 datasets。 +`dataset add` 只登记 Dataset,不创建或删除后端数据。通过 CLI 追加 `s3://` 时,若文件中已有其它 `s3://`,新条目的 endpoint / region / 后端密钥 MUST 与之完全一致(手写多后端配置仍合法,但 worker 须按第 6 条选库)。`issue` 生成用户 AK/SK(secret 只在本次 stdout 输出),MUST NOT 写入任何 grant。`grant` / `revoke` 修改 `[[grants]]`;NAME 为 `*` 时对**当前全部用户**逐个写入/删除具名 grant(不是写入 `user = "*"` 公开行)。所有写操作 MUST 原子替换文件,失败时保留原文件。 ## HTTP @@ -183,16 +204,17 @@ Directory 路由与 Warehouse 共用 `/api` 与 `/api/v1` 前缀。鉴权头: | `x-pchronicle-secret-key` | 用户 secret key | 缺失、空白或密钥不匹配 MUST 返回 `401`,且 MUST NOT 区分“用户不存在”与“密钥错误”。 +完全无 catalog 头时,列表接口 MAY 只返回 `user = "*"` 的公开 Dataset;带残缺头仍 MUST `401`。 未授权的 library 名与不存在的 library 名 MUST 都返回 `404`。 | 路由 | 父进程 | 响应 | |---|---|---| -| `GET /api/v1/catalog/datasets` | 是 | 该用户可见 library 的 `name`、`uri`、可选 `endpoint`/`region`;**不含**后端密钥 | -| `GET /api/v1/catalog/datasets/{name}` | 是 | 授权时返回完整票,含后端 `access_key` / `secret_key` | +| `GET /api/v1/catalog/datasets` | 是 | 鉴权用户可见 library,或匿名时的公开 library:`name`、`uri`、可选 `endpoint`/`region`;**不含**后端密钥 | +| `GET /api/v1/catalog/datasets/{name}` | 是 | 鉴权且授权时返回完整票,含后端 `access_key` / `secret_key`;匿名 MUST `401`(公开库只允许无密钥列表/浏览) | | `GET /api/health` | 是 | 无鉴权 | | 静态 UI | 是 | 无鉴权 | -| 其余 `/api/*`(含 `GET /api/catalog`,返回当前 Snapshot) | 否,转发 worker | 先鉴权,再按用户 mounts 执行 | +| 其余 `/api/*`(含 `GET /api/catalog`,返回当前 Snapshot) | 否,转发 worker | 先鉴权,再按用户 mounts 执行;多后端时须 `dataset=` | 错误 JSON 沿用 Warehouse 的 `code`、`message`、`request_id`。日志可以包含用户段名、library 名和 `request_id`,MUST NOT 打印用户钥或后端钥。 @@ -234,7 +256,7 @@ Settings(左侧 **Keys**)保存 catalog 用户钥到 `localStorage`: 浏览器把这两项作为上述 HTTP 头附加到 **发往当前 pChronicle serve 的** `/api/` 请求。这与 Assistant 的 Browser BYOK 相反:Assistant 钥只发给模型端点,catalog 钥必须到达 serve 才能鉴权。 -未配置用户钥时,Web MUST NOT 假装本地 Warehouse 已授权;catalog 模式下无头请求在数据面得到 `401`。无 `--catalog-config` 的普通 serve 不要求这些头。 +未配置用户钥时,Web 仍可浏览 `user = "*"` 公开库;需鉴权的数据面请求 MUST `401`。无 `--catalog-config` 的普通 serve 不要求这些头。 查询在 worker 中执行。浏览器不直接持有后端对象存储密钥。 @@ -286,19 +308,21 @@ Worker 用票构造 `ChronicleServerConfig` mounts,执行与普通 Warehouse - 无 `--catalog-config` 时,现有 Dataset 引用、普通 pin 的 `@name/suffix` 路径拼接、以及无鉴权 loopback Warehouse MUST 保持不变。 - `catalog://` MUST NOT 成为 `DatasetLocation` 可打开的存储 scheme;只有 dataset pin 解析器认识它。 -- 新增 library 字段、鉴权头或 worker 协议属于破坏性变更,需要修订本 RFC。 -- 未来的 STS 或热加载可以作为后续 RFC,不得 silently 改变“透传后端密钥 / 重启生效”的语义。 +- 权威配置键为 `meta` / `users` / `datasets` / `grants`;旧式 `[libraries.*]` 或把授权嵌在 `users.*.datasets` 的写法 MUST NOT 再作为规范。 +- 新增 Dataset 字段、鉴权头或 worker 协议属于破坏性变更,需要修订本 RFC。 +- 未来的 STS 或 Dataset 热加载可以作为后续 RFC,不得 silently 改变“透传后端密钥 / Dataset 变更须重启”的语义。 -本 RFC 修正架构文档中“loopback Warehouse 完全没有 authentication”的表述:在 `--catalog-config` 下,数据面和 Directory 路由使用用户钥请求头;它仍不是公网多租户服务。 +本 RFC 修正架构文档中“loopback Warehouse 完全没有 authentication”的表述:在 `--catalog-config` 下,数据面和 Directory 路由使用用户钥请求头;公开库除外。它仍不是公网多租户服务。 ## 实施状态 当前实现覆盖本 RFC 的核心范围: -- `catalog.toml` 解析与启动期校验; -- `pchronicle serve catalog issue|grant|revoke` 改写 ACL(签发不授权,sk 只打一次 stdout); -- `GET /api/v1/catalog/datasets` 与 `/{name}`; -- `--catalog-config` front-only 父进程与 `--catalog-query-worker`; +- TOML Directory 配置解析与启动期校验(`meta` / `users` / `datasets` / `[[grants]]`,含 `user = "*"`); +- `pchronicle serve catalog issue|grant|revoke|dataset …` 改写配置(签发不授权,sk 只打一次 stdout); +- 用户与授权约 3 秒热加载;Dataset / 后端凭证变更拒绝热加载并保留旧 ACL; +- `GET /api/v1/catalog/datasets` 与 `/{name}`(含匿名公开列表); +- `--catalog-config` front-only 父进程与 `--catalog-query-worker`;多后端时按 `dataset=` 收窄 mount; - `catalog://` pin、`@team/prod` 换票与进程内票缓存; - Web `localStorage` 用户钥与数据面请求头。 @@ -306,4 +330,5 @@ Worker 用票构造 `ChronicleServerConfig` mounts,执行与普通 Warehouse 1. 覆盖真实 worker 子进程的集成测试(环境中不得出现未授权 library 的密钥); 2. 评估是否为本地路径 library 提供与 S3 相同的显式审计日志字段; -3. `issue --rotate`:轮换已有用户密钥(当前重名签发直接拒绝)。 +3. `issue --rotate`:轮换已有用户密钥(当前重名签发直接拒绝); +4. 是否让 CLI `dataset add` 正式支持写入多套 S3 后端(当前手写合法,CLI 追加仍要求与已有 s3 后端一致)。 diff --git a/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md b/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md index 43f6797ea..de04ab1bb 100644 --- a/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md +++ b/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md @@ -71,20 +71,30 @@ unreadable cache MUST rebuild it and continue with an empty cache. Cache errors MUST NOT make `serve` fail when the authoritative location is still available. -The cache exposes three refresh paths: - -1. A foreground tree request refreshes only the requested level when its - cached view is missing or explicitly stale. -2. A periodic catalog refresh runs every 30 seconds and performs a breadth - first walk of each mount. This publishes shallow levels before deeper - levels and bounds the frontier. -3. A successful refresh updates memory and persistent storage atomically from - the caller's point of view. A failed refresh leaves the previous view - available and records the error for the UI. - -At most one manifest refresh is active per cache instance. This prevents -duplicate timers and concurrent scans from multiplying local or object-store -I/O. The UI browse worker also has a bounded process-level I/O gate. +The Warehouse separates interactive directory browsing from manifest maintenance: + +| Work | Remote I/O | Local result | Admission and backoff | +|---|---|---|---| +| `GET /api/explorer/tree` | One-level LIST on a directory cache miss or refresh; no remote manifest probes | Names enriched from `ManifestCache` with Dataset types and trajectory counts | Foreground AIMD, scoped by endpoint + bucket | +| Background manifest worker | Probe the current prefix and list immediate children, then walk breadth first | Update persistent manifest observations and the browse projection | Independent background AIMD, also scoped by endpoint + bucket | + +Foreground requests use an existing directory projection or manifest observation +immediately. A cold request starts a shallow LIST and waits up to 250 ms before +returning an explicit loading view. Metadata that has not been observed is +partial, not a certified zero count. Background work is bounded to 32 prefixes +per 30-second round and a 10,000-directory frontier; unfinished observations +remain partial. Each refresh job has a 10-second deadline. + +The two workloads do not share in-flight deduplication or failure cooldowns. +A foreground LIST of a prefix can run while that same prefix's background +manifest request is stalled. They may reuse an OpenDAL client, but their AIMD +semaphores, failure counters and cooldowns are independent. Each workload +continues to share admission across paths on the same endpoint and bucket. + +A successful observation updates memory and attempts persistent storage. A +failed observation retains the previous cache. The directory projection stores +navigation data; Dataset identity and aggregate counts come only from the +local manifest observations, scoped to the selected mount and prefix. ## Aggregation semantics diff --git a/pchronicle-web/Cargo.toml b/pchronicle-web/Cargo.toml index 8b43b8025..e092db4fd 100644 --- a/pchronicle-web/Cargo.toml +++ b/pchronicle-web/Cargo.toml @@ -18,7 +18,7 @@ time = { version = "=0.3.55", features = ["formatting", "parsing"] } urlencoding = "2" wasm-bindgen = "0.2" wasm-bindgen-futures = "0.4" -web-sys = { version = "0.3", features = ["Clipboard", "Document", "DomRect", "Element", "EventTarget", "History", "HtmlElement", "HtmlTextAreaElement", "KeyboardEvent", "Location", "Navigator", "Storage", "UrlSearchParams", "Window"] } +web-sys = { version = "0.3", features = ["Crypto", "Clipboard", "Document", "DomRect", "Element", "EventTarget", "History", "HtmlElement", "HtmlTextAreaElement", "KeyboardEvent", "Location", "Navigator", "Storage", "UrlSearchParams", "Window"] } web-time = "=1.1.0" [features] diff --git a/pchronicle-web/assets/inline-trace.css b/pchronicle-web/assets/inline-trace.css index 08797bd01..51ace0c2e 100644 --- a/pchronicle-web/assets/inline-trace.css +++ b/pchronicle-web/assets/inline-trace.css @@ -1328,4 +1328,4 @@ details.pc2-context-message-body pre { } .pc2-search-mode{display:inline-flex;align-items:center;height:24px;padding:0 6px;border:1px solid #d0d5dd;border-radius:999px;background:#f8fafc;color:#667085;font-size:9px;font-weight:700;white-space:nowrap}.pc2-search-mode.available{border-color:#bbf7d0;background:#f0fdf4;color:#15803d}.pc2-search-mode.unavailable{border-color:#e4e7ec;background:#f8fafc;color:#98a2b3}.pc2-filter-clear{display:inline-flex;align-items:center;justify-content:center;width:22px;height:22px;margin-right:5px;padding:0;border:0;border-radius:999px;background:transparent;color:#667085;font-size:18px;line-height:1;cursor:pointer}.pc2-filter-clear:hover{background:#eaecf0;color:#1d2939}.pc2-filterbar .pc2-filter-clear{height:22px;min-height:22px;padding:0;border:0;background:transparent;color:#667085}.pc2-search-hit{padding:0 2px;border-radius:3px;background:#fef08a;color:inherit;box-shadow:inset 0 -1px #eab308} .pc2-detail-filter-search{min-width:240px!important;height:36px!important;padding:0 10px!important;border-radius:8px!important}.pc2-detail-filter-search input{width:auto!important;height:auto!important;padding:0!important;font-size:11px!important}.pc2-detail-filter-search .pc2-filter-clear{flex:none;margin-right:0} -.pc2-run-search-preview{max-width:100%;margin-top:2px;overflow:hidden;color:#667085;font-size:9px;line-height:1.35;text-overflow:ellipsis;display:-webkit-box;-webkit-box-orient:vertical;-webkit-line-clamp:2}.pc2-run-search-preview .pc2-search-hit{background:#fef08a;color:#344054;font-weight:700} +.pc2-run-search-preview{max-width:100%;margin-top:4px;color:#475467;font-size:11px;line-height:1.45;white-space:normal;overflow-wrap:anywhere}.pc2-run-search-preview .pc2-search-hit{background:#fef08a;color:#344054;font-weight:700} diff --git a/pchronicle-web/assets/path-explorer.css b/pchronicle-web/assets/path-explorer.css index 8933dbe2e..c87dcc912 100644 --- a/pchronicle-web/assets/path-explorer.css +++ b/pchronicle-web/assets/path-explorer.css @@ -1,4 +1,4 @@ -.pc2-runs-layout{min-height:0;display:grid;grid-template-columns:270px minmax(0,1fr);gap:12px;flex:1}.pc2-path-explorer{min-width:0;min-height:0;display:flex;flex-direction:column;border:1px solid #dfe3e8;border-radius:10px;background:#fff;overflow:hidden;box-shadow:0 1px 2px #10182808}.pc2-path-explorer>header{min-height:55px;display:flex;align-items:center;justify-content:space-between;gap:8px;padding:9px 11px;border-bottom:1px solid #eceef1;background:#f8fafc}.pc2-path-explorer>header>div{min-width:0;display:flex;flex-direction:column;gap:2px}.pc2-path-explorer>header strong{color:#1d2939;font-size:11px}.pc2-path-explorer>header div span{color:#98a2b3;font-size:8px}.pc2-path-explorer>header>span{min-width:23px;padding:3px 6px;border:1px solid #d0d5dd;border-radius:999px;background:#fff;color:#667085;font:600 8px ui-monospace,SFMono-Regular,Menlo,monospace;text-align:center}.pc2-path-explorer>footer{padding:8px 10px;border-top:1px solid #eef0f3;background:#fafbfc;color:#98a2b3;font-size:7px;line-height:1.4}.pc2-path-tree{min-height:0;flex:1;padding:7px 6px 14px;overflow:auto;scrollbar-gutter:stable}.pc2-path-all,.pc2-path-row{width:100%;min-height:31px;display:grid;grid-template-columns:20px minmax(0,1fr) 8px 28px;gap:3px;align-items:center;padding:2px 5px;border:0;border-radius:6px;background:transparent;color:#475467;text-align:left}.pc2-path-all{grid-template-columns:20px minmax(0,1fr) 28px;margin-bottom:4px;cursor:pointer}.pc2-path-all:hover,.pc2-path-row:hover{background:#f2f6fc}.pc2-path-all.active,.pc2-path-row.active{background:#eaf2ff;color:#1d4ed8}.pc2-path-row.branch{background:#f7faff}.pc2-path-all strong{overflow:hidden;font-size:9px;text-overflow:ellipsis;white-space:nowrap}.pc2-path-all code,.pc2-path-row>code{color:#98a2b3;font-size:8px;text-align:right}.pc2-path-all.active code,.pc2-path-row.active>code{color:#2563eb}.pc2-path-toggle{width:19px;height:23px;padding:0;border:0;border-radius:4px;background:transparent;color:#98a2b3;font-size:13px;cursor:pointer}.pc2-path-toggle:hover{background:#e4eaf2;color:#475467}.pc2-path-toggle.leaf{display:block}.pc2-path-name{min-width:0;height:27px;display:flex;align-items:center;gap:6px;padding:0;border:0;background:transparent;color:inherit;text-align:left;cursor:pointer}.pc2-path-name>span:last-child{overflow:hidden;font-size:9px;font-weight:600;text-overflow:ellipsis;white-space:nowrap}.pc2-path-icon{position:relative;width:13px;height:11px;flex:none;color:#94a3b8}.pc2-path-icon.root{display:grid;place-items:center;font-size:11px}.pc2-path-icon.folder:before{position:absolute;inset:2px 0 0;border:1px solid currentColor;border-radius:2px;background:#f8fafc;content:""}.pc2-path-icon.folder:after{position:absolute;top:0;left:1px;width:6px;height:4px;border:1px solid currentColor;border-bottom:0;border-radius:2px 2px 0 0;background:#f8fafc;content:""}.pc2-path-icon.run:before{position:absolute;inset:0 1px;border:1px solid currentColor;border-radius:2px;background:#fff;content:""}.pc2-path-icon.run:after{position:absolute;top:3px;left:4px;width:5px;height:1px;background:currentColor;box-shadow:0 3px currentColor;content:""}.pc2-path-row.active .pc2-path-icon{color:#2563eb}.pc2-path-health{width:6px;height:6px;border-radius:50%;background:#98a2b3}.pc2-path-health.good{background:#22c55e}.pc2-path-health.live{background:#3b82f6}.pc2-path-health.bad{background:#ef4444}.pc2-path-children{margin-left:10px;padding-left:5px;border-left:1px solid #dfe3e8}.pc2-path-loading,.pc2-path-empty{min-height:120px;display:flex;align-items:center;justify-content:center;gap:6px;padding:16px;color:#98a2b3;font-size:8px;text-align:center}.pc2-path-loading .spinner{width:12px;height:12px;margin:0}.pc2-path-filter{max-width:180px;height:30px!important;overflow:hidden;padding:0 8px;border-color:#bfdbfe!important;background:#eff6ff!important;color:#1d4ed8!important;font-size:8px!important;text-overflow:ellipsis;white-space:nowrap;cursor:pointer}.pc2-path-all:focus-visible,.pc2-path-toggle:focus-visible,.pc2-path-name:focus-visible{outline:2px solid #60a5fa;outline-offset:1px}@media(max-width:1150px){.pc2-runs-layout{grid-template-columns:230px minmax(0,1fr)}.pc2-filter-search{min-width:240px}}@media(max-width:850px){.pc2-runs-layout{display:flex;overflow:auto;flex-direction:column}.pc2-path-explorer{min-height:220px;max-height:260px}.pc2-path-explorer>footer{display:none}} +.pc2-runs-layout{min-height:0;display:grid;grid-template-columns:270px minmax(0,1fr);gap:12px;flex:1}.pc2-path-explorer{min-width:0;min-height:0;display:flex;flex-direction:column;border:1px solid #dfe3e8;border-radius:10px;background:#fff;overflow:hidden;box-shadow:0 1px 2px #10182808}.pc2-path-explorer>header{min-height:55px;display:flex;align-items:center;justify-content:space-between;gap:8px;padding:9px 11px;border-bottom:1px solid #eceef1;background:#f8fafc}.pc2-path-explorer>header>div{min-width:0;display:flex;flex-direction:column;gap:2px}.pc2-path-explorer>header strong{color:#1d2939;font-size:11px}.pc2-path-explorer>header div span{color:#98a2b3;font-size:8px}.pc2-path-explorer>header>span{min-width:23px;padding:3px 6px;border:1px solid #d0d5dd;border-radius:999px;background:#fff;color:#667085;font:600 8px ui-monospace,SFMono-Regular,Menlo,monospace;text-align:center}.pc2-path-explorer>footer{padding:8px 10px;border-top:1px solid #eef0f3;background:#fafbfc;color:#98a2b3;font-size:7px;line-height:1.4}.pc2-path-explorer>footer.paged{display:flex;align-items:center;justify-content:space-between;gap:8px}.pc2-path-page{min-width:0;display:flex;align-items:center;gap:5px;color:#667085;font-size:8px}.pc2-path-page select{max-width:132px;height:22px;padding:0 4px;border:1px solid #d0d5dd;border-radius:5px;background:#fff;color:#344054;font:600 8px ui-monospace,SFMono-Regular,Menlo,monospace}.pc2-path-page select:focus-visible{outline:2px solid #60a5fa;outline-offset:1px}.pc2-path-tree{min-height:0;flex:1;padding:7px 6px 14px;overflow:auto;scrollbar-gutter:stable}.pc2-path-all,.pc2-path-row{width:100%;min-height:31px;display:grid;grid-template-columns:20px minmax(0,1fr) 8px 28px;gap:3px;align-items:center;padding:2px 5px;border:0;border-radius:6px;background:transparent;color:#475467;text-align:left}.pc2-path-all{grid-template-columns:20px minmax(0,1fr) 28px;margin-bottom:4px;cursor:pointer}.pc2-path-all:hover,.pc2-path-row:hover{background:#f2f6fc}.pc2-path-all.active,.pc2-path-row.active{background:#eaf2ff;color:#1d4ed8}.pc2-path-row.branch{background:#f7faff}.pc2-path-all strong{overflow:hidden;font-size:9px;text-overflow:ellipsis;white-space:nowrap}.pc2-path-all code,.pc2-path-row>code{color:#98a2b3;font-size:8px;text-align:right}.pc2-path-all.active code,.pc2-path-row.active>code{color:#2563eb}.pc2-path-toggle{width:19px;height:23px;padding:0;border:0;border-radius:4px;background:transparent;color:#98a2b3;font-size:13px;cursor:pointer}.pc2-path-toggle:hover{background:#e4eaf2;color:#475467}.pc2-path-toggle.leaf{display:block}.pc2-path-name{min-width:0;height:27px;display:flex;align-items:center;gap:6px;padding:0;border:0;background:transparent;color:inherit;text-align:left;cursor:pointer}.pc2-path-name>span:last-child{overflow:hidden;font-size:9px;font-weight:600;text-overflow:ellipsis;white-space:nowrap}.pc2-path-icon{position:relative;width:13px;height:11px;flex:none;color:#94a3b8}.pc2-path-icon.root{display:grid;place-items:center;font-size:11px}.pc2-path-icon.folder:before{position:absolute;inset:2px 0 0;border:1px solid currentColor;border-radius:2px;background:#f8fafc;content:""}.pc2-path-icon.folder:after{position:absolute;top:0;left:1px;width:6px;height:4px;border:1px solid currentColor;border-bottom:0;border-radius:2px 2px 0 0;background:#f8fafc;content:""}.pc2-path-icon.run:before{position:absolute;inset:0 1px;border:1px solid currentColor;border-radius:2px;background:#fff;content:""}.pc2-path-icon.run:after{position:absolute;top:3px;left:4px;width:5px;height:1px;background:currentColor;box-shadow:0 3px currentColor;content:""}.pc2-path-row.active .pc2-path-icon{color:#2563eb}.pc2-path-health{width:6px;height:6px;border-radius:50%;background:#98a2b3}.pc2-path-health.good{background:#22c55e}.pc2-path-health.live{background:#3b82f6}.pc2-path-health.bad{background:#ef4444}.pc2-path-children{margin-left:10px;padding-left:5px;border-left:1px solid #dfe3e8}.pc2-path-loading,.pc2-path-empty{min-height:120px;display:flex;align-items:center;justify-content:center;gap:6px;padding:16px;color:#98a2b3;font-size:8px;text-align:center}.pc2-path-loading .spinner{width:12px;height:12px;margin:0}.pc2-path-filter{max-width:180px;height:30px!important;overflow:hidden;padding:0 8px;border-color:#bfdbfe!important;background:#eff6ff!important;color:#1d4ed8!important;font-size:8px!important;text-overflow:ellipsis;white-space:nowrap;cursor:pointer}.pc2-path-all:focus-visible,.pc2-path-toggle:focus-visible,.pc2-path-name:focus-visible{outline:2px solid #60a5fa;outline-offset:1px}@media(max-width:1150px){.pc2-runs-layout{grid-template-columns:230px minmax(0,1fr)}.pc2-filter-search{min-width:240px}}@media(max-width:850px){.pc2-runs-layout{display:flex;overflow:auto;flex-direction:column}.pc2-path-explorer{min-height:220px;max-height:260px}.pc2-path-explorer>footer{display:none}} .pc2-path-row{grid-template-columns:20px minmax(0,1fr) 24px 8px 28px} .pc2-path-explorer>header>.pc2-path-view-toggle{flex:0 0 auto;display:flex;flex-direction:row;gap:2px;padding:2px;border:1px solid #d0d5dd;border-radius:6px;background:#fff}.pc2-path-view-toggle button{height:22px;padding:0 5px;border:0;border-radius:4px;background:transparent;color:#667085;font-size:8px;cursor:pointer}.pc2-path-view-toggle button.active{background:#eaf2ff;color:#1d4ed8;font-weight:700}.pc2-path-view-toggle button:focus-visible{outline:2px solid #60a5fa;outline-offset:1px} .pc2-path-explorer>header>div:first-child{flex:1} diff --git a/pchronicle-web/assets/workbench.css b/pchronicle-web/assets/workbench.css index 45f22ce1e..850c9a782 100644 --- a/pchronicle-web/assets/workbench.css +++ b/pchronicle-web/assets/workbench.css @@ -150,3 +150,8 @@ .pc2-shell .rail-secondary .rail-button{color:#b8cbe4} @media(max-width:850px){.pc2-shell .rail-secondary{padding-top:0;border-top:0}} .pc2-form select{width:100%;height:42px;padding:0 12px;border:1px solid #cbd5e1;border-radius:9px;background:#fff;color:#344054;font:inherit}.pc2-identity-list{display:flex;flex-direction:column;gap:8px}.pc2-identity-list .button{width:100%;text-align:left}.pc2-settings .button.danger{border-color:#fecaca;color:#b42318;background:#fff}.pc2-settings .button.danger:hover{background:#fff1f0} + +/* Request diagnostics uses the existing workspace rail and scroll surface. */ +.request-indicator{display:flex;align-items:center;gap:7px;width:100%;padding:10px 14px;border:0;border-radius:8px;background:#ffffff0b;color:#cbd5e1;text-align:left;font-size:12px;cursor:pointer}.request-indicator .spinner{margin:0;width:12px;height:12px;flex:none}.requests-panel{padding:24px;overflow:auto;height:100%;box-sizing:border-box}.requests-panel header p{color:var(--muted)}.requests-panel form{display:flex;gap:8px;margin:16px 0}.requests-panel input{min-width:260px;padding:8px;border:1px solid var(--border);border-radius:6px}.requests-grid{display:grid;grid-template-columns:minmax(240px,1fr) minmax(400px,2fr);gap:20px;align-items:start}.requests-grid aside{max-height:65vh;overflow:auto}.request-row{display:flex;flex-direction:column;gap:4px;width:100%;padding:12px;border:1px solid var(--border);background:white;text-align:left;cursor:pointer;overflow-wrap:anywhere}.request-row span{font-size:11px;color:var(--muted)}.request-row-summary{color:#344054!important;font-weight:600}.request-row-id{font:500 10px ui-monospace,SFMono-Regular,Menlo,monospace}.request-row.active{background:#eff6ff;border-color:#2563eb}.requests-grid article{min-width:0;max-height:65vh;padding:20px;border:1px solid var(--border);border-radius:10px;background:white;overflow:auto;overflow-wrap:anywhere}.request-phases{width:100%;border-collapse:collapse;font-size:13px}.request-phases caption{text-align:left;font-weight:600;padding:14px 0 8px}.request-phases td,.request-phases th{text-align:left;padding:10px 6px;border-bottom:1px solid var(--border)}.request-error{color:#b42318}.request-indicator:focus-visible,.request-row:focus-visible{outline:2px solid #60a5fa;outline-offset:2px}@media(max-width:850px){.requests-grid{grid-template-columns:1fr}.requests-grid aside{max-height:220px}.requests-grid article{max-height:none}} + +.request-indicator{flex-wrap:wrap;box-sizing:border-box}.request-indicator small{width:100%;font-size:10px;overflow-wrap:anywhere}.request-row strong{font-size:13px}.requests-panel input:focus-visible{outline:2px solid #2563eb;outline-offset:2px}@media(max-width:850px){.request-indicator{font-size:9px;padding:8px 2px;overflow-wrap:anywhere}.request-indicator small{display:none}.requests-panel input{min-width:0;width:100%}} diff --git a/pchronicle-web/src/analysis_agent.rs b/pchronicle-web/src/analysis_agent.rs index 3549fea53..656557820 100644 --- a/pchronicle-web/src/analysis_agent.rs +++ b/pchronicle-web/src/analysis_agent.rs @@ -1287,6 +1287,7 @@ mod tests { fn plan_prompt_sends_only_approved_catalog_and_scope_context() { let mut catalog = catalog(); catalog.datasets = vec![QueryDatasetSummary { + browse: None, name: "private-dataset".into(), uri: "s3://secret-bucket/?token=private".into(), ready_sources: 17, diff --git a/pchronicle-web/src/api.rs b/pchronicle-web/src/api.rs index dc50a679f..fb5a7704f 100644 --- a/pchronicle-web/src/api.rs +++ b/pchronicle-web/src/api.rs @@ -9,6 +9,7 @@ use crate::model::{ PhysicalSource, QueryCatalog, QueryEvidence, RunAnalysis, RunPage, RunSummary, TurnDetail, TurnPage, }; +use crate::requests::TrackedSend; use gloo_net::http::{Request, RequestBuilder, Response}; use serde::de::DeserializeOwned; use serde_json::json; @@ -164,7 +165,12 @@ pub async fn explorer_runs( urlencoding::encode(path), urlencoding::encode(file), ); - json_checked(with_catalog_headers(Request::get(&url)).send().await).await + json_checked( + with_catalog_headers(Request::get(&url)) + .send_tracked() + .await, + ) + .await } pub async fn explorer_tree(dataset: &str, prefix: &str) -> Result { @@ -173,7 +179,12 @@ pub async fn explorer_tree(dataset: &str, prefix: &str) -> Result Result { - json_checked( - with_catalog_headers(Request::get(&format!("/api/explorer/run?{}", run.query()))) - .send() - .await, - ) - .await + json_checked(Request::get(&url).send_tracked().await).await } pub async fn compact_record(run: &RunSummary) -> Result { @@ -203,20 +205,40 @@ pub async fn compact_record(run: &RunSummary) -> Result Result { +pub async fn turns( + run: &RunSummary, + q: &str, + source: &str, + include_analysis: bool, +) -> Result { let url = format!( - "/api/explorer/turns?{}&q={}&source={}&offset=0&limit=500", + "/api/explorer/turns?{}&q={}&source={}&include_analysis={include_analysis}", run.query(), urlencoding::encode(q), urlencoding::encode(source), ); - json_checked(with_catalog_headers(Request::get(&url)).send().await).await + json_checked( + with_catalog_headers(Request::get(&url)) + .send_tracked() + .await, + ) + .await +} + +pub async fn run_analysis(run: &RunSummary) -> Result { + let url = format!("/api/explorer/run?{}", run.query()); + json_checked( + with_catalog_headers(Request::get(&url)) + .send_tracked() + .await, + ) + .await } pub async fn turn_detail(run: &RunSummary, turn_id: i64) -> Result { @@ -225,7 +247,7 @@ pub async fn turn_detail(run: &RunSummary, turn_id: i64) -> Result Result { json_checked( with_catalog_headers(Request::get("/api/query/tables?ui=true")) - .send() + .send_tracked() .await, ) .await } pub async fn ui_config() -> Result { - json_checked(Request::get("/api/ui").send().await).await + json_checked(Request::get("/api/ui").send_tracked().await).await } pub async fn refresh_catalog() -> Result<(), ApiFailure> { send_checked( with_catalog_headers(Request::post("/api/catalog")) - .send() + .send_tracked() .await, ) .await?; @@ -303,7 +325,7 @@ pub async fn refresh_catalog() -> Result<(), ApiFailure> { pub async fn physical_sources() -> Result, ApiFailure> { json_checked( with_catalog_headers(Request::get("/api/physical/sources")) - .send() + .send_tracked() .await, ) .await @@ -315,7 +337,12 @@ pub async fn physical_layout(dataset: &str, file: &str) -> Result, pub name: String, pub uri: String, pub ready_sources: usize, pub error_sources: usize, } -#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize)] pub struct CatalogTree { #[serde(default)] pub browse: Option, @@ -588,6 +590,8 @@ pub struct TurnSummary { #[derive(Clone, Debug, PartialEq, Deserialize)] pub struct TurnPage { + #[serde(default)] + pub analysis: Option, pub snapshot: PageSnapshot, pub records: Vec, #[serde(default)] @@ -773,12 +777,14 @@ mod tests { path_column: "_file_".into(), datasets: vec![ QueryDatasetSummary { + browse: None, name: "atif".into(), uri: "atif".into(), ready_sources: 1, error_sources: 0, }, QueryDatasetSummary { + browse: None, name: "actf".into(), uri: "actf".into(), ready_sources: 1, @@ -842,10 +848,58 @@ mod tests { } /// Directory observations are independent of the query's pinned revisions. -#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize)] pub struct BrowseStatus { + #[serde(default)] + pub partial: bool, + #[serde(default)] + pub state: String, pub observed_at: i64, pub stale: bool, pub refreshing: bool, pub last_error: Option, } + +impl QueryDatasetSummary { + pub fn label(&self) -> String { + let suffix = self + .browse + .as_ref() + .map(|status| { + if status.observed_at == 0 && status.last_error.is_some() { + " · Unavailable" + } else if status.observed_at == 0 { + " · Loading…" + } else if status.partial { + " · Partial" + } else if status.stale { + " · Cached" + } else { + "" + } + }) + .unwrap_or(""); + format!("{}{suffix}", self.name) + } +} + +#[cfg(test)] +mod browse_status_tests { + use super::*; + #[test] + fn distinguish_unknown_counts_from_an_empty_loaded_dataset() { + let mut dataset: QueryDatasetSummary = serde_json::from_value(serde_json::json!({ + "name": "mount", "uri": "s3://bucket", "ready_sources": 0, "error_sources": 0, + "browse": {"observed_at": 0, "refreshing": true, "stale": true, "last_error": null} + })) + .unwrap(); + assert_eq!(dataset.label(), "mount · Loading…"); + dataset.browse.as_mut().unwrap().observed_at = 1; + dataset.browse.as_mut().unwrap().stale = false; + assert_eq!(dataset.label(), "mount"); + dataset.browse.as_mut().unwrap().partial = true; + assert_eq!(dataset.label(), "mount · Partial"); + dataset.browse = None; + assert_eq!(dataset.label(), "mount"); + } +} diff --git a/pchronicle-web/src/requests.rs b/pchronicle-web/src/requests.rs new file mode 100644 index 000000000..be1b7790f --- /dev/null +++ b/pchronicle-web/src/requests.rs @@ -0,0 +1,439 @@ +use dioxus::prelude::*; +use gloo_net::http::{Request, RequestBuilder, Response}; +use gloo_timers::future::TimeoutFuture; +use serde::Deserialize; +use web_time::Instant; + +const OBSERVER: &str = "x-pchronicle-observer"; +const LIMIT: usize = 40; +#[derive(Clone, Debug, PartialEq, Deserialize)] +pub struct Phase { + pub name: String, + pub state: String, + pub elapsed_ms: u64, +} +#[derive(Clone, Debug, PartialEq, Deserialize)] +pub struct Snapshot { + pub request_id: String, + pub method: String, + pub path: String, + pub state: String, + pub elapsed_ms: u64, + pub status: Option, + pub error: Option, + pub note: Option, + pub phases: Vec, + pub worker: Option>, +} +#[derive(Clone, PartialEq)] +pub struct Entry { + pub id: String, + pub method: String, + pub path: String, + pub transport: String, + pub started: Instant, + pub finished_ms: Option, + pub snapshot: Option, + pub diagnostic_error: Option, +} +pub static REQUESTS: GlobalSignal> = Signal::global(Vec::new); + +fn random_id() -> Option { + let mut bytes = [0u8; 16]; + web_sys::window()? + .crypto() + .ok()? + .get_random_values_with_u8_array(&mut bytes) + .ok()?; + Some(bytes.iter().map(|b| format!("{b:02x}")).collect()) +} +fn observer() -> Option { + let storage = web_sys::window()?.session_storage().ok()??; + if let Ok(Some(token)) = storage.get_item("pchronicle.request_observer") { + return Some(token); + } + let token = random_id()?; + storage + .set_item("pchronicle.request_observer", &token) + .ok()?; + Some(token) +} +fn update(id: &str, apply: impl FnOnce(&mut Entry)) { + if let Some(entry) = REQUESTS.write().iter_mut().find(|e| e.id == id) { + apply(entry); + } +} +struct CancelOnDrop(String); +impl Drop for CancelOnDrop { + fn drop(&mut self) { + update(&self.0, |e| { + if e.transport == "running" { + e.transport = "cancelled".into(); + e.finished_ms = Some(e.started.elapsed().as_millis() as u64); + } + }); + } +} + +pub trait TrackedSend { + async fn send_tracked(self) -> Result; +} +impl TrackedSend for RequestBuilder { + async fn send_tracked(self) -> Result { + self.build()?.send_tracked().await + } +} +impl TrackedSend for Request { + async fn send_tracked(self) -> Result { + let (Some(token), Some(id)) = (observer(), random_id()) else { + return self.send().await; + }; + self.headers().set("x-request-id", &id); + self.headers().set(OBSERVER, &token); + let url = self.url(); + let path = format!( + "/api/{}", + url.split("/api/") + .nth(1) + .unwrap_or("") + .split('?') + .next() + .unwrap_or("") + ); + { + let mut entries = REQUESTS.write(); + // Keep failures available while routine catalog polling continues. + if entries.len() >= LIMIT { + let index = entries + .iter() + .position(|e| e.transport == "completed") + .unwrap_or(0); + entries.remove(index); + } + entries.push(Entry { + id: id.clone(), + method: self.method().to_string(), + path, + transport: "running".into(), + started: Instant::now(), + finished_ms: None, + snapshot: None, + diagnostic_error: None, + }); + } + let _cancel = CancelOnDrop(id.clone()); + let result = self.send().await; + update(&id, |e| { + e.finished_ms = Some(e.started.elapsed().as_millis() as u64); + e.transport = match &result { + Ok(r) if r.ok() => "completed", + _ => "failed", + } + .into(); + if let Err(error) = &result { + e.diagnostic_error = Some(format!("Could not reach the server: {error}")); + } + }); + result + } +} + +fn selected_entry<'a>(entries: &'a [Entry], selected: &str) -> Option<&'a Entry> { + if selected.is_empty() { + entries.last() + } else { + entries.iter().find(|entry| entry.id == selected) + } +} + +fn needs_diagnostics(entry: &Entry) -> bool { + entry.diagnostic_error.is_none() && entry.snapshot.as_ref().is_none_or(|s| s.state == "running") +} + +fn use_request_polling(selected: Signal) { + use_future(move || async move { + loop { + // Only the visible inspector needs server progress. Transport status + // for the sidebar is already maintained by send_tracked. + if web_sys::window() + .and_then(|window| window.document()) + .is_none_or(|document| document.hidden()) + { + TimeoutFuture::new(1000).await; + continue; + } + let pending = { + let entries = REQUESTS.peek(); + let selected = selected.peek(); + selected_entry(&entries, &selected) + .filter(|entry| needs_diagnostics(entry)) + .map(|entry| (entry.id.clone(), entry.started)) + }; + if let Some((id, started)) = pending + && let Some(token) = observer() + { + let fetch = async { + let response = Request::get(&format!("/api/requests/{id}")) + .header(OBSERVER, &token) + .send() + .await + .map_err(|_| "Diagnostics connection failed".to_string())?; + if response.status() == 404 && started.elapsed().as_secs() < 10 { + return Ok(None); + } + if !response.ok() { + return Err("Diagnostics expired or are unavailable; the original request may still be running".into()); + } + response + .json::() + .await + .map(Some) + .map_err(|_| "Invalid diagnostics response".into()) + }; + let result = match futures_util::future::select( + Box::pin(fetch), + Box::pin(TimeoutFuture::new(3000)), + ) + .await + { + futures_util::future::Either::Left((r, _)) => r, + _ => Err("Diagnostics timed out; server status is unknown".into()), + }; + update(&id, |e| match result { + Ok(Some(s)) => { + if e.path == "Lookup" { + e.method = s.method.clone(); + e.path = s.path.clone(); + } + e.snapshot = Some(s); + } + Ok(None) => {} + Err(error) => e.diagnostic_error = Some(error), + }); + } + TimeoutFuture::new(1000).await; + } + }); +} +fn label(name: &str) -> &str { + match name { + "authentication" => "Check catalog identity", + "worker_queue" => "Wait for worker", + "worker_start" => "Start worker", + "worker_execution" => "Execute in worker", + "execution" => "Accept request", + "browse_cache" => "Read directory cache", + "manifest_summary" => "Read local manifest summaries", + "directory_wait" => "Wait for directory listing", + "query_queue" => "Wait for query slot / shared result", + "source_metadata" => "Resolve source metadata", + "storage_read" => "Read storage", + "query" => "Build / execute query", + "catalog_wait" => "Wait for catalog refresh", + "response" => "Prepare response", + _ => name, + } +} + +fn entry_elapsed_ms(entry: &Entry) -> u64 { + entry.finished_ms.unwrap_or_else(|| { + entry + .snapshot + .as_ref() + .map(|snapshot| snapshot.elapsed_ms) + .unwrap_or_else(|| entry.started.elapsed().as_millis() as u64) + }) +} + +fn entry_summary(entry: &Entry) -> String { + let mut parts = vec![entry.transport.clone()]; + if let Some(status) = entry.snapshot.as_ref().and_then(|snapshot| snapshot.status) { + parts.push(format!("HTTP {status}")); + } + parts.push(format!("{} ms", entry_elapsed_ms(entry))); + parts.join(" · ") +} + +fn short_request_id(id: &str) -> String { + if id.len() <= 16 { + id.to_owned() + } else { + format!("{}…{}", &id[..8], &id[id.len().saturating_sub(4)..]) + } +} +#[component] +pub fn RequestIndicator(on_open: EventHandler<()>) -> Element { + let entries = REQUESTS.read(); + let running = entries + .iter() + .filter(|e| { + e.transport == "running" + || (e.transport == "unknown" + && e.snapshot.as_ref().is_some_and(|s| s.state == "running")) + }) + .count(); + let failed = entries + .iter() + .filter(|e| { + e.transport == "failed" || e.snapshot.as_ref().is_some_and(|s| s.state == "failed") + }) + .count(); + let current = entries + .iter() + .rev() + .find(|e| { + e.transport == "running" + || (e.transport == "unknown" + && e.snapshot.as_ref().is_some_and(|s| s.state == "running")) + }) + .and_then(|e| e.snapshot.as_ref()) + .and_then(|s| { + let s = s.worker.as_deref().unwrap_or(s); + s.phases + .iter() + .find(|p| p.state == "running") + .map(|p| format!("{} · {} ms", label(&p.name), p.elapsed_ms)) + }); + rsx! { button { class:"request-indicator", onclick:move |_|on_open.call(()), + if running>0 { span { class:"spinner" } "{running} running" if let Some(stage)=current { small { "{stage}" } } } + else if failed>0 { "{failed} failed · View requests" } + else { "Requests · Idle" } + } } +} +#[component] +fn Phases(snapshot: Snapshot) -> Element { + rsx! { + if let Some(note)=&snapshot.note { p { role:"status", "{note}" } } + table { class:"request-phases", caption { "Execution stages" } + thead { tr { th { "Stage" } th { "Status" } th { "Elapsed" } } } + tbody { for phase in &snapshot.phases { tr { key:"{phase.name}", + td { "{label(&phase.name)}" } td { "{phase.state}" } td { "{phase.elapsed_ms} ms" } + } } } + } + if let Some(worker)=snapshot.worker { h3 { "Worker execution" } Phases { snapshot:*worker } } + } +} +#[component] +fn RequestRow(entry: Entry, active: bool, on_select: EventHandler) -> Element { + let summary = entry_summary(&entry); + let short_id = short_request_id(&entry.id); + let id = entry.id.clone(); + rsx! { + button { + class: if active { "request-row active" } else { "request-row" }, + onclick: move |_| on_select.call(id.clone()), + strong { "{entry.method} {entry.path}" } + span { class: "request-row-summary", "{summary}" } + span { class: "request-row-id", title: "{entry.id}", "{short_id}" } + } + } +} + +#[component] +pub fn RequestsPanel() -> Element { + let mut selected = use_signal(String::new); + let mut lookup = use_signal(String::new); + use_request_polling(selected); + let entries = REQUESTS.read().clone(); + let active = selected_entry(&entries, &selected()).cloned(); + rsx! { section { class:"requests-panel", + header { h1 { "Requests" } p { "Inspect this browser’s recent requests, execution stages and failures. Server history is retained for up to 10 minutes." } + form { onsubmit:move |event| { + event.prevent_default(); + let id=lookup().trim().to_owned(); + if id.is_empty() || id.len()>64 || !id.bytes().all(|b|b.is_ascii_alphanumeric() || b==b'-' || b==b'_') { return; } + selected.set(id.clone()); + let mut entries=REQUESTS.write(); + if let Some(entry)=entries.iter_mut().find(|e|e.id==id) { entry.diagnostic_error=None; } + else { + if entries.len()>=LIMIT { entries.remove(0); } + entries.push(Entry { id,method:"GET".into(),path:"Lookup".into(),transport:"unknown".into(),started:Instant::now()-std::time::Duration::from_secs(10),finished_ms:None,snapshot:None,diagnostic_error:None }); + } + }, + input { aria_label:"Request ID", placeholder:"Find by request ID", value:"{lookup}", oninput:move |e|lookup.set(e.value()) } + button { class:"button", r#type:"submit", "Find" } + } + } + if !selected().is_empty() && !entries.iter().any(|e|e.id==selected()) { p { role:"status", "This request is not in this browser’s recent history." } } + div { class:"requests-grid", + aside { aria_label:"Recent requests", + for entry in entries.iter().rev() { + RequestRow { + key: "{entry.id}", + entry: entry.clone(), + active: active.as_ref().is_some_and(|e| e.id == entry.id), + on_select: move |id| selected.set(id), + } + } + } + article { + if let Some(entry)=active { + h2 { "{entry.method} {entry.path}" } code { "{entry.id}" } + p { "Browser request: {entry.transport}" } + if let Some(error)=entry.diagnostic_error { p { class:"request-error", role:"alert", "{error}" } + button { class:"button", onclick:{let id=entry.id.clone();move |_|update(&id,|e|e.diagnostic_error=None)}, "Retry diagnostics" } + } + if let Some(snapshot)=entry.snapshot { + p { role:"status", "Server: {snapshot.state} · {snapshot.elapsed_ms} ms" } + if let Some(status)=snapshot.status { p { "HTTP {status}" } } + if let Some(ref error)=snapshot.error { p { class:"request-error", role:"alert", "{error}" } } + Phases { snapshot } + } else { p { role:"status", "Waiting for server diagnostics…" } } + } else { p { "Browse a dataset or open Runs to inspect a request." } } + } + } + } } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn diagnostics_stop_on_terminal_snapshot_or_error() { + let mut entry = Entry { + id: "test".into(), + method: "GET".into(), + path: "test".into(), + transport: "completed".into(), + started: Instant::now(), + finished_ms: None, + snapshot: None, + diagnostic_error: None, + }; + let entries = (0..40) + .map(|id| Entry { + id: id.to_string(), + ..entry.clone() + }) + .collect::>(); + assert_eq!(selected_entry(&entries, "").unwrap().id, "39"); + assert_eq!(selected_entry(&entries, "5").unwrap().id, "5"); + assert!(selected_entry(&entries, "missing").is_none()); + assert!(selected_entry(&[], "").is_none()); + entry.finished_ms = Some(12); + assert_eq!(entry_elapsed_ms(&entry), 12); + // Opening the inspector can retrieve diagnostics for a completed request. + assert!(needs_diagnostics(&entry)); + entry.snapshot = Some(Snapshot { + request_id: "test".into(), + method: "GET".into(), + path: "test".into(), + state: "running".into(), + elapsed_ms: 0, + status: None, + error: None, + note: None, + phases: vec![], + worker: None, + }); + assert!(needs_diagnostics(&entry)); + for state in ["completed", "failed", "cancelled"] { + entry.snapshot.as_mut().unwrap().state = state.into(); + assert!(!needs_diagnostics(&entry)); + } + entry.snapshot = None; + entry.diagnostic_error = Some("unavailable".into()); + assert!(!needs_diagnostics(&entry)); + } +} diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 9598a9148..2f230594f 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -28,14 +28,14 @@ use crate::llm_settings::LlmSettings; #[cfg(test)] use crate::model::RunSearchStatus; use crate::model::{ - CatalogTree, CompactRecordDetail, DimensionAggregate, HistogramBucket, PageSnapshot, - QueryCatalog, QueryDatasetSummary, RunAnalysis, RunExplorerItem, RunPage, RunSummary, - ToolAggregate, TurnDetail, TurnSearchStatus, TurnSummary, + BrowseStatus, CatalogTree, CompactRecordDetail, DimensionAggregate, HistogramBucket, + PageSnapshot, QueryCatalog, QueryDatasetSummary, RunAnalysis, RunExplorerItem, RunPage, + RunSummary, ToolAggregate, TurnDetail, TurnSearchStatus, TurnSummary, }; use crate::notice::{ErrorNotice, WorkspaceNotice, workspace_notice}; use crate::terminology::{ANALYSIS, ASSISTANT, DATASETS, RUNS, STEPS, STORAGE, TIMELINE}; -const SEARCH_DEBOUNCE_MS: u32 = 1_000; +const STEP_SEARCH_DEBOUNCE_MS: u32 = 1_000; const CATALOG_REFRESH_MS: u32 = 5_000; fn evidence_notice(turn_id: i64, detail: &str) -> WorkspaceNotice { @@ -88,6 +88,7 @@ fn page_from_query(page: Option<&str>, has_run: bool) -> &'static str { Some("runs") => "runs", Some("physical") => "physical", Some("catalog") => "catalog", + Some("requests") => "requests", Some("detail") => "detail", _ => "home", } @@ -245,15 +246,10 @@ pub fn App() -> Element { let mut page = use_signal(move || initial_page.to_string()); let runs = use_signal(|| None::); let runs_loading = use_signal(|| true); - let initial_query = url_param("q").unwrap_or_default(); - let mut query = use_signal({ - let initial_query = initial_query.clone(); - move || initial_query - }); - // Keep the input text separate from the query that drives the network - // request so typing can be debounced without making the input lag. - let mut applied_query = use_signal(move || initial_query); - let mut query_debounce_id = use_signal(|| 0u64); + let runs_generation = use_signal(|| 0u64); + let turn_generation = use_signal(|| 0_u64); + let mut last_runs_key = use_signal(|| None::); + let mut query = use_signal(|| url_param("q").unwrap_or_default()); let mut dataset_filter = use_signal(|| url_param("dataset_filter").unwrap_or_else(|| "all".into())); let mut status = use_signal(|| url_param("status").unwrap_or_else(|| "all".into())); @@ -280,8 +276,10 @@ pub fn App() -> Element { let last_place = use_signal(String::new); let history_ready = use_signal(|| false); let mut history_seq = use_signal(|| 0i32); - let catalog_tree = use_signal(|| None::); - let catalog_loading = use_signal(|| false); + let mut catalog_tree = use_signal(|| None::); + let mut catalog_loading = use_signal(|| false); + let mut catalog_generation = use_signal(|| 0u64); + let mut catalog_task = use_signal(|| None::); let mut offset = use_signal(|| 0usize); let mut error = use_signal(|| None::); let mut catalog_auth_configured = use_signal(|| catalog_auth::load().is_configured()); @@ -301,6 +299,8 @@ pub fn App() -> Element { let mut expanded_turn_id = use_signal(|| url_param("turn").and_then(|value| value.parse::().ok())); let detail_loading = use_signal(|| false); + let detail_failed = use_signal(|| false); + let detail_generation = use_signal(|| 0u64); let turn_loading = use_signal(|| false); let mut detail_mode = use_signal(|| url_param("workspace").unwrap_or_else(|| "trace".into())); let mut trace_mode = use_signal(|| { @@ -322,73 +322,105 @@ pub fn App() -> Element { let mut llm_config = use_signal(llm::load_config); use_effect(move || { - if !matches!(page().as_str(), "runs" | "detail") { + // Selecting a run navigates to `detail` while retaining the already + // loaded run page. Do not refetch the list just because the detail + // pane changed; only bootstrap it for a direct detail URL with no + // existing page. + let on_detail_without_runs = page() == "detail" && runs.peek().is_none(); + if page() != "runs" && !on_detail_without_runs { return; } - load_runs( - RunFilters { - query: applied_query(), - dataset: dataset_filter(), - status: status(), - sort: sort(), - direction: direction(), - path: run_path(), - file: file_prefix(), - offset: offset(), - }, - runs, - runs_loading, - error, + let filters = RunFilters { + query: query(), + dataset: dataset_filter(), + status: status(), + sort: sort(), + direction: direction(), + path: run_path(), + file: file_prefix(), + offset: offset(), + }; + let key = format!( + "{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}", + filters.query, + filters.dataset, + filters.status, + filters.sort, + filters.direction, + filters.path, + filters.file, + filters.offset, + page(), ); + if last_runs_key.peek().as_deref() == Some(key.as_str()) && runs.peek().is_some() { + return; + } + last_runs_key.set(Some(key)); + load_runs(filters, runs, runs_loading, runs_generation, error); }); use_effect(move || { + // One task per navigation. Cancel even on A -> B -> A, so old pollers + // cannot resume or overwrite the current page. + if let Some(task) = *catalog_task.peek() { + task.cancel(); + } + let requested_generation = (*catalog_generation.peek()).saturating_add(1); + catalog_generation.set(requested_generation); if page() != "catalog" { return; } let dataset = catalog_dataset(); let prefix = catalog_prefix(); - load_catalog_tree( - dataset.clone(), - prefix.clone(), - catalog_tree, - catalog_loading, - error, - ); - spawn(async move { + catalog_task.set(Some(spawn(async move { loop { - TimeoutFuture::new(CATALOG_REFRESH_MS).await; - if page() != "catalog" || catalog_dataset() != dataset || catalog_prefix() != prefix - { - break; - } load_catalog_tree( dataset.clone(), prefix.clone(), catalog_tree, catalog_loading, error, - ); + catalog_generation, + requested_generation, + ) + .await; + // Wait after completion. Slow requests must never overlap the + // next poll, and failures must not cause a tight retry loop. + TimeoutFuture::new(CATALOG_REFRESH_MS).await; } - }); + }))); }); + let mut detail_request_key = use_signal(|| None::); use_effect(move || { - if analysis().is_none() - && let Some(run) = selected_run() + let Some(run) = selected_run() else { + detail_request_key.set(None); + return; + }; + let key = run.query(); + if detail_request_key.peek().as_deref() == Some(key.as_str()) + && (analysis().is_some() || *detail_loading.peek() || *detail_failed.peek()) { - load_workspace( - run, - turn_query(), - source(), - analysis, - turns, - turn_search, - compact_record, - detail_loading, - error, - ); + return; } + detail_request_key.set(Some(key)); + let debounce = *turn_query_debounce_id.peek() + 1; + turn_query_debounce_id.set(debounce); + load_workspace( + run, + turn_query.peek().clone(), + source.peek().clone(), + analysis, + turns, + turn_search, + turn_loading, + turn_generation, + compact_record, + detail_loading, + detail_failed, + detail_generation, + error, + ); }); use_effect(move || { @@ -410,6 +442,31 @@ pub fn App() -> Element { } }); + // A filtered result should be useful immediately. Pick the first real + // match and open its parent conversation instead of leaving the hit hidden + // behind a collapsed summary row. + use_effect(move || { + let query = turn_query(); + if query.trim().is_empty() || expanded_turn_id().is_some() { + return; + } + let Some(turn) = turns().into_iter().find(|turn| turn.id >= 0) else { + return; + }; + expanded_turn_id.set(Some(turn.id)); + selected_turn.set(None); + if let Some(run) = selected_run() { + load_turn( + run, + turn.id, + expanded_turn_id, + selected_turn, + turn_loading, + error, + ); + } + }); + // Route signals are the only tracked inputs. last_place / history_ready are // written inside the helper with peek() + change-only set so this effect // cannot reschedule itself (Dioxus Signal::set always notifies). @@ -467,7 +524,6 @@ pub fn App() -> Element { file_prefix, run_path, query, - applied_query, status, sort, direction, @@ -490,8 +546,20 @@ pub fn App() -> Element { if page() == "home" { return; } - if catalog().is_none() { + let initial = catalog().is_none(); + let waiting = catalog().as_ref().is_some_and(|catalog| { + catalog.datasets.iter().any(|dataset| { + dataset + .browse + .as_ref() + .is_some_and(|status| status.observed_at == 0) + }) + }); + if initial || waiting { spawn(async move { + if !initial { + TimeoutFuture::new(CATALOG_REFRESH_MS).await; + } match api::query_catalog().await { Ok(value) => { if selected_table().is_empty() { @@ -540,11 +608,13 @@ pub fn App() -> Element { RailButton { active: page() == "tools", icon: "analysis", label: ANALYSIS, onclick: move |_| page.set("tools".into()) } RailButton { active: page() == "physical", icon: "storage", label: STORAGE, onclick: move |_| page.set("physical".into()) } } + RailButton { active: page() == "requests", icon: "analysis", label: "Requests", onclick: move |_| page.set("requests".into()) } div { class: "rail-spacer" } div { class: "rail-secondary", aria_label: "Assistant and settings", button { class: if copilot_open() { "rail-button active" } else { "rail-button" }, aria_label: "Toggle Assistant", aria_expanded: copilot_open(), onclick: move |_| copilot_open.set(!copilot_open()), WorkspaceIcon { name: "assistant" } span { {ASSISTANT} } } button { class: if settings_open() { "rail-button active" } else { "rail-button" }, aria_label: "Settings", onclick: move |_| settings_open.set(true), WorkspaceIcon { name: "keys" } span { "Keys" } } } + crate::requests::RequestIndicator { on_open: move |_| page.set("requests".into()) } { let identity = catalog_auth::load(); let configured = identity.is_configured(); @@ -570,6 +640,7 @@ pub fn App() -> Element { } } match page().as_str() { + "requests" => rsx! { crate::requests::RequestsPanel {} }, "catalog" => rsx! { CatalogExplorer { tree: catalog_tree(), @@ -577,6 +648,22 @@ pub fn App() -> Element { auth_required: !catalog_auth_configured() && catalog_tree().is_none(), on_settings: move |_| settings_open.set(true), on_open: move |(dataset, prefix): (String, String)| { + // Move immediately. The old tree must not remain visible while + // a slow remote prefix is loading. + catalog_tree.set(Some(CatalogTree { + dataset: (!dataset.is_empty()).then_some(dataset.clone()), + prefix: prefix.clone(), + browse: Some(BrowseStatus { + state: "refreshing".into(), + refreshing: true, + observed_at: 0, + stale: true, + ..Default::default() + }), + ..Default::default() + })); + catalog_loading.set(true); + catalog_generation.set(catalog_generation().saturating_add(1)); catalog_dataset.set(dataset); catalog_prefix.set(prefix); }, @@ -606,9 +693,13 @@ pub fn App() -> Element { let path_runs = runs().map(|page| page.path_index).unwrap_or_default(); let selected_path = analysis().map(|value| value.run.path).or_else(|| selected_run().map(|run| run.path)).unwrap_or_default(); rsx! { div { class: "pc2-detail-layout", - PathExplorer { runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path, loading: runs_loading(), + PathExplorer { paged: !query().trim().is_empty(), runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path, loading: runs_loading(), + page_total: runs().map(|page| page.snapshot.total).unwrap_or_default(), + page_offset: runs().map(|page| page.snapshot.offset).unwrap_or_default(), + page_limit: runs().map(|page| page.snapshot.limit).unwrap_or(50), on_path: move |value| { run_path.set(value); offset.set(0); page.set("runs".into()); }, on_view_mode: move |mode| path_list_mode.set(mode), + on_page: move |value| offset.set(value), on_select: move |run: RunSummary| { turn_query.set(query()); selected_run.set(Some(run)); analysis.set(None); turns.set(Vec::new()); turn_search.set(TurnSearchStatus::default()); selected_turn.set(None); drawer_turn.set(None); drawer_details.set(Vec::new()); drawer_turn_id.set(None); drawer_turn_ids.set(Vec::new()); drawer_title.set(String::new()); drawer_loading.set(false); expanded_turn_id.set(None); }, on_open_chat: move |run: RunSummary| { turn_query.set(query()); @@ -623,10 +714,16 @@ pub fn App() -> Element { copilot_open.set(true); }, } - if let (Some(_run), Some(value)) = (selected_run(), analysis()) { + if let Some(run) = selected_run() { RunDetailWorkspace { - run: value.run.clone(), - analysis: value, + failed: detail_failed(), + on_retry: move |_| { + if let Some(run) = selected_run.peek().clone() { + load_workspace(run,turn_query(),source(),analysis,turns,turn_search,turn_loading,turn_generation,compact_record,detail_loading,detail_failed,detail_generation,error); + } + }, + run, + analysis: analysis(), compact_record: compact_record(), turns: turns(), search: turn_search(), @@ -652,13 +749,18 @@ pub fn App() -> Element { on_view: move |value: String| { trace_mode.set(normalize_trace_view(&value).to_string()); }, - on_source: move |value| source.set(value), + on_source: move |value: String| { + source.set(value.clone()); + if let Some(run) = selected_run() { + load_turns(run, turn_query(), value, turns, turn_search, turn_loading, error, turn_generation); + } + }, on_query: move |value: String| { turn_query.set(value.clone()); let request_id = turn_query_debounce_id() + 1; turn_query_debounce_id.set(request_id); spawn(async move { - TimeoutFuture::new(SEARCH_DEBOUNCE_MS).await; + TimeoutFuture::new(STEP_SEARCH_DEBOUNCE_MS).await; if turn_query_debounce_id() == request_id && let Some(run) = selected_run() { @@ -670,6 +772,7 @@ pub fn App() -> Element { turn_search, turn_loading, error, + turn_generation, ); } }); @@ -677,6 +780,8 @@ pub fn App() -> Element { on_apply_query: move |value: String| { turn_query_debounce_id.set(turn_query_debounce_id() + 1); turn_query.set(value.clone()); + expanded_turn_id.set(None); + selected_turn.set(None); if let Some(run) = selected_run() { load_turns( run, @@ -686,6 +791,7 @@ pub fn App() -> Element { turn_search, turn_loading, error, + turn_generation, ); } }, @@ -750,9 +856,13 @@ pub fn App() -> Element { _ => { let path_runs = runs().map(|value| value.path_index).unwrap_or_default(); rsx! { div { class: "pc2-runs-layout", - PathExplorer { runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path: run_path(), loading: runs_loading(), + PathExplorer { paged: !query().trim().is_empty(), runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path: run_path(), loading: runs_loading(), + page_total: runs().map(|value| value.snapshot.total).unwrap_or_default(), + page_offset: runs().map(|value| value.snapshot.offset).unwrap_or_default(), + page_limit: runs().map(|value| value.snapshot.limit).unwrap_or(50), on_path: move |value| { run_path.set(value); offset.set(0); }, on_view_mode: move |mode| path_list_mode.set(mode), + on_page: move |value| offset.set(value), on_select: move |run: RunSummary| { turn_query.set(query()); selected_run.set(Some(run)); analysis.set(None); turns.set(Vec::new()); turn_search.set(TurnSearchStatus::default()); selected_turn.set(None); drawer_turn.set(None); drawer_details.set(Vec::new()); drawer_turn_id.set(None); drawer_turn_ids.set(Vec::new()); drawer_title.set(String::new()); drawer_loading.set(false); expanded_turn_id.set(None); detail_mode.set("trace".into()); page.set("detail".into()); }, on_open_chat: move |run: RunSummary| { turn_query.set(query()); @@ -779,24 +889,8 @@ pub fn App() -> Element { datasets: catalog().map(|value| value.datasets).unwrap_or_default(), dataset: dataset_filter(), chat_sessions: assistant_index().sessions.clone(), - on_query: move |value: String| { - query.set(value.clone()); - // A new search starts from the first page; retaining a - // previous offset can make valid matches look absent. - offset.set(0); - let request_id = query_debounce_id() + 1; - query_debounce_id.set(request_id); - spawn(async move { - TimeoutFuture::new(SEARCH_DEBOUNCE_MS).await; - if query_debounce_id() == request_id { - applied_query.set(value); - } - }); - }, on_apply_query: move |value: String| { - query_debounce_id.set(query_debounce_id() + 1); - query.set(value.clone()); - applied_query.set(value); + query.set(value); offset.set(0); }, on_dataset: move |value| { dataset_filter.set(value); run_path.set(String::new()); file_prefix.set(String::new()); offset.set(0); }, @@ -824,7 +918,7 @@ pub fn App() -> Element { if let Ok(value) = api::query_catalog().await { catalog.set(Some(value)); } - load_runs(filters, runs, runs_loading, error); + load_runs(filters, runs, runs_loading, runs_generation, error); }); }, on_page: move |value| offset.set(value), @@ -925,23 +1019,35 @@ fn load_runs( filters: RunFilters, mut page: Signal>, mut loading: Signal, + mut generation: Signal, mut error: Signal>, ) { + // This function is called from a reactive effect. Reading the signal here + // would subscribe that effect to its own generation writes and create a + // request loop. + let request_generation = *generation.peek() + 1; + generation.set(request_generation); page.set(None); loading.set(true); + error.set(None); spawn(async move { let all_datasets = filters.dataset.trim().is_empty() || filters.dataset == "all"; let dataset_names = if all_datasets { - api::query_catalog() - .await - .map(|catalog| { - catalog - .datasets - .into_iter() - .map(|dataset| dataset.name) - .collect::>() - }) - .unwrap_or_default() + match api::query_catalog().await { + Ok(catalog) => catalog + .datasets + .into_iter() + .map(|dataset| dataset.name) + .collect::>(), + Err(failure) => { + if generation() != request_generation { + return; + } + error.set(Some(workspace_notice(&failure))); + loading.set(false); + return; + } + } } else { vec![filters.dataset.clone()] }; @@ -949,13 +1055,13 @@ fn load_runs( // A separate request per mounted Dataset lets a fast Lance source // paint immediately while a slower JSON source is still scanning. // Each response is merged into the same page as it arrives. - let request_limit = if all_datasets { 200 } else { 50 }; + let request_limit = 50; let mut pending = FuturesUnordered::new(); for dataset in dataset_names { let mut scoped = filters.clone(); scoped.dataset = dataset; pending.push(async move { - api::explorer_runs( + let mut value = api::explorer_runs( &scoped.query, &scoped.dataset, &scoped.status, @@ -966,7 +1072,34 @@ fn load_runs( if all_datasets { 0 } else { scoped.offset }, request_limit, ) - .await + .await?; + // Merge only the sorted prefix needed for this global page. + // A fixed first-200 cap silently hid later matches. + while all_datasets + && value.snapshot.has_more + && value.records.len() < scoped.offset.saturating_add(request_limit) + && generation() == request_generation + { + let next = api::explorer_runs( + &scoped.query, + &scoped.dataset, + &scoped.status, + &scoped.sort, + &scoped.direction, + &scoped.path, + &scoped.file, + value.snapshot.next_offset, + request_limit, + ) + .await?; + if next.snapshot.next_offset <= value.snapshot.next_offset { + break; + } + value.snapshot = next.snapshot; + value.records.extend(next.records); + value.path_index.extend(next.path_index); + } + Ok::<_, crate::api::ApiFailure>(value) }); } @@ -976,16 +1109,33 @@ fn load_runs( match result { Ok(value) => { partials.push(value); - page.set(Some(merge_run_pages(&partials, &filters))); + if generation() == request_generation { + page.set(Some(merge_run_pages(&partials, &filters))); + } } Err(message) => { first_error.get_or_insert(message); } } } + if let Some(failure) = &first_error { + // Keep successful datasets visible, but never present a partial + // all-dataset result as complete. + if generation() == request_generation { + error.set(Some(workspace_notice(failure))); + } + } if partials.is_empty() { - // Preserve the previous all-Dataset behavior if catalog discovery - // failed before fan-out could be started. + // Keep the selected dataset path usable even when one scoped scan + // fails; for an all-dataset request the catalog error was already + // surfaced above, so an empty page is preferable to a second + // unscoped remote scan. + if all_datasets { + if generation() == request_generation { + loading.set(false); + } + return; + } match api::explorer_runs( &filters.query, &filters.dataset, @@ -999,11 +1149,16 @@ fn load_runs( ) .await { - Ok(value) => page.set(Some(value)), - Err(failure) => error.set(Some(workspace_notice(&first_error.unwrap_or(failure)))), + Ok(value) if generation() == request_generation => page.set(Some(value)), + Err(failure) if generation() == request_generation => { + error.set(Some(workspace_notice(&first_error.unwrap_or(failure)))) + } + _ => {} } } - loading.set(false); + if generation() == request_generation { + loading.set(false); + } }); } @@ -1024,11 +1179,11 @@ fn merge_run_pages(pages: &[RunPage], filters: &RunFilters) -> RunPage { let all_datasets = filters.dataset.is_empty() || filters.dataset == "all"; let snapshot = if all_datasets { - let total = records.len(); + let total = pages.iter().map(|page| page.snapshot.total).sum(); let limit = 50; let offset = filters.offset.min(total); - let next_offset = (offset + limit).min(total); records = records.into_iter().skip(offset).take(limit).collect(); + let next_offset = offset + records.len(); PageSnapshot { offset, next_offset, @@ -1051,10 +1206,14 @@ fn merge_run_pages(pages: &[RunPage], filters: &RunFilters) -> RunPage { }) }; - let mut path_index = pages - .iter() - .flat_map(|page| page.path_index.iter().cloned()) - .collect::>(); + let mut path_index = if filters.query.trim().is_empty() { + pages + .iter() + .flat_map(|page| page.path_index.iter().cloned()) + .collect::>() + } else { + records.iter().map(|item| item.run.clone()).collect() + }; path_index.sort_by(|left, right| left.path.cmp(&right.path)); path_index.dedup_by(|left, right| left.query() == right.query()); @@ -1079,31 +1238,37 @@ fn merge_run_pages(pages: &[RunPage], filters: &RunFilters) -> RunPage { } } -fn load_catalog_tree( +async fn load_catalog_tree( dataset: String, prefix: String, mut tree: Signal>, mut loading: Signal, mut error: Signal>, + generation: Signal, + requested_generation: u64, ) { loading.set(true); - spawn(async move { - match api::explorer_tree(&dataset, &prefix).await { - Ok(value) => tree.set(Some(value)), - Err(failure) - if matches!(failure.status, 400 | 401) - && dataset.is_empty() - && prefix.is_empty() => - { - match api::explorer_tree_anonymous(&dataset, &prefix).await { - Ok(value) => tree.set(Some(value)), - Err(failure) => error.set(Some(workspace_notice(&failure))), + match api::explorer_tree(&dataset, &prefix).await { + Ok(value) if *generation.peek() == requested_generation => tree.set(Some(value)), + Err(failure) + if matches!(failure.status, 400 | 401) && dataset.is_empty() && prefix.is_empty() => + { + match api::explorer_tree_anonymous(&dataset, &prefix).await { + Ok(value) if *generation.peek() == requested_generation => tree.set(Some(value)), + Err(failure) if *generation.peek() == requested_generation => { + error.set(Some(workspace_notice(&failure))) } + _ => {} } - Err(failure) => error.set(Some(workspace_notice(&failure))), } + Err(failure) if *generation.peek() == requested_generation => { + error.set(Some(workspace_notice(&failure))) + } + _ => {} + } + if *generation.peek() == requested_generation { loading.set(false); - }); + } } #[allow(clippy::too_many_arguments)] @@ -1114,38 +1279,106 @@ fn load_workspace( mut analysis: Signal>, mut turns: Signal>, mut turn_search: Signal, + mut turn_loading: Signal, + mut turn_generation: Signal, mut compact_record: Signal>, mut loading: Signal, + mut failed: Signal, + mut generation: Signal, mut error: Signal>, ) { + // Untracked reads avoid subscribing the caller's effect to its own writes. + let requested = *generation.peek() + 1; + generation.set(requested); compact_record.set(None); + analysis.set(None); + failed.set(false); loading.set(true); - spawn({ - let run = run.clone(); - async move { - let (next_analysis, next_turns) = - futures_util::join!(api::run_analysis(&run), api::turns(&run, &query, &source),); - match (next_analysis, next_turns) { - (Ok(next_analysis), Ok(next_turns)) => { - if next_analysis.run.is_compact_jsonl() { - match api::compact_record(&next_analysis.run).await { - Ok(value) => compact_record.set(Some(value)), - Err(failure) => error.set(Some(workspace_notice(&failure))), - } - } - analysis.set(Some(next_analysis)); - turns.set(next_turns.records); - turn_search.set(next_turns.search); + turn_loading.set(true); + turns.set(Vec::new()); + let step_requested = *turn_generation.peek() + 1; + turn_generation.set(step_requested); + spawn(async move { + // Steps first (no embedded analysis), paint the timeline, then load + // statistics. The follow-up /run hits the worker trajectory memo when + // the pool reuses the same child, so aggregates are usually cheap. + let turns_result = match futures_util::future::select( + Box::pin(api::turns(&run, &query, &source, false)), + Box::pin(TimeoutFuture::new(65_000)), + ) + .await + { + futures_util::future::Either::Left((result, _)) => result, + _ => Err(api::ApiFailure::network( + "Loading steps timed out. Open Requests to inspect server progress, then retry.", + )), + }; + if *generation.peek() != requested { + return; + } + match turns_result { + Ok(page) => { + if *turn_generation.peek() == step_requested { + turns.set(page.records); + turn_search.set(page.search); + turn_loading.set(false); + } + } + Err(failure) => { + if *turn_generation.peek() == step_requested { + turn_loading.set(false); + } + error.set(Some(workspace_notice(&failure))); + loading.set(false); + return; + } + } + + if run.is_compact_jsonl() { + match api::compact_record(&run).await { + Ok(record) if *generation.peek() == requested => { + compact_record.set(Some(record)); } - (Err(failure), _) | (_, Err(failure)) => { + Err(failure) if *generation.peek() == requested => { error.set(Some(workspace_notice(&failure))); } + _ => {} } - loading.set(false); + if *generation.peek() == requested { + loading.set(false); + } + return; } + + let analysis_result = match futures_util::future::select( + Box::pin(api::run_analysis(&run)), + Box::pin(TimeoutFuture::new(65_000)), + ) + .await + { + futures_util::future::Either::Left((result, _)) => result, + _ => Err(api::ApiFailure::network( + "Run statistics timed out. Steps remain available; open Requests to inspect progress.", + )), + }; + if *generation.peek() != requested { + return; + } + match analysis_result { + Ok(next_analysis) => { + analysis.set(Some(next_analysis)); + failed.set(false); + } + Err(failure) => { + failed.set(true); + error.set(Some(workspace_notice(&failure))); + } + } + loading.set(false); }); } +#[allow(clippy::too_many_arguments)] fn load_turns( run: RunSummary, query: String, @@ -1154,10 +1387,17 @@ fn load_turns( mut turn_search: Signal, mut loading: Signal, mut error: Signal>, + mut generation: Signal, ) { + let requested = *generation.peek() + 1; + generation.set(requested); loading.set(true); spawn(async move { - match api::turns(&run, &query, &source).await { + let result = api::turns(&run, &query, &source, false).await; + if *generation.peek() != requested { + return; + } + match result { Ok(value) => { turns.set(value.records); turn_search.set(value.search); @@ -1407,15 +1647,35 @@ fn PathExplorer( view_mode: PathListMode, selected_path: String, loading: bool, + paged: bool, + #[props(default)] page_total: usize, + #[props(default)] page_offset: usize, + #[props(default)] page_limit: usize, on_path: EventHandler, on_view_mode: EventHandler, on_select: EventHandler, on_open_chat: EventHandler, + on_page: EventHandler, ) -> Element { let import_path_tree = build_import_path_tree(&runs); + let limit = page_limit.max(1); + let total = if page_total > 0 { + page_total + } else { + runs.len() + }; + let page_count = total.div_ceil(limit).max(1); + let current_page = (page_offset / limit).min(page_count.saturating_sub(1)); + let page_start = if total == 0 { + 0 + } else { + page_offset.saturating_add(1) + }; + let page_end = page_offset.saturating_add(runs.len()).min(total); + let show_pager = !paged && total > limit; rsx! { aside { class: "pc2-path-explorer", header { - div { strong { "Run paths" } span { if view_mode == PathListMode::Flat { "All runs in this dataset" } else { "Tree by import path" } } } + div { strong { "Run paths" } span { if paged { "Search results on this page" } else if view_mode == PathListMode::Flat { "All runs in this dataset" } else { "Tree by import path" } } } div { class: "pc2-path-view-toggle", role: "radiogroup", aria_label: "Run list view", button { class: if view_mode == PathListMode::Flat { "active" } else { "" }, role: "radio", aria_checked: view_mode == PathListMode::Flat, onclick: move |_| on_view_mode.call(PathListMode::Flat), "Flat" } button { class: if view_mode == PathListMode::Tree { "active" } else { "" }, role: "radio", aria_checked: view_mode == PathListMode::Tree, onclick: move |_| on_view_mode.call(PathListMode::Tree), "Tree" } @@ -1423,7 +1683,7 @@ fn PathExplorer( span { "{runs.len()}" } } div { class: "pc2-path-tree", - button { class: if selected_path.is_empty() { "pc2-path-all active" } else { "pc2-path-all" }, onclick: move |_| on_path.call(String::new()), span { class: "pc2-path-icon root", "⌂" } strong { "All runs" } code { "{runs.len()}" } } + button { class: if selected_path.is_empty() { "pc2-path-all active" } else { "pc2-path-all" }, onclick: move |_| on_path.call(String::new()), span { class: "pc2-path-icon root", "⌂" } strong { if paged { "Search results" } else { "All runs" } } code { "{total}" } } if loading && runs.is_empty() { div { class: "pc2-path-loading", span { class: "spinner" } "Loading paths…" } } else if runs.is_empty() { div { class: "pc2-path-empty", "No captured run paths." } } else if view_mode == PathListMode::Flat { @@ -1436,7 +1696,38 @@ fn PathExplorer( } } } - footer { if view_mode == PathListMode::Flat { "Showing all runs in this dataset." } else { "Tree follows the imported path." } } + footer { class: if show_pager { "pc2-path-footer paged" } else { "pc2-path-footer" }, + if show_pager { + label { class: "pc2-path-page", + span { "Page" } + select { + value: "{current_page}", + aria_label: "Jump to run path page", + onchange: move |event| { + if let Ok(index) = event.value().parse::() { + on_page.call(index.saturating_mul(limit)); + } + }, + for index in 0..page_count { + option { value: "{index}", selected: index == current_page, + { + let start = index.saturating_mul(limit).saturating_add(1); + let end = index.saturating_mul(limit).saturating_add(limit).min(total); + format!("{}/{} · {start}–{end}", index + 1, page_count) + } + } + } + } + } + span { "{page_start}–{page_end} of {total}" } + } else if paged { + "Showing the current search page." + } else if view_mode == PathListMode::Flat { + "Showing all runs in this dataset." + } else { + "Tree follows the imported path." + } + } } } } @@ -1490,7 +1781,6 @@ fn RunsExplorer( direction: String, path: String, file: String, - on_query: EventHandler, on_apply_query: EventHandler, on_dataset: EventHandler, on_status: EventHandler, @@ -1503,6 +1793,11 @@ fn RunsExplorer( on_open_chat: EventHandler, on_select: EventHandler, ) -> Element { + // Typing stays local; only submitted queries reach the page's request effect. + let mut draft_query = use_signal(|| query.clone()); + use_effect(use_reactive((&query,), move |(query,)| { + draft_query.set(query) + })); let total = page.as_ref().map_or(0, |page| page.snapshot.total); let page_offset = page.as_ref().map_or(0, |page| page.snapshot.offset); let page_limit = page.as_ref().map_or(50, |page| page.snapshot.limit); @@ -1522,7 +1817,7 @@ fn RunsExplorer( _ => "FTS unavailable", }; let search_placeholder = if search.fts_available { - "Search runs/content or JSONB (find syntax)" + "Search message body · Enter to search" } else { "Search unavailable for this Dataset" }; @@ -1533,10 +1828,10 @@ fn RunsExplorer( button { class: "button", onclick: on_refresh, "↻ Refresh" } } div { class: "pc2-filterbar", - label { class: "pc2-filter-search", span { "⌕" } input { value: "{query}", placeholder: "{search_placeholder}", aria_label: "Search runs and content", oninput: move |event| on_query.call(event.value()), onkeydown: move |event| { if event.key() == Key::Enter { event.prevent_default(); on_apply_query.call(query.clone()); } } } if !query.is_empty() { button { r#type: "button", class: "pc2-filter-clear", aria_label: "Clear run search", title: "Clear search", onclick: move |event| { event.prevent_default(); on_apply_query.call(String::new()); }, "×" } } } + label { class: "pc2-filter-search", span { "⌕" } input { value: "{draft_query}", placeholder: "{search_placeholder}", aria_label: "Search runs and content", title: "Search message body. Use #all(...) for all fields or an explicit field/JSON filter. Press Enter to search", oninput: move |event| draft_query.set(event.value()), onkeydown: move |event| { if event.key() == Key::Enter { event.prevent_default(); on_apply_query.call(draft_query()); } } } if !draft_query().is_empty() { button { r#type: "button", class: "pc2-filter-clear", aria_label: "Clear run search", title: "Clear search", onclick: move |event| { event.prevent_default(); draft_query.set(String::new()); on_apply_query.call(String::new()); }, "×" } } } select { value: "{dataset}", aria_label: "Filter by Dataset", onchange: move |event| on_dataset.call(event.value()), option { value: "all", "All Datasets" } - for mounted in datasets { option { value: "{mounted.name}", "{mounted.name}" } } + for mounted in datasets { option { value: "{mounted.name}", "{mounted.label()}" } } } select { value: "{status}", aria_label: "Filter by run status", onchange: move |event| on_status.call(event.value()), option { value: "all", "All statuses" } option { value: "active", "Active" } option { value: "completed", "Completed" } option { value: "failed", "Failed" } } select { value: "{sort}", aria_label: "Sort runs", onchange: move |event| on_sort.call(event.value()), option { value: "session", "Session" } option { value: "events", "Events" } option { value: "status", "Status" } option { value: "agent", "Agent" } } @@ -1546,6 +1841,13 @@ fn RunsExplorer( span { class: if search.fts_available { "pc2-search-mode available" } else { "pc2-search-mode unavailable" }, title: "{search_label}", "{search_label}" } span { class: "pc2-result-count", "{total} runs" } } + if page.is_some() { + footer { class: "pc2-pagination", + button { disabled: page_offset == 0, onclick: move |_| on_page.call(page_offset.saturating_sub(page_limit)), "← Previous" } + span { "{page_offset + usize::from(total > 0)}–{page_next} of {total}" } + button { disabled: !page_has_more, onclick: move |_| on_page.call(page_next), "Next →" } + } + } div { class: "pc2-table-wrap", table { class: "pc2-run-table", thead { tr { th { "Session" } th { "Agent / model" } th { "Status" } th { "Events" } th { "Root" } } } @@ -1564,13 +1866,7 @@ fn RunsExplorer( } } } - if page.is_some() { - footer { class: "pc2-pagination", - button { disabled: page_offset == 0, onclick: move |_| on_page.call(page_offset.saturating_sub(page_limit)), "← Previous" } - span { "{page_offset + usize::from(total > 0)}–{page_next} of {total}" } - button { disabled: !page_has_more, onclick: move |_| on_page.call(page_next), "Next →" } - } - } + } } } @@ -1595,7 +1891,7 @@ fn RunTableRow( .map(|value| search_preview_excerpt(value, &highlight_query)); rsx! { tr { tabindex: "0", onclick: move |_| on_select.call(run.clone()), onkeydown: move |event| if event.key() == Key::Enter { on_select.call(keyboard_run.clone()) }, - td { div { class: "pc2-session-cell", div { class: "pc2-session-heading", strong { if compact { "Record · " } HighlightedText { text: item.run.session_id.clone(), query: query.clone() } } if has_chat && !compact { ChatMarker { run: run.clone(), on_open_chat } } } span { if compact { "1 JSON record · {item.run.file}" } else { "{item.run.row_count} captured rows" } } if let Some(preview) = preview { div { class: "pc2-run-search-preview", title: "Matched content preview", HighlightedText { text: preview, query: highlight_query.clone() } } } } } + td { div { class: "pc2-session-cell", div { class: "pc2-session-heading", strong { if compact { "Record · " } HighlightedText { text: item.run.session_id.clone(), query: query.clone() } } if has_chat && !compact { ChatMarker { run: run.clone(), on_open_chat } } } span { if compact { "1 JSON record · {item.run.file}" } else { "{item.run.row_count} captured rows" } } if let Some(preview) = preview { div { class: "pc2-run-search-preview", title: "{preview}", span { "Match: " } HighlightedText { text: preview.clone(), query: highlight_query.clone() } } } } } td { div { class: "pc2-session-cell", strong { if compact { "Compact JSONL" } else { HighlightedText { text: item.run.agent_id.clone(), query: query.clone() } } } span { if !compact { HighlightedText { text: model_text.clone(), query: query.clone() } } } } } td { StatusBadge { value: item.run.status.clone() } } td { class: "pc2-number", "{item.run.row_count}" } @@ -1676,8 +1972,10 @@ fn StatusBadge(value: String) -> Element { #[component] #[allow(clippy::too_many_arguments)] fn RunDetailWorkspace( + failed: bool, + on_retry: EventHandler, run: RunSummary, - analysis: RunAnalysis, + analysis: Option, compact_record: Option, turns: Vec, search: TurnSearchStatus, @@ -1737,6 +2035,11 @@ fn RunDetailWorkspace( }; rsx! { section { class: if compact_header() { "pc2-detail is-condensed" } else { "pc2-detail" }, + if failed { + div { role: "alert", "Run statistics could not be loaded. Steps remain available." + button { class: "button", onclick: on_retry, "Retry" } + } + } header { class: "pc2-detail-head", div { class: "pc2-detail-title", button { class: "pc2-back", onclick: on_back, "← Runs" } div { p { "{run.agent_id}" } h1 { title: "{run.session_id}", "{run.session_id}" } div { StatusBadge { value: run.status.clone() } if let Some(root) = &run.root_session_id { code { "root {short(root, 24)}" } } } } } div { class: "pc2-head-actions", @@ -1762,22 +2065,29 @@ fn RunDetailWorkspace( } } } else { - MetricsStrip { analysis: analysis.clone() } + if let Some(value) = analysis.clone() { + MetricsStrip { analysis: value.clone() } if detail_mode == "trace" { CompactOverviewStrip { - analysis: analysis.clone(), + analysis: value, turns: turns.clone(), on_open_analysis: move |_| on_detail_mode.call("analysis".into()), } } + } else if loading { + div { class: "pc2-inline-loading", role: "status", "Loading run statistics…" } + } nav { class: "pc2-detail-tabs", aria_label: "Run detail view", button { class: if detail_mode == "trace" { "active" } else { "" }, onclick: move |_| on_detail_mode.call("trace".into()), {TIMELINE} } button { class: if detail_mode == "analysis" { "active" } else { "" }, onclick: move |_| on_detail_mode.call("analysis".into()), "Analysis" } - span { "{turns.len()} of {analysis.turn_count} steps loaded for interactive charts" } + if let Some(value) = &analysis { + span { "{turns.len()} of {value.turn_count} steps loaded for interactive charts" } + } else { span { "{turns.len()} steps loaded" } } } if detail_mode == "analysis" { + if let Some(value) = analysis { AnalysisWorkspace { - analysis: analysis.clone(), + analysis: value, turns: turns.clone(), on_turn: move |id| { on_turn.call(id); @@ -1785,6 +2095,7 @@ fn RunDetailWorkspace( }, on_scroll: on_detail_scroll, } + } else { div { role: "status", "Run statistics are not available yet." } } } else { section { class: "pc2-trace-surface pc2-inline-trace", div { class: "pc2-trace-toolbar", @@ -1800,8 +2111,8 @@ fn RunDetailWorkspace( } } div { id: RUN_DETAIL_SCROLL_ID, class: "pc2-turn-list pc2-span-scroll", onscroll: on_detail_scroll, - if loading { div { class: "pc2-inline-loading", span { class: "spinner" } "Refreshing run details…" } } - if turns.is_empty() { div { class: "pc2-empty", strong { "No visible steps" } span { "No loaded steps match this filter." } } } + if turn_loading { div { class: "pc2-inline-loading", role: "status", span { class: "spinner" } "Loading steps…" } } + if turns.is_empty() && !turn_loading { div { class: "pc2-empty", strong { "No visible steps" } span { "No loaded steps match this filter." } } } else { TrajectoryView { turns, expanded_turn_id, detail: selected, loading: turn_loading, view: view_for_list, source: source_for_list, query: query_for_list, on_turn, on_open_drawer } } } } @@ -2946,7 +3257,6 @@ fn apply_workspace_search( mut file_prefix: Signal, mut run_path: Signal, mut query: Signal, - mut applied_query: Signal, mut status: Signal, mut sort: Signal, mut direction: Signal, @@ -2971,8 +3281,7 @@ fn apply_workspace_search( file_prefix.set(query_value(search, "file_prefix").unwrap_or_default()); run_path.set(query_value(search, "path").unwrap_or_default()); let next_query = query_value(search, "q").unwrap_or_default(); - query.set(next_query.clone()); - applied_query.set(next_query); + query.set(next_query); status.set(query_value(search, "status").unwrap_or_else(|| "all".into())); sort.set(query_value(search, "sort").unwrap_or_else(|| "session".into())); direction.set(query_value(search, "direction").unwrap_or_else(|| "asc".into())); @@ -3125,6 +3434,38 @@ mod tests { ); } + #[test] + fn detail_renders_steps_while_statistics_are_pending() { + let mut dom = VirtualDom::new(|| { + let turn: TurnSummary = serde_json::from_value(serde_json::json!({ + "id": 7, "source": "user", "kind": null, "timestamp": null, + "call_id": null, "preview": "ready-step-before-statistics", "model_name": null, + "latency_ms": null, "ttft_ms": null, "prompt_tokens": null, + "completion_tokens": null, "total_tokens": null, "tool_names": [], + "event_seqs": [], "has_error": false + })) + .unwrap(); + rsx! { RunDetailWorkspace { + failed: false, on_retry: |_| {}, + run: run_at("a/run"), analysis: None, compact_record: None, + turns: vec![turn], search: TurnSearchStatus::default(), + selected: None, drawer: None, drawer_details: vec![], drawer_ids: vec![], + drawer_title: String::new(), drawer_loading: false, expanded_turn_id: None, + loading: true, turn_loading: false, detail_mode: "trace".to_string(), + view: "steps".to_string(), source: "all".to_string(), query: String::new(), + on_back: |_| {}, on_detail_mode: |_| {}, on_view: |_| {}, on_source: |_| {}, + on_query: |_| {}, on_apply_query: |_| {}, on_turn: |_| {}, + on_open_drawer: |_| {}, on_close_drawer: |_| {}, on_open_copilot: |_| {}, + on_analyze: |_| {}, + } } + }); + let mutations = format!("{:?}", dom.rebuild_to_vec()); + assert!( + mutations.contains("ready-step-before-statistics"), + "{mutations}" + ); + } + fn drain_effects(dom: &mut VirtualDom) { use futures_util::FutureExt; for _ in 0..8 { @@ -3361,6 +3702,18 @@ mod tests { assert!(!merged.snapshot.has_more); } + #[test] + fn merged_search_keeps_remote_totals_and_only_visible_paths() { + let first = server_page(0, 300, &["a", "b"]); + let second = server_page(0, 400, &["c", "d"]); + let mut filters = run_filters("all", 0); + filters.query = "rust".into(); + let merged = merge_run_pages(&[first, second], &filters); + assert_eq!(merged.snapshot.total, 700); + assert!(merged.snapshot.has_more); + assert_eq!(merged.path_index.len(), merged.records.len()); + } + #[test] fn analyze_workspace_url_retains_only_the_session_id() { assert_eq!(