From 47ca6810e3a44700d0f2989270331668b9575fdc Mon Sep 17 00:00:00 2001 From: Reiase Date: Mon, 14 Sep 2026 23:23:39 +0800 Subject: [PATCH 01/22] feat(catalog): introduce catalog state management and consistency tracking - Added new `status` module to define `CatalogConsistency`, `CatalogState`, and `CatalogStatus` for improved catalog state management. - Updated various modules to utilize the new catalog state definitions, enhancing clarity and consistency in catalog operations. - Refactored error handling in the CLI server to include execution stages, improving error reporting and debugging capabilities. This commit aims to enhance the robustness of catalog management and provide clearer insights into catalog states during operations. --- .../src/server/catalog_worker.rs | 10 ++--- .../src/server/mod.rs | 26 +++++++------ .../src/server/problem.rs | 18 +++++++++ .../src/server/ui_cache.rs | 27 ++++++++++++-- crates/persisting-pchronicle/src/storage.rs | 37 ++++++++++--------- .../src/store/catalog/mod.rs | 2 + .../src/store/catalog/status.rs | 30 +++++++++++++++ crates/persisting-pchronicle/src/store/mod.rs | 17 +++++---- 8 files changed, 121 insertions(+), 46 deletions(-) create mode 100644 crates/persisting-pchronicle/src/store/catalog/status.rs diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index 52343a515..12d244f2f 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -17,7 +17,7 @@ use tower::ServiceExt; use super::{ catalog::{CatalogLibrary, apply_library_env}, - problem::ApiError, + problem::{ApiError, ExecutionStage}, }; const MAX_WORKERS: usize = 8; @@ -78,7 +78,7 @@ impl WorkerPool { pub(super) fn admit(&self) -> Result, ApiError> { self.requests .try_acquire() - .map_err(|_| ApiError::unavailable()) + .map_err(|_| ApiError::unavailable().with_stage(ExecutionStage::Admission)) } async fn slot(&self, scope: &str) -> Result { @@ -94,7 +94,7 @@ impl WorkerPool { .find(|(_, slot)| Arc::strong_count(slot) == 1) .map(|(key, _)| key.clone()); let Some(idle) = idle else { - return Err(ApiError::unavailable()); + return Err(ApiError::unavailable().with_stage(ExecutionStage::Admission)); }; if let Some(slot) = slots.remove(&idle) && let Some(mut worker) = slot.lock().await.take() @@ -128,13 +128,13 @@ impl WorkerPool { Ok(response) }) .await - .map_err(|_| ApiError::unavailable())? + .map_err(|_| ApiError::unavailable().with_stage(ExecutionStage::Query))? } } fn worker_error(error: anyhow::Error) -> ApiError { // Protocol/OS diagnostics only; never log bootstrap payloads or child stderr. - ApiError::internal("", "catalog_worker", error) + ApiError::internal("", "catalog_worker", error).with_stage(ExecutionStage::Worker) } struct Worker { diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 1c3676663..767f9ba4e 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -33,15 +33,16 @@ use persisting_pchronicle::search::storyline_steps_fts_available; #[cfg(test)] use persisting_pchronicle::storage::StoryCoords; use persisting_pchronicle::storage::{ - CatalogErrorPolicy, CatalogEventProvenance, CatalogSnapshotOptions, CatalogStorylineKey, - DEFAULT_DATASET_NAME, DatasetCatalogSnapshot, DatasetMount, + CatalogConsistency, CatalogErrorPolicy, CatalogEventProvenance, CatalogSnapshotOptions, + CatalogStorylineKey, DEFAULT_DATASET_NAME, DatasetCatalogSnapshot, DatasetMount, }; use serde::{Deserialize, Serialize}; use serde_json::{Value, json}; use acceleration::{AccelerationStatus, ServerAcceleration}; use problem::{ - ApiError, CHAIN_LIMIT, LOG_TARGET, QUERY_LOG_LIMIT, ROOT_CAUSE_LIMIT, truncate_utf8, + ApiError, CHAIN_LIMIT, ExecutionStage, LOG_TARGET, QUERY_LOG_LIMIT, ROOT_CAUSE_LIMIT, + truncate_utf8, }; use request_log::{FtsDiagnostics, RequestId, RequestMetrics}; @@ -556,7 +557,9 @@ async fn current_catalog( } let runtime = build_catalog_runtime(&state.config) .await - .map_err(|error| fail(request_id, "current_catalog", error))?; + .map_err(|error| { + fail(request_id, "current_catalog", error).with_stage(ExecutionStage::Catalog) + })?; *state.catalog.write().await = Some(Arc::clone(&runtime)); Ok(runtime) } @@ -628,7 +631,7 @@ async fn rebuild_catalog_for_runs( #[derive(Debug, Serialize)] struct CatalogResponse { - consistency: &'static str, + consistency: CatalogConsistency, snapshot_id: String, created_at: String, default_dataset: Option, @@ -639,7 +642,7 @@ struct CatalogResponse { fn catalog_response(state: &AppState, runtime: &CatalogRuntime) -> CatalogResponse { CatalogResponse { - consistency: "per_source_pinned", + consistency: CatalogConsistency::Pinned, snapshot_id: runtime.snapshot.snapshot_id().to_string(), created_at: runtime.snapshot.created_at().to_string(), default_dataset: runtime.snapshot.default_dataset().map(str::to_owned), @@ -667,10 +670,9 @@ async fn refresh_catalog( let warehouse = PreparedWarehouse { state: state.clone(), }; - let runtime = warehouse - .refresh_runtime() - .await - .map_err(|error| fail(&request_id, "refresh_catalog", error))?; + let runtime = warehouse.refresh_runtime().await.map_err(|error| { + fail(&request_id, "refresh_catalog", error).with_stage(ExecutionStage::Catalog) + })?; Ok(Json(catalog_response(&state, &runtime))) } @@ -1412,7 +1414,9 @@ async fn explorer_tree( .await .tree(mount, prefix) .await - .map_err(|error| fail(&request_id, "explorer_tree", error))?; + .map_err(|error| { + fail(&request_id, "explorer_tree", error).with_stage(ExecutionStage::Manifest) + })?; metrics.record("browse", started); Ok(Json(serde_json::to_value(view).unwrap())) } diff --git a/crates/persisting-pchronicle-cli/src/server/problem.rs b/crates/persisting-pchronicle-cli/src/server/problem.rs index 2c3abf862..7d519b4c8 100644 --- a/crates/persisting-pchronicle-cli/src/server/problem.rs +++ b/crates/persisting-pchronicle-cli/src/server/problem.rs @@ -21,6 +21,16 @@ pub(crate) enum BoundaryCode { Internal, } +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub(crate) enum ExecutionStage { + Admission, + Catalog, + Manifest, + Worker, + Query, +} + impl BoundaryCode { pub(crate) const fn as_str(self) -> &'static str { match self { @@ -73,6 +83,8 @@ pub(super) struct ApiError { pub(super) code: BoundaryCode, message: String, request_id: String, + #[serde(skip_serializing_if = "Option::is_none")] + stage: Option, #[serde(skip)] root_cause: Option, } @@ -84,6 +96,7 @@ impl ApiError { code, message: message.into(), request_id: String::new(), + stage: None, root_cause: None, } } @@ -93,6 +106,11 @@ impl ApiError { self } + pub(super) fn with_stage(mut self, stage: ExecutionStage) -> Self { + self.stage = Some(stage); + self + } + pub(super) fn with_4xx_root_cause(self, error: &anyhow::Error) -> Self { if error.source().is_some() { self.with_4xx_root_cause_text(truncate_utf8( diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index 1f50ed030..ce9bf9c29 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -7,7 +7,9 @@ use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant}; use anyhow::{Context, Result}; -use persisting_pchronicle::storage::{DatasetLocation, DatasetMount, ManifestCache}; +use persisting_pchronicle::storage::{ + CatalogConsistency, CatalogState, DatasetLocation, DatasetMount, ManifestCache, +}; use serde::{Deserialize, Serialize}; use tokio::sync::{RwLock, mpsc, oneshot}; @@ -87,7 +89,8 @@ pub(crate) struct BrowseSnapshot { #[derive(Clone, Debug, Serialize)] struct BrowseStatus { - consistency: &'static str, + consistency: CatalogConsistency, + state: CatalogState, generation: String, observed_at: i64, stale: bool, @@ -253,7 +256,14 @@ impl BrowseCoordinator { .fold(0usize, |sum, c| sum.saturating_add(c.failed_count)); BrowseSnapshot { browse: BrowseStatus { - consistency: "best_effort", + consistency: CatalogConsistency::BestEffort, + state: if refreshing { + CatalogState::Refreshing + } else if !complete || error.is_some() || now() - observed_at >= 30 { + CatalogState::Stale + } else { + CatalogState::Ready + }, generation: blake3::hash(&serde_json::to_vec(&tree).unwrap()) .to_hex() .to_string(), @@ -335,7 +345,16 @@ impl BrowseCoordinator { BrowseSnapshot { tree: entry.tree, browse: BrowseStatus { - consistency: "best_effort", + consistency: CatalogConsistency::BestEffort, + state: if self.pending.lock().unwrap().contains_key(key) { + CatalogState::Refreshing + } else if error.is_some() + || now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64 + { + CatalogState::Stale + } else { + CatalogState::Ready + }, generation: entry.generation, observed_at: entry.observed_at, stale: error.is_some() diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 096522bc7..131b7afcf 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -64,24 +64,25 @@ pub use crate::store::object_store_io_gate::{ #[cfg(feature = "lance-store")] pub use crate::store::{ AppendOutcome, AttemptRecord, AttemptRecordState, AttemptRegistry, CachedDataset, - CatalogDataset, CatalogErrorPolicy, CatalogEventProvenance, CatalogEventView, CatalogNamespace, - CatalogPage, CatalogProjectionStatus, CatalogSnapshotOptions, CatalogSourceDescription, - CatalogSourceKind, CatalogSourceRevision, CatalogSourceStatus, CatalogStorylineKey, - CatalogTrajectoryBundle, ChronicleManifest, CommitRunOutcome, CompactJsonlBuildPhase, - CompactJsonlColumn, CompactJsonlImportEvent, CompactJsonlOffload, CompactJsonlOptions, - CompactJsonlRecord, CompactJsonlStore, DEFAULT_CONTENT_OFFLOAD_THRESHOLD, - DEFAULT_CONTENT_PREVIEW_BYTES, DEFAULT_DATASET_NAME, DEFAULT_MAX_CHUNK_BYTES, - DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, DEFAULT_PHYSICAL_PAGE_LIMIT, - Dataset, DatasetCatalogSnapshot, DatasetLocation, DatasetLocationKind, DatasetMount, - DatasetResolver, DiscoveredSource, EventFactSnapshot, EventLogLayoutStats, EventWriterFence, - ExportOutcome, ImportableObjectEvent, LanceMaintenanceOptions, LanceMaintenanceReport, - LeaseAcquireOutcome, LocationSummary, ManifestCache, ManifestKind, ManifestListing, - ManifestReadMode, ManifestStats, NamespacePath, PathListEntry, PathListKind, PersistentCache, - PhysicalColumn, PhysicalDataFile, PhysicalFileLayout, PhysicalFragment, PhysicalLayout, - PhysicalPage, PhysicalPagePreview, PhysicalPageQuery, PhysicalSource, PhysicalTable, - ProjectionSourceSnapshot, QueryScope, RawEventLanceAppender, RawEventLanceStore, ReplayOutcome, - ResolveMode, ResolveTarget, RunControlStore, ShallowNavEntry, StorylineContentOptions, - StorylineContentReadMode, StorylineDataSource, StorylineDataSourceOptions, StorylineLanceStore, + CatalogConsistency, CatalogDataset, CatalogErrorPolicy, CatalogEventProvenance, + CatalogEventView, CatalogNamespace, CatalogPage, CatalogProjectionStatus, + CatalogSnapshotOptions, CatalogSourceDescription, CatalogSourceKind, CatalogSourceRevision, + CatalogSourceStatus, CatalogState, CatalogStatus, CatalogStorylineKey, CatalogTrajectoryBundle, + ChronicleManifest, CommitRunOutcome, CompactJsonlBuildPhase, CompactJsonlColumn, + CompactJsonlImportEvent, CompactJsonlOffload, CompactJsonlOptions, CompactJsonlRecord, + CompactJsonlStore, DEFAULT_CONTENT_OFFLOAD_THRESHOLD, DEFAULT_CONTENT_PREVIEW_BYTES, + DEFAULT_DATASET_NAME, DEFAULT_MAX_CHUNK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_BYTES, + DEFAULT_MAX_EVENT_FALLBACK_ROWS, DEFAULT_PHYSICAL_PAGE_LIMIT, Dataset, DatasetCatalogSnapshot, + DatasetLocation, DatasetLocationKind, DatasetMount, DatasetResolver, DiscoveredSource, + EventFactSnapshot, EventLogLayoutStats, EventWriterFence, ExportOutcome, ImportableObjectEvent, + LanceMaintenanceOptions, LanceMaintenanceReport, LeaseAcquireOutcome, LocationSummary, + ManifestCache, ManifestKind, ManifestListing, ManifestReadMode, ManifestStats, NamespacePath, + PathListEntry, PathListKind, PersistentCache, PhysicalColumn, PhysicalDataFile, + PhysicalFileLayout, PhysicalFragment, PhysicalLayout, PhysicalPage, PhysicalPagePreview, + PhysicalPageQuery, PhysicalSource, PhysicalTable, ProjectionSourceSnapshot, QueryScope, + RawEventLanceAppender, RawEventLanceStore, ReplayOutcome, ResolveMode, ResolveTarget, + RunControlStore, ShallowNavEntry, StorylineContentOptions, StorylineContentReadMode, + StorylineDataSource, StorylineDataSourceOptions, StorylineLanceStore, StorylineMaintenanceReport, StorylineProjectionLineage, StorylineSearchIndexSuppressGuard, StorylineStreamImportReport, StorylineStreamOptions, StorylineTablePaths, TrajectoryStats, attempt_registry_now_ms, distinct_session_ids_in_run, export_source_dirs, export_story_bundle, diff --git a/crates/persisting-pchronicle/src/store/catalog/mod.rs b/crates/persisting-pchronicle/src/store/catalog/mod.rs index 61f101834..accd2888f 100644 --- a/crates/persisting-pchronicle/src/store/catalog/mod.rs +++ b/crates/persisting-pchronicle/src/store/catalog/mod.rs @@ -14,6 +14,7 @@ mod namespace; mod provider; mod resolver; mod source; +mod status; pub use identity::{CatalogSourceRevision, DatasetMount, NamespacePath}; #[allow(unused_imports)] @@ -34,6 +35,7 @@ pub use namespace::{CatalogNamespace, CatalogPage, CatalogSourceDescription}; use provider::*; pub use resolver::{CachedDataset, Dataset, DatasetResolver, ResolveMode, ResolveTarget}; use source::*; +pub use status::{CatalogConsistency, CatalogState, CatalogStatus}; use discovery::{ bind_canonical_storyline_projections, discover_cached_candidates, discover_candidate_at, diff --git a/crates/persisting-pchronicle/src/store/catalog/status.rs b/crates/persisting-pchronicle/src/store/catalog/status.rs new file mode 100644 index 000000000..5c5456efc --- /dev/null +++ b/crates/persisting-pchronicle/src/store/catalog/status.rs @@ -0,0 +1,30 @@ +//! Shared catalog state vocabulary for UI, CLI, and query callers. + +use serde::Serialize; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum CatalogConsistency { + BestEffort, + #[serde(rename = "per_source_pinned")] + Pinned, + Exact, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum CatalogState { + Ready, + Stale, + Refreshing, + Unavailable, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct CatalogStatus { + pub consistency: CatalogConsistency, + pub state: CatalogState, + pub generation: String, + pub observed_at: i64, + pub last_error: Option, +} diff --git a/crates/persisting-pchronicle/src/store/mod.rs b/crates/persisting-pchronicle/src/store/mod.rs index 8e4b293a8..5972c9b69 100644 --- a/crates/persisting-pchronicle/src/store/mod.rs +++ b/crates/persisting-pchronicle/src/store/mod.rs @@ -84,14 +84,15 @@ pub use catalog::manifest::{ }; #[cfg(feature = "lance-store")] pub use catalog::{ - CATALOG_SOURCES_TABLE, CATALOG_TRAJECTORIES_TABLE, CachedDataset, CatalogDataset, - CatalogErrorPolicy, CatalogEventProvenance, CatalogEventView, CatalogNamespace, CatalogPage, - CatalogProjectionStatus, CatalogSnapshotOptions, CatalogSourceDescription, CatalogSourceKind, - CatalogSourceRevision, CatalogSourceStatus, CatalogStorylineKey, CatalogTrajectoryBundle, - DEFAULT_DATASET_NAME, DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, - Dataset, DatasetCatalogSnapshot, DatasetMount, DatasetResolver, DiscoveredSource, - LocationSummary, ManifestCache, ManifestListing, ManifestReadMode, NamespacePath, QueryScope, - ResolveMode, ResolveTarget, + CATALOG_SOURCES_TABLE, CATALOG_TRAJECTORIES_TABLE, CachedDataset, CatalogConsistency, + CatalogDataset, CatalogErrorPolicy, CatalogEventProvenance, CatalogEventView, CatalogNamespace, + CatalogPage, CatalogProjectionStatus, CatalogSnapshotOptions, CatalogSourceDescription, + CatalogSourceKind, CatalogSourceRevision, CatalogSourceStatus, CatalogState, CatalogStatus, + CatalogStorylineKey, CatalogTrajectoryBundle, DEFAULT_DATASET_NAME, + DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, Dataset, + DatasetCatalogSnapshot, DatasetMount, DatasetResolver, DiscoveredSource, LocationSummary, + ManifestCache, ManifestListing, ManifestReadMode, NamespacePath, QueryScope, ResolveMode, + ResolveTarget, }; #[cfg(feature = "lance-store")] pub use compact_jsonl::{ From 2791a0cf0ebec9c6e5cdbd1ef4b493559a832abb Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 15:41:08 +0800 Subject: [PATCH 02/22] feat(pchronicle): enhance caching and storage mechanisms - Introduced `lance-io` and `object_store` dependencies to improve object storage capabilities. - Added `BlockCache` and `CacheConfig` for efficient caching of object store reads, enhancing performance. - Implemented a new `lance_cache` benchmark to evaluate caching performance. - Refactored dataset access methods to utilize the new caching mechanisms, improving data retrieval efficiency. - Updated various modules to integrate the new caching features, ensuring consistency across data operations. This commit aims to optimize data handling and improve the overall performance of the pChronicle application. --- Cargo.lock | 3 + Cargo.toml | 2 + .../src/gateway_partition.rs | 10 +- .../src/server/query_admission.rs | 22 +- .../src/server/ui_cache.rs | 341 +++++++++----- crates/persisting-pchronicle-cli/src/sync.rs | 58 ++- crates/persisting-pchronicle/Cargo.toml | 9 + .../benches/lance_cache.rs | 24 + crates/persisting-pchronicle/src/revision.rs | 5 +- crates/persisting-pchronicle/src/storage.rs | 16 + .../src/store/blockcache/adapter.rs | 422 ++++++++++++++++++ .../src/store/blockcache/block.rs | 201 +++++++++ .../src/store/blockcache/config.rs | 51 +++ .../src/store/blockcache/mod.rs | 14 + .../src/store/catalog/discovery.rs | 2 +- .../src/store/catalog/manifest_cache.rs | 98 ++-- .../src/store/compact_jsonl.rs | 19 +- .../src/store/events/mod.rs | 4 +- .../src/store/inspect.rs | 6 +- crates/persisting-pchronicle/src/store/mod.rs | 2 + .../src/store/object_store_io_gate.rs | 330 ++++++++++++-- .../src/store/opendal_store.rs | 354 ++++++++++----- .../src/store/persistent_cache.rs | 40 +- .../src/store/storyline/mod.rs | 24 +- .../src/store/storyline/mutation.rs | 1 + 25 files changed, 1708 insertions(+), 350 deletions(-) create mode 100644 crates/persisting-pchronicle/benches/lance_cache.rs create mode 100644 crates/persisting-pchronicle/src/store/blockcache/adapter.rs create mode 100644 crates/persisting-pchronicle/src/store/blockcache/block.rs create mode 100644 crates/persisting-pchronicle/src/store/blockcache/config.rs create mode 100644 crates/persisting-pchronicle/src/store/blockcache/mod.rs diff --git a/Cargo.lock b/Cargo.lock index aec661e0d..2a5f13ba3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7254,6 +7254,7 @@ dependencies = [ "async-trait", "base64 0.22.1", "blake3", + "bytes", "chrono", "criterion", "datafusion", @@ -7266,9 +7267,11 @@ dependencies = [ "lance-datafusion", "lance-file", "lance-index", + "lance-io", "lance-linalg", "lance-table", "libc", + "object_store", "opendal 0.57.0", "persisting-agentctl", "persisting-events", diff --git a/Cargo.toml b/Cargo.toml index 4a6e19ae5..480791773 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -60,6 +60,7 @@ lance-core = "11.0.0" lance-file = "11.0.0" lance-datafusion = "11.0.0" lance-index = { version = "11.0.0", features = ["tokenizer-jieba"] } +lance-io = "11.0.0" lance-linalg = "11.0.0" lance-table = "11.0.0" libc = "0.2" @@ -75,6 +76,7 @@ opendal = { version = "0.57.0", default-features = true, features = [ "services-s3", "services-tos", ] } +object_store = "0.13.2" persisting-agentctl = { path = "crates/persisting-agentctl" } persisting-events = { path = "crates/persisting-events" } persisting-gateway = { path = "crates/persisting-gateway" } diff --git a/crates/persisting-pchronicle-cli/src/gateway_partition.rs b/crates/persisting-pchronicle-cli/src/gateway_partition.rs index 47c31dbf2..bdf2b329a 100644 --- a/crates/persisting-pchronicle-cli/src/gateway_partition.rs +++ b/crates/persisting-pchronicle-cli/src/gateway_partition.rs @@ -1,7 +1,7 @@ //! Safe, bounded physical partitioning below one logical Gateway Dataset. use std::collections::HashMap; -use std::sync::Mutex; +use std::sync::{Mutex, MutexGuard}; use anyhow::{Context, Result}; use chrono::{DateTime, Datelike, Timelike, Utc}; @@ -10,6 +10,12 @@ const MAX_TEMPLATE_BYTES: usize = 256; const MAX_TEMPLATE_SEGMENTS: usize = 16; const MAX_USER_SEGMENT_BYTES: usize = 80; +fn lock_recover(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) +} + #[derive(Debug, Clone, PartialEq, Eq)] enum Segment { Literal(String), @@ -149,7 +155,7 @@ impl GatewayPartitionRouter { let Some(split) = &self.split else { return self.dataset_uri.clone(); }; - let mut routes = self.routes.lock().unwrap(); + let mut routes = lock_recover(&self.routes); routes .entry(route_key.to_string()) .or_insert_with(|| { diff --git a/crates/persisting-pchronicle-cli/src/server/query_admission.rs b/crates/persisting-pchronicle-cli/src/server/query_admission.rs index 3c2002a1a..17938fc03 100644 --- a/crates/persisting-pchronicle-cli/src/server/query_admission.rs +++ b/crates/persisting-pchronicle-cli/src/server/query_admission.rs @@ -3,7 +3,7 @@ use std::collections::HashMap; use std::future::Future; -use std::sync::{Arc, Mutex, Weak}; +use std::sync::{Arc, Mutex, MutexGuard, Weak}; use std::time::{Duration, Instant}; use anyhow::Result; @@ -20,6 +20,12 @@ const MAX_CACHE_BYTES: usize = 64 * 1024 * 1024; // One background summary refresh per process, with no waiting task queue. pub(super) static REFRESH_SLOT: Semaphore = Semaphore::const_new(1); +fn lock_recover(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) +} + struct CachedSummaries { summaries: Summaries, built_at: Instant, @@ -83,16 +89,16 @@ impl Default for ScopedQueries { impl ScopedQueries { pub(super) fn invalidate(&self) { - let mut cache = self.cache.lock().unwrap(); + let mut cache = lock_recover(&self.cache); cache.generation += 1; cache.entries.clear(); // Requests started after explicit refresh must not join an older build. - self.flights.lock().unwrap().clear(); + lock_recover(&self.flights).clear(); } fn publish(&self, scope: QueryScope, generation: u64, summaries: Summaries) { let bytes = summary_bytes(&summaries); - let mut cache = self.cache.lock().unwrap(); + let mut cache = lock_recover(&self.cache); if cache.generation != generation { return; } @@ -140,7 +146,7 @@ impl ScopedQueries { Fut: Future> + Send + 'static, { let (generation, cached) = { - let mut cache = self.cache.lock().unwrap(); + let mut cache = lock_recover(&self.cache); let generation = cache.generation; let cached = cache.entries.get_mut(&scope).map(|entry| { entry.last_used = Instant::now(); @@ -163,7 +169,7 @@ impl ScopedQueries { match queries.run(scope.clone(), || execute(true)).await { Ok(summaries) => queries.publish(scope, generation, summaries), Err(error) => { - let mut cache = queries.cache.lock().unwrap(); + let mut cache = lock_recover(&queries.cache); if cache.generation == generation && let Some(entry) = cache.entries.get_mut(&scope) { @@ -195,7 +201,7 @@ impl ScopedQueries { Fut: Future>, { let flight = { - let mut flights = self.flights.lock().unwrap(); + let mut flights = lock_recover(&self.flights); flights.retain(|_, entry| entry.strong_count() > 0); if let Some(flight) = flights.get(&scope).and_then(Weak::upgrade) { flight @@ -234,7 +240,7 @@ mod tests { } fn expire(queries: &ScopedQueries, key: &QueryScope) { - let mut cache = queries.cache.lock().unwrap(); + let mut cache = lock_recover(&queries.cache); let entry = cache.entries.get_mut(key).unwrap(); entry.built_at = Instant::now() - REFRESH_INTERVAL; entry.refresh_after = Instant::now(); diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index ce9bf9c29..61316a478 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -3,7 +3,7 @@ use std::collections::{HashMap, HashSet, VecDeque}; use std::path::PathBuf; -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Mutex, MutexGuard}; use std::time::{Duration, Instant}; use anyhow::{Context, Result}; @@ -17,9 +17,10 @@ use super::explorer::{CatalogTree, catalog_tree_from_mount_specs, catalog_tree_f const REFRESH_INTERVAL: Duration = Duration::from_secs(30); const QUEUE_CAPACITY: usize = 128; -// ponytail: one process-wide browse scan at a time; per-backend budgets if -// multiple independent stores need more throughput. This does not gate SQL. -static BROWSE_IO: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); +// Keep browse work bounded while allowing a foreground request to run beside +// one background walk. +static BROWSE_IO: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(2); +static BACKGROUND_IO: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(1); #[derive(Clone, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] struct TreeKey { @@ -108,6 +109,12 @@ struct RefreshState { type Reply = oneshot::Sender>; type Pending = Arc>>>; +fn lock_recover(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) +} + /// One bounded worker per serve instance, shared by timer and HTTP requests. /// The task owns the index, not the coordinator; dropping the last AppState /// aborts it, including an in-flight list. No permanent process singleton. @@ -210,11 +217,8 @@ impl BrowseCoordinator { for mount in mounts { let key = TreeKey::new(mount, "").expect("root prefix"); let _ = self.enqueue(&key, None); - refreshing |= self.pending.lock().unwrap().contains_key(&key); - if let Some(failure) = self - .refresh - .lock() - .unwrap() + refreshing |= lock_recover(&self.pending).contains_key(&key); + if let Some(failure) = lock_recover(&self.refresh) .get(&key) .and_then(|s| s.error.clone()) { @@ -259,6 +263,8 @@ impl BrowseCoordinator { consistency: CatalogConsistency::BestEffort, state: if refreshing { CatalogState::Refreshing + } else if !complete && error.is_some() { + CatalogState::Unavailable } else if !complete || error.is_some() || now() - observed_at >= 30 { CatalogState::Stale } else { @@ -268,7 +274,7 @@ impl BrowseCoordinator { .to_hex() .to_string(), observed_at: if complete { observed_at } else { 0 }, - stale: !complete || error.is_some() || now() - observed_at >= 30, + stale: complete && (error.is_some() || now() - observed_at >= 30), refreshing, last_error: error, }, @@ -289,27 +295,26 @@ impl BrowseCoordinator { let _ = self.enqueue(&key, None); return Ok(self.snapshot_with_summary(&key, entry).await); } - let (reply, wait) = oneshot::channel(); - self.enqueue(&key, Some(reply))?; - // Bound cold requests even if many prefixes precede them in the queue. - tokio::time::timeout(Duration::from_secs(30), wait) - .await - .context("browse refresh timed out")? - .context("browse worker stopped")? - .map_err(anyhow::Error::msg)?; - let entry = self - .index - .values - .read() - .await - .get(&key) - .cloned() - .context("browse refresh produced no view")?; - Ok(self.snapshot_with_summary(&key, entry).await) + self.enqueue(&key, None)?; + // Cold browse requests return an empty, explicitly refreshing view; + // the worker fills the index asynchronously instead of blocking HTTP. + let tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]); + Ok(BrowseSnapshot { + tree, + browse: BrowseStatus { + consistency: CatalogConsistency::BestEffort, + state: CatalogState::Refreshing, + generation: String::new(), + observed_at: 0, + stale: true, + refreshing: true, + last_error: None, + }, + }) } fn enqueue(&self, key: &TreeKey, reply: Option) -> Result<()> { - if let Some(state) = self.refresh.lock().unwrap().get(key) + if let Some(state) = lock_recover(&self.refresh).get(key) && state .retry_at .is_some_and(|deadline| deadline > Instant::now()) @@ -319,7 +324,7 @@ impl BrowseCoordinator { } return Ok(()); } - let mut pending = self.pending.lock().unwrap(); + let mut pending = lock_recover(&self.pending); if let Some(waiters) = pending.get_mut(key) { waiters.retain(|reply| !reply.is_closed()); if let Some(reply) = reply { @@ -336,17 +341,14 @@ impl BrowseCoordinator { } fn snapshot(&self, key: &TreeKey, entry: IndexEntry) -> BrowseSnapshot { - let error = self - .refresh - .lock() - .unwrap() + let error = lock_recover(&self.refresh) .get(key) .and_then(|s| s.error.clone()); BrowseSnapshot { tree: entry.tree, browse: BrowseStatus { consistency: CatalogConsistency::BestEffort, - state: if self.pending.lock().unwrap().contains_key(key) { + state: if lock_recover(&self.pending).contains_key(key) { CatalogState::Refreshing } else if error.is_some() || now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64 @@ -359,7 +361,7 @@ impl BrowseCoordinator { observed_at: entry.observed_at, stale: error.is_some() || now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64, - refreshing: self.pending.lock().unwrap().contains_key(key), + refreshing: lock_recover(&self.pending).contains_key(key), last_error: error, }, } @@ -502,118 +504,159 @@ async fn run_worker( ) { let mut interval = tokio::time::interval(REFRESH_INTERVAL); interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - // Incremental round over roots and previously browsed prefixes. Foreground - // requests are checked between each scan, not after the entire mount set. - // Breadth-first keeps shallow datasets visible while a deep subtree is - // still being indexed. let mut background = VecDeque::new(); let mut visited = HashSet::new(); + let mut background_budget = 0usize; + let mut background_running = false; + // JoinSet aborts outstanding work when the coordinator is dropped. + let mut jobs = tokio::task::JoinSet::<(TreeKey, bool, Result)>::new(); loop { - let key = tokio::select! { + let (key, is_background) = tokio::select! { biased; - _ = std::future::ready(()), if !background.is_empty() => background.pop_front().unwrap(), - request = receiver.recv() => match request { Some(key) => key, None => break }, + completed = jobs.join_next(), if !jobs.is_empty() => { + let (key, was_background, result) = match completed.unwrap() { + Ok(completed) => completed, + Err(error) => { + tracing::error!(target: "pchronicle.serve", %error, "browse refresh task stopped"); + for (_, waiters) in lock_recover(&pending).drain() { + for waiter in waiters { + let _ = waiter.send(Err(format!("browse refresh task stopped: {error}"))); + } + } + return; + } + }; + if was_background { background_running = false; } + let outcome = match result { + Ok(tree) => { + visited.insert(key.clone()); + for child in &tree.children { + if child.kind == "dir" && visited.len() < 10_000 { + let next = TreeKey { prefix: child.path.clone(), ..key.clone() }; + if visited.insert(next.clone()) { background.push_back(next); } + } + } + lock_recover(&states).insert(key.clone(), RefreshState { + retry_at: Some(Instant::now() + Duration::from_secs(2)), + ..Default::default() + }); + Ok(()) + } + Err(error) => { + let error = format!("{error:#}"); + let cached_view = index.values.read().await.contains_key(&key); + let mut states = lock_recover(&states); + let state = states.entry(key.clone()).or_default(); + state.failures = state.failures.saturating_add(1); + state.retry_at = Some(Instant::now() + Duration::from_secs( + (30u64 * (1u64 << state.failures.min(4))).min(300))); + state.error = Some(error.clone()); + tracing::warn!(target: "pchronicle.serve", dataset = %key.dataset, + prefix = %key.prefix, cached_view, %error, "browse refresh failed"); + Err(error) + } + }; + finish(&pending, &key, outcome); + continue; + } + request = receiver.recv(), if jobs.len() < 2 => match request { + Some(key) => (key, false), None => break + }, + _ = std::future::ready(()), if !background.is_empty() + && background_budget > 0 && !background_running && jobs.len() < 2 => { + background_budget -= 1; + (background.pop_front().unwrap(), true) + }, _ = interval.tick() => { - tracing::info!(target: "pchronicle.serve", mounts = mounts.len(), "browse manifest refresh round started"); - if background.is_empty() { + if background.is_empty() && !background_running { visited.clear(); background.extend(mounts.iter().filter_map(|m| TreeKey::new(m, "").ok())); visited.extend(background.iter().cloned()); } + background_budget = 32; continue; } }; let Some(mount) = mounts .iter() .find(|m| TreeKey::new(m, &key.prefix).ok().as_ref() == Some(&key)) + .cloned() else { continue; }; - if states - .lock() - .unwrap() + // A queued foreground request or an active job already owns this key. + // Leave its waiters attached; never start a duplicate background scan. + if is_background && lock_recover(&pending).contains_key(&key) { + continue; + } + if lock_recover(&states) .get(&key) .is_some_and(|s| s.retry_at.is_some_and(|deadline| deadline > Instant::now())) { finish(&pending, &key, Ok(())); continue; } - pending.lock().unwrap().entry(key.clone()).or_default(); - let result = async { - let _permit = BROWSE_IO.acquire().await?; - // Bound prefix-list start rate as well as concurrency. One list may - // contain several storage requests; backend I/O gates still apply. - tokio::time::sleep(Duration::from_millis(100)).await; - let location = DatasetLocation::parse(&mount.uri)?; - if let Some(root) = location.local_path() { - // Navigation's list API treats a missing path as empty. For a - // cached view that could erase an offline mount's descendants. - // Deletions are instead established by a successful parent list. - tokio::fs::metadata(root.join(&key.prefix)) - .await - .context("browse path unavailable")?; - } - let manifest_key = if key.prefix.is_empty() { - format!("{}\0{}", key.dataset, key.uri_fingerprint) + lock_recover(&pending).entry(key.clone()).or_default(); + if is_background { + background_running = true; + } + let index = index.clone(); + let manifests = manifests.clone(); + jobs.spawn(async move { + // Reserve at least one global browse slot for foreground work. + let _background = if is_background { + BACKGROUND_IO.acquire().await.ok() } else { - format!("{}\0{}\0{}", key.dataset, key.uri_fingerprint, key.prefix) + None }; - let listing = tokio::time::timeout( - Duration::from_secs(20), - manifests.refresh(manifest_key, &location, &key.prefix), - ) + let result = refresh_tree(&mount, &key, &index, &manifests).await; + (key, is_background, result) + }); + } +} + +async fn refresh_tree( + mount: &DatasetMount, + key: &TreeKey, + index: &BrowseTreeProjection, + manifests: &ManifestCache, +) -> Result { + let _permit = BROWSE_IO.acquire().await?; + #[cfg(test)] + let block = lock_recover(&index.refresh_blocks).get(key).cloned(); + #[cfg(test)] + let _block = if let Some(block) = block { + Some(block.acquire_owned().await?) + } else { + None + }; + let location = DatasetLocation::parse(&mount.uri)?; + if location.local_path().is_none() { + tokio::time::sleep(Duration::from_millis(100)).await; + } + if let Some(root) = location.local_path() { + tokio::fs::metadata(root.join(&key.prefix)) .await - .context("browse list timed out")??; - let entries = listing.entries; - let tree = catalog_tree_from_path_list(&mount.name, &key.prefix, &entries); - // Walk only navigational directories; a Dataset leaf is opaque. - // A bounded frontier prevents the background walk growing without limit. - visited.insert(key.clone()); - for child in &tree.children { - if child.kind == "dir" && visited.len() < 10_000 { - let next = TreeKey::new(mount, &child.path)?; - if visited.insert(next.clone()) { - background.push_back(next); - } - } - } - index.put(key.clone(), tree).await; - Ok::<_, anyhow::Error>(()) - } - .await; - let outcome = match result { - Ok(()) => { - tracing::info!(target: "pchronicle.serve", dataset = %key.dataset, prefix = %key.prefix, "browse manifest level refreshed"); - states.lock().unwrap().insert( - key.clone(), - RefreshState { - retry_at: Some(Instant::now() + Duration::from_secs(2)), - ..Default::default() - }, - ); - Ok(()) - } - Err(error) => { - let error = format!("{error:#}"); - let cached_view = index.values.read().await.contains_key(&key); - let mut states = states.lock().unwrap(); - let state = states.entry(key.clone()).or_default(); - state.failures = state.failures.saturating_add(1); - state.retry_at = Some( - Instant::now() - + Duration::from_secs((30u64 * (1u64 << state.failures.min(4))).min(300)), - ); - state.error = Some(error.clone()); - tracing::warn!(target: "pchronicle.serve", dataset = %key.dataset, prefix = %key.prefix, cached_view, error = %error, "browse refresh failed"); - Err(error) - } - }; - finish(&pending, &key, outcome); + .context("browse path unavailable")?; } + let manifest_key = if key.prefix.is_empty() { + format!("{}\0{}", key.dataset, key.uri_fingerprint) + } else { + format!("{}\0{}\0{}", key.dataset, key.uri_fingerprint, key.prefix) + }; + let listing = tokio::time::timeout( + Duration::from_secs(20), + manifests.refresh(manifest_key, &location, &key.prefix), + ) + .await + .context("browse list timed out")??; + let tree = catalog_tree_from_path_list(&mount.name, &key.prefix, &listing.entries); + index.put(key.clone(), tree.clone()).await; + Ok(tree) } fn finish(pending: &Pending, key: &TreeKey, outcome: std::result::Result<(), String>) { - if let Some(waiters) = pending.lock().unwrap().remove(key) { + if let Some(waiters) = lock_recover(&pending).remove(key) { for waiter in waiters { let _ = waiter.send(outcome.clone()); } @@ -626,6 +669,8 @@ fn now() -> i64 { /// UI-only tree projection over the core ManifestCache. It preserves the legacy tree wire format. struct BrowseTreeProjection { + #[cfg(test)] + refresh_blocks: Mutex>>, disk: Arc, values: RwLock>, } @@ -647,6 +692,8 @@ impl BrowseTreeProjection { Self { disk, values: RwLock::new(values), + #[cfg(test)] + refresh_blocks: Mutex::new(HashMap::new()), } } @@ -727,6 +774,60 @@ mod tests { DatasetMount::new("test", path.to_string_lossy()).unwrap() } + #[tokio::test] + async fn foreground_finishes_while_background_is_blocked_and_drop_cancels_work() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + std::fs::create_dir_all(source.join("nested")).unwrap(); + let mount = mount(&source); + // Construct before starting the worker so the root background scan is + // deterministically blocked while a child foreground request arrives. + let index = Arc::new(BrowseTreeProjection::open(temp.path().join("tree.lance")).await); + let root_key = TreeKey::new(&mount, "").unwrap(); + let blocker = Arc::new(tokio::sync::Semaphore::new(0)); + lock_recover(&index.refresh_blocks).insert(root_key.clone(), blocker.clone()); + let manifests = Arc::new(ManifestCache::open(temp.path().join("manifest.lance")).await); + let pending = Arc::new(Mutex::new(HashMap::new())); + let states = Arc::new(Mutex::new(HashMap::new())); + let (sender, receiver) = mpsc::channel(QUEUE_CAPACITY); + let task = tokio::spawn(run_worker( + vec![mount.clone()], + index, + manifests, + pending.clone(), + states, + receiver, + )); + tokio::time::timeout(Duration::from_secs(5), async { + while !lock_recover(&pending).contains_key(&root_key) { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let child = TreeKey::new(&mount, "nested").unwrap(); + let (reply, wait) = oneshot::channel(); + lock_recover(&pending).insert(child.clone(), vec![reply]); + sender.send(child).await.unwrap(); + tokio::time::timeout(Duration::from_secs(5), wait) + .await + .unwrap() + .unwrap() + .unwrap(); + assert!( + lock_recover(&pending).contains_key(&root_key), + "background must still be blocked" + ); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + // Aborting the scheduler drops its JoinSet and releases active I/O. + let permits = tokio::time::timeout(Duration::from_secs(5), BROWSE_IO.acquire_many(2)) + .await + .unwrap() + .unwrap(); + drop(permits); + } + #[tokio::test] async fn directory_statistics_use_cached_descendants_without_child_projections() { let temp = tempfile::tempdir().unwrap(); @@ -864,21 +965,21 @@ mod tests { let coordinator = BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; // Hold the global I/O permit to make duplicate cold requests observable. - let gate = BROWSE_IO.acquire().await.unwrap(); + let gate = BROWSE_IO.acquire_many(2).await.unwrap(); let key = TreeKey::new(&mount, "").unwrap(); let (a, ar) = oneshot::channel(); let (b, br) = oneshot::channel(); coordinator.enqueue(&key, Some(a)).unwrap(); coordinator.enqueue(&key, Some(b)).unwrap(); - assert_eq!(coordinator.pending.lock().unwrap().len(), 1); - assert_eq!(coordinator.pending.lock().unwrap()[&key].len(), 2); + assert_eq!(lock_recover(&coordinator.pending).len(), 1); + assert_eq!(lock_recover(&coordinator.pending)[&key].len(), 2); drop(gate); ar.await.unwrap().unwrap(); br.await.unwrap().unwrap(); let old = coordinator.tree(&mount, "").await.unwrap(); assert_eq!(old.tree.children[0].name, "nested"); std::fs::remove_dir_all(&source).unwrap(); - coordinator.refresh.lock().unwrap().remove(&key); + lock_recover(&coordinator.refresh).remove(&key); let (reply, wait) = oneshot::channel(); coordinator.enqueue(&key, Some(reply)).unwrap(); assert!(wait.await.unwrap().is_err()); diff --git a/crates/persisting-pchronicle-cli/src/sync.rs b/crates/persisting-pchronicle-cli/src/sync.rs index d0cb1e84c..74f5e4d44 100644 --- a/crates/persisting-pchronicle-cli/src/sync.rs +++ b/crates/persisting-pchronicle-cli/src/sync.rs @@ -174,13 +174,7 @@ pub(crate) async fn run( let mut pending = BTreeSet::new(); let mut failures = 0u32; loop { - tokio::time::sleep(interval).await; - while let Ok(path) = changes_rx.try_recv() { - pending.insert(path); - } - if pending.is_empty() { - continue; - } + collect_pending_changes(&mut changes_rx, &mut pending).await?; match super::exchange::sync_snapshot( &source_uri, @@ -239,6 +233,21 @@ pub(crate) async fn run( } } +// A failed batch remains pending. Only idle syncs wait for a new change; +// retries also absorb any changes received during backoff. +async fn collect_pending_changes( + changes: &mut tokio::sync::mpsc::Receiver, + pending: &mut BTreeSet, +) -> Result<()> { + if pending.is_empty() { + pending.insert(changes.recv().await.context("sync watcher stopped")?); + } + while let Ok(path) = changes.try_recv() { + pending.insert(path); + } + Ok(()) +} + fn prepare_destination(uri: &str, name: &str) -> Result { anyhow::ensure!(!uri.is_empty(), "sync {name} target must not be empty"); let location = DatasetLocation::parse(uri)?; @@ -348,6 +357,41 @@ fn changed_paths( mod tests { use super::*; + #[tokio::test] + async fn failed_sync_retries_without_new_changes_and_coalesces_backoff_changes() { + let (sender, mut receiver) = tokio::sync::mpsc::channel(4); + sender.send(PathBuf::from("a.json")).await.unwrap(); + let mut pending = BTreeSet::new(); + collect_pending_changes(&mut receiver, &mut pending) + .await + .unwrap(); + // Simulate a failed attempt: retain the batch, with no new file event. + tokio::time::timeout( + Duration::from_millis(100), + collect_pending_changes(&mut receiver, &mut pending), + ) + .await + .unwrap() + .unwrap(); + sender.send(PathBuf::from("b.json")).await.unwrap(); + collect_pending_changes(&mut receiver, &mut pending) + .await + .unwrap(); + assert_eq!( + pending, + BTreeSet::from([PathBuf::from("a.json"), PathBuf::from("b.json")]) + ); + pending.clear(); + assert!( + tokio::time::timeout( + Duration::from_millis(20), + collect_pending_changes(&mut receiver, &mut pending) + ) + .await + .is_err() + ); + } + #[test] fn prepare_destination_preserves_object_store_uri() { assert_eq!( diff --git a/crates/persisting-pchronicle/Cargo.toml b/crates/persisting-pchronicle/Cargo.toml index 48bb277a3..ba813798d 100644 --- a/crates/persisting-pchronicle/Cargo.toml +++ b/crates/persisting-pchronicle/Cargo.toml @@ -18,7 +18,9 @@ lance-store = [ "dep:lance-datafusion", "dep:lance-file", "dep:lance-index", + "dep:lance-io", "dep:opendal", + "dep:object_store", ] # Remote object stores are separated from the local Lance engine so pPilot and # other local-only consumers do not compile cloud SDKs. `s3-store` remains a @@ -40,6 +42,7 @@ proptest = ["lance-store"] [dependencies] anyhow.workspace = true async-trait.workspace = true +bytes.workspace = true base64.workspace = true blake3.workspace = true chrono.workspace = true @@ -53,9 +56,11 @@ lance-core = { workspace = true, optional = true } lance-file = { workspace = true, optional = true } lance-datafusion = { workspace = true, optional = true } lance-index = { workspace = true, optional = true } +lance-io = { workspace = true, optional = true } lance-linalg = { workspace = true, optional = true } lance-table = { workspace = true, optional = true } opendal = { workspace = true, optional = true } +object_store = { workspace = true, optional = true } persisting-agentctl.workspace = true persisting-events.workspace = true serde = { workspace = true, features = ["derive"] } @@ -375,3 +380,7 @@ required-features = ["proptest"] name = "proptest_unknown_fields_roundtrip" path = "tests/proptests/unknown_fields_roundtrip.rs" required-features = ["proptest"] + +[[bench]] +name = "lance_cache" +harness = false diff --git a/crates/persisting-pchronicle/benches/lance_cache.rs b/crates/persisting-pchronicle/benches/lance_cache.rs new file mode 100644 index 000000000..72641f012 --- /dev/null +++ b/crates/persisting-pchronicle/benches/lance_cache.rs @@ -0,0 +1,24 @@ +use bytes::Bytes; +use criterion::{Criterion, criterion_group, criterion_main}; +use persisting_pchronicle::storage::{BlockCache, CacheConfig}; +use std::hint::black_box; +use tempfile::tempdir; + +fn block_cache(c: &mut Criterion) { + let rt = tokio::runtime::Runtime::new().unwrap(); + let dir = tempdir().unwrap(); + let cache = BlockCache::new(CacheConfig::new( + dir.path().into(), + 64 * 1024 * 1024, + 64 * 1024, + )); + let path = dir.path().join("block"); + rt.block_on(cache.get_or_fetch(&path, 4, async { Ok(Bytes::from_static(b"data")) })) + .unwrap(); + c.bench_function("block_cache_hit", |b| { + b.iter(|| rt.block_on(cache.get_or_fetch(black_box(&path), 4, async { unreachable!() }))) + }); +} + +criterion_group!(benches, block_cache); +criterion_main!(benches); diff --git a/crates/persisting-pchronicle/src/revision.rs b/crates/persisting-pchronicle/src/revision.rs index b59fe6ee9..411e2d9f9 100644 --- a/crates/persisting-pchronicle/src/revision.rs +++ b/crates/persisting-pchronicle/src/revision.rs @@ -4,7 +4,6 @@ use std::sync::Arc; use anyhow::{Context, Result}; use futures::TryStreamExt; -use lance::Dataset; use lance::dataset::{InsertBuilder, MergeInsertBuilder, WhenMatched, WhenNotMatched}; use lance::deps::arrow_array::{Array, RecordBatch, RecordBatchIterator, StringArray}; use lance::deps::arrow_schema::{DataType, Field, Schema}; @@ -94,7 +93,7 @@ fn text(batch: &RecordBatch, column: &str, row: usize) -> Result { pub async fn read_revisions(session: &StoryCoords) -> Result> { let uri = revision_dataset_path(session)?; - let dataset = match Dataset::open(&uri).await { + let dataset = match crate::storage::open_lance_dataset(&uri).await { Ok(dataset) => dataset, Err(lance::Error::DatasetNotFound { .. }) => return Ok(Vec::new()), Err(error) => return Err(anyhow::anyhow!(error)).context("open revisions.lance"), @@ -139,7 +138,7 @@ pub async fn write_revisions(session: &StoryCoords, rows: &[RevisionRow]) -> Res } let _guard = crate::store::dataset_write_lock::acquire(&uri).await?; let batch = batch(rows)?; - match Dataset::open(&uri).await { + match crate::storage::open_lance_dataset(&uri).await { Ok(dataset) => { let reader = Box::new(RecordBatchIterator::new(vec![Ok(batch)], schema())); MergeInsertBuilder::try_new(Arc::new(dataset), vec!["revision_id".into()])? diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 131b7afcf..7a0133461 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -49,6 +49,12 @@ pub use crate::discovery::{ drop_lifecycle_run_partitions, expand_story_locations, expand_story_locations_blocking, }; +#[cfg(feature = "lance-store")] +pub use crate::store::blockcache::{ + BlockCache, CacheConfig, CacheStats, CachedObjectStore, DEFAULT_BLOCK_SIZE_BYTES, + DEFAULT_CAPACITY_BYTES, LanceCacheWrapper, SERVE_CAPACITY_BYTES, capacity_for_serve, + configured_capacity_bytes, default_cache_dir, lance_store_params, +}; #[cfg(feature = "lance-store")] pub use crate::store::index_build_progress::{ Guard as IndexBuildProgressGuard, install as install_index_build_progress, @@ -61,6 +67,16 @@ pub use crate::store::object_store_io_gate::{ snapshot as object_store_gate_snapshot, }; +#[cfg(feature = "lance-store")] +pub async fn open_lance_dataset(uri: &str) -> lance::Result { + lance::dataset::builder::DatasetBuilder::from_uri(uri) + .with_store_params(crate::store::blockcache::lance_store_params( + crate::store::blockcache::configured_capacity_bytes(), + )) + .load() + .await +} + #[cfg(feature = "lance-store")] pub use crate::store::{ AppendOutcome, AttemptRecord, AttemptRecordState, AttemptRegistry, CachedDataset, diff --git a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs new file mode 100644 index 000000000..5a9db89fb --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs @@ -0,0 +1,422 @@ +//! Lance bridge and ObjectStore read-through adapter. Only validated range reads +//! are cached; HEAD and full GET retain the backend's native semantics/stream. + +use super::{BlockCache, CacheConfig}; +use crate::store::object_store_io_gate::{self as io_gate, IoKind}; +use async_trait::async_trait; +use bytes::Bytes; +use futures::{StreamExt, stream::BoxStream}; +use object_store::{ + CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, ObjectMeta, + ObjectStore, PutMultipartOptions, PutOptions, PutPayload, PutResult, Result as ObjectResult, + UploadPart, path::Path, +}; +use std::{ops::Range, sync::Arc}; + +/// Admission and feedback live at the object-request boundary. The backend +/// already retries transport errors; do not replay an entire Lance open here. +async fn remote_request( + uri: &str, + request: impl std::future::Future>, +) -> ObjectResult { + let _permit = io_gate::acquire(uri, IoKind::Read).await; + let result = request.await; + match &result { + Ok(_) => io_gate::note_success(uri), + Err(error) if io_gate::is_transient_error(error) => { + io_gate::note_failure(uri, IoKind::Read) + } + Err(_) => {} + } + result +} + +async fn remote_write_request( + uri: &str, + request: impl std::future::Future>, +) -> ObjectResult { + let _permit = io_gate::acquire(uri, IoKind::Write).await; + let result = request.await; + match &result { + Ok(_) => io_gate::note_success(uri), + Err(error) if io_gate::is_transient_error(error) => { + io_gate::note_failure(uri, IoKind::Write) + } + Err(_) => {} + } + result +} + +struct GatedMultipartUpload { + inner: Arc>>, + io_scope: String, +} + +impl std::fmt::Debug for GatedMultipartUpload { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("GatedMultipartUpload") + .field("io_scope", &self.io_scope) + .finish() + } +} + +#[async_trait] +impl MultipartUpload for GatedMultipartUpload { + fn put_part(&mut self, data: PutPayload) -> UploadPart { + let inner = Arc::clone(&self.inner); + let scope = self.io_scope.clone(); + Box::pin(async move { + // ponytail: serialize multipart parts because object_store's trait + // returns a 'static future from &mut self; parallelism can be added + // when the trait exposes owned part handles. + let mut inner = inner.lock().await; + remote_write_request(&scope, inner.put_part(data)).await + }) + } + + async fn complete(&mut self) -> ObjectResult { + let mut inner = self.inner.lock().await; + remote_write_request(&self.io_scope, inner.complete()).await + } + + async fn abort(&mut self) -> ObjectResult<()> { + let mut inner = self.inner.lock().await; + remote_write_request(&self.io_scope, inner.abort()).await + } +} + +#[derive(Debug, Clone)] +pub struct LanceCacheWrapper { + pub capacity_bytes: u64, + pub block_size_bytes: u64, + pub root: std::path::PathBuf, +} + +fn backend_identity_from_env() -> String { + let backend: std::collections::BTreeMap<_, _> = std::env::vars() + .filter(|(key, _)| { + key.starts_with("AWS_") || key.starts_with("AZURE_") || key.starts_with("GOOGLE_") + }) + .collect(); + blake3::hash(serde_json::to_string(&backend).unwrap().as_bytes()) + .to_hex() + .to_string() +} + +impl lance_io::object_store::WrappingObjectStore for LanceCacheWrapper { + fn wrap(&self, prefix: &str, original: Arc) -> Arc { + // Lance uses e.g. s3$bucket, not a URI. Convert only its scheme delimiter. + let uri = prefix.replacen('$', "://", 1); + if !io_gate::is_remote_uri(&uri) { + return original; + } + let config = CacheConfig::new( + self.root.clone(), + self.capacity_bytes, + self.block_size_bytes, + ); + Arc::new(CachedObjectStore::new( + original, + config, + format!("{uri}#{}", backend_identity_from_env()), + )) + } +} + +pub fn lance_wrapper(capacity_bytes: u64) -> Arc { + let mut config = CacheConfig::from_env(); + config.capacity_bytes = capacity_bytes; + Arc::new(LanceCacheWrapper { + capacity_bytes, + block_size_bytes: config.block_size_bytes, + root: config.root, + }) +} + +pub fn lance_store_params(capacity_bytes: u64) -> lance_io::object_store::ObjectStoreParams { + lance_io::object_store::ObjectStoreParams { + object_store_wrapper: Some(lance_wrapper(capacity_bytes)), + ..Default::default() + } +} + +#[derive(Debug, Clone)] +pub struct CachedObjectStore { + inner: Arc, + cache: BlockCache, + block_size_bytes: u64, + store_uri: String, + io_scope: String, +} + +impl CachedObjectStore { + /// `store_uri` is the cache namespace: include the backend/credential scope + /// as well as bucket. Use a digest for any sensitive configuration. + pub fn new(inner: Arc, config: CacheConfig, store_uri: String) -> Self { + let block_size_bytes = config.block_size_bytes; + Self { + inner, + cache: BlockCache::new(config), + block_size_bytes, + io_scope: io_gate::scope_key(&store_uri), + store_uri, + } + } + + async fn read_block( + &self, + path: &Path, + meta: &ObjectMeta, + options: GetOptions, + key: &str, + range: Range, + ) -> ObjectResult { + let block = range.start / self.block_size_bytes; + let file = self.cache.block_path(key, block); + self.cache + .get_or_fetch(&file, (range.end - range.start) as usize, async { + remote_request(&self.io_scope, async { + let result = self.inner.get_opts(path, options.clone()).await?; + // Do not publish mismatched bytes even if a compatible backend + // ignores If-Match or the requested VersionId. + if result.range != range + || result.meta.size != meta.size + || options + .version + .as_ref() + .is_some_and(|v| result.meta.version.as_ref() != Some(v)) + || options + .if_match + .as_ref() + .is_some_and(|tag| result.meta.e_tag.as_ref() != Some(tag)) + { + return Err(object_store::Error::Precondition { + path: path.to_string(), + source: "object changed while reading cached block".into(), + }); + } + result.bytes().await + }) + .await + }) + .await + } +} + +impl std::fmt::Display for CachedObjectStore { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "cached({})", self.inner) + } +} + +#[async_trait] +impl ObjectStore for CachedObjectStore { + async fn put_opts(&self, p: &Path, b: PutPayload, o: PutOptions) -> ObjectResult { + remote_write_request(&self.io_scope, self.inner.put_opts(p, b, o)).await + } + async fn put_multipart_opts( + &self, + p: &Path, + o: PutMultipartOptions, + ) -> ObjectResult> { + let upload = + remote_write_request(&self.io_scope, self.inner.put_multipart_opts(p, o)).await?; + Ok(Box::new(GatedMultipartUpload { + inner: Arc::new(tokio::sync::Mutex::new(upload)), + io_scope: self.io_scope.clone(), + })) + } + async fn get_opts(&self, p: &Path, o: GetOptions) -> ObjectResult { + if o.head || o.range.is_none() { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + } + // Fetch metadata for the requested version with all caller conditions. + let head = remote_request( + &self.io_scope, + self.inner.get_opts( + p, + GetOptions { + head: true, + range: None, + ..o.clone() + }, + ), + ) + .await?; + o.check_preconditions(&head.meta)?; + let version = head + .meta + .version + .clone() + .filter(|v| !v.is_empty() && v != "null"); + let etag = head + .meta + .e_tag + .clone() + .filter(|v| !v.is_empty() && !v.starts_with("W/")); + if version.is_none() && etag.is_none() { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + } + // A backend that did not identify the requested version cannot safely + // populate a version-keyed cache. Preserve its native GET semantics. + if o.version.is_some() && o.version != head.meta.version { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + } + let range = o + .range + .as_ref() + .unwrap() + .as_range(head.meta.size) + .map_err(|source| object_store::Error::Generic { + store: "pchronicle-cache", + source: Box::new(source), + })?; + // v2 deliberately never reads the previous unnamespaced cache entries. + let key = serde_json::to_string(&( + "v2", + &self.store_uri, + p.as_ref(), + &version, + &etag, + head.meta.size, + self.block_size_bytes, + )) + .unwrap(); + let options = GetOptions { + version: version.or(o.version), + if_match: etag.or(o.if_match), + ..o + }; + let this = self.clone(); + let path = p.clone(); + let meta = head.meta.clone(); + let end = range.end; + let stream = futures::stream::try_unfold(range.start, move |start| { + let (this, path, meta, key, mut options) = ( + this.clone(), + path.clone(), + meta.clone(), + key.clone(), + options.clone(), + ); + async move { + if start >= end { + return Ok(None); + } + let block_start = (start / this.block_size_bytes) * this.block_size_bytes; + let block_end = block_start + .saturating_add(this.block_size_bytes) + .min(meta.size); + options.range = Some((block_start..block_end).into()); + let bytes = this + .read_block(&path, &meta, options, &key, block_start..block_end) + .await?; + let next = block_end.min(end); + Ok(Some(( + bytes.slice((start - block_start) as usize..(next - block_start) as usize), + next, + ))) + } + }); + Ok(GetResult { + payload: GetResultPayload::Stream(stream.boxed()), + meta: head.meta, + range, + attributes: head.attributes, + }) + } + fn delete_stream( + &self, + p: BoxStream<'static, ObjectResult>, + ) -> BoxStream<'static, ObjectResult> { + let uri = self.io_scope.clone(); + let stream = self.inner.delete_stream(p); + futures::stream::unfold(stream, move |mut stream| { + let uri = uri.clone(); + async move { + let _permit = io_gate::acquire(&uri, IoKind::Write).await; + match stream.next().await { + Some(result) => { + if result.is_ok() { + io_gate::note_success(&uri); + } else if result + .as_ref() + .err() + .is_some_and(io_gate::is_transient_error) + { + io_gate::note_failure(&uri, IoKind::Write); + } + Some((result, stream)) + } + None => None, + } + } + }) + .boxed() + } + fn list(&self, p: Option<&Path>) -> BoxStream<'static, ObjectResult> { + let uri = self.io_scope.clone(); + let stream = self.inner.list(p); + futures::stream::unfold(stream, move |mut stream| { + let uri = uri.clone(); + async move { + let _permit = io_gate::acquire(&uri, IoKind::Read).await; + match stream.next().await { + Some(result) => { + if result.is_ok() { + io_gate::note_success(&uri); + } else if result + .as_ref() + .err() + .is_some_and(io_gate::is_transient_error) + { + io_gate::note_failure(&uri, IoKind::Read); + } + Some((result, stream)) + } + None => None, + } + } + }) + .boxed() + } + async fn list_with_delimiter(&self, p: Option<&Path>) -> ObjectResult { + remote_request(&self.io_scope, self.inner.list_with_delimiter(p)).await + } + async fn copy_opts(&self, a: &Path, b: &Path, o: CopyOptions) -> ObjectResult<()> { + remote_write_request(&self.io_scope, self.inner.copy_opts(a, b, o)).await + } +} + +#[cfg(test)] +mod tests { + use super::*; + use object_store::{ObjectStoreExt, memory::InMemory}; + + #[tokio::test] + async fn range_cache_preserves_head_and_full_get_semantics() { + let root = tempfile::tempdir().unwrap(); + let inner = Arc::new(InMemory::new()); + let path = Path::from("dataset/data.lance"); + inner + .put(&path, Bytes::from_static(b"0123456789").into()) + .await + .unwrap(); + let cached = CachedObjectStore::new( + inner.clone(), + CacheConfig::new(root.path().into(), 1024, 4), + "s3://test-bucket".into(), + ); + + let head = cached.head(&path).await.unwrap(); + assert_eq!(head.size, 10); + assert_eq!( + cached.get_range(&path, 3..8).await.unwrap(), + Bytes::from_static(b"34567") + ); + assert_eq!( + cached.get(&path).await.unwrap().bytes().await.unwrap(), + Bytes::from_static(b"0123456789") + ); + } +} diff --git a/crates/persisting-pchronicle/src/store/blockcache/block.rs b/crates/persisting-pchronicle/src/store/blockcache/block.rs new file mode 100644 index 000000000..82321cc71 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/block.rs @@ -0,0 +1,201 @@ +use super::config::CacheConfig; +use bytes::Bytes; +use std::{ + collections::HashMap, + future::Future, + path::{Path, PathBuf}, + sync::{ + Arc, + atomic::{AtomicU64, Ordering}, + }, +}; + +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +pub struct CacheStats { + pub hits: u64, + pub misses: u64, + pub evictions: u64, +} + +#[derive(Debug, Default)] +struct Counters { + hits: AtomicU64, + misses: AtomicU64, + evictions: AtomicU64, +} + +#[derive(Debug, Clone)] +pub struct BlockCache { + config: CacheConfig, + counters: Arc, + flights: Arc>>>>, + last_trim_ms: Arc, +} + +impl BlockCache { + pub fn new(config: CacheConfig) -> Self { + Self { + config, + counters: Arc::new(Counters::default()), + flights: Arc::new(tokio::sync::Mutex::new(HashMap::new())), + last_trim_ms: Arc::new(AtomicU64::new(0)), + } + } + pub fn stats(&self) -> CacheStats { + CacheStats { + hits: self.counters.hits.load(Ordering::Relaxed), + misses: self.counters.misses.load(Ordering::Relaxed), + evictions: self.counters.evictions.load(Ordering::Relaxed), + } + } + pub fn block_path(&self, key: &str, block: u64) -> PathBuf { + self.config.root.join( + blake3::hash(format!("{key}:{block}").as_bytes()) + .to_hex() + .to_string(), + ) + } + pub async fn get_or_fetch( + &self, + path: &Path, + expected: usize, + fetch: F, + ) -> object_store::Result + where + F: Future>, + { + match tokio::fs::read(path).await { + Ok(bytes) if bytes.len() == expected => { + self.counters.hits.fetch_add(1, Ordering::Relaxed); + return Ok(Bytes::from(bytes)); + } + _ => { + self.counters.misses.fetch_add(1, Ordering::Relaxed); + } + }; + let flight = { + let mut flights = self.flights.lock().await; + flights + .entry(path.to_path_buf()) + .or_insert_with(|| Arc::new(tokio::sync::Mutex::new(()))) + .clone() + }; + let flight_guard = flight.lock().await; + let result = async { + if let Ok(bytes) = tokio::fs::read(path).await { + if bytes.len() == expected { + self.counters.hits.fetch_add(1, Ordering::Relaxed); + return Ok(Bytes::from(bytes)); + } + } + let bytes = fetch.await?; + if bytes.len() != expected { + return Err(object_store::Error::Generic { + store: "pchronicle-cache", + source: format!("block length {}, expected {expected}", bytes.len()).into(), + }); + } + if tokio::fs::create_dir_all(&self.config.root).await.is_ok() { + let tmp = path.with_extension(format!("{}.tmp", std::process::id())); + if tokio::fs::write(&tmp, &bytes).await.is_ok() { + let _ = tokio::fs::rename(&tmp, path).await; + self.schedule_trim(); + } + } + Ok(bytes) + } + .await; + drop(flight_guard); + let mut flights = self.flights.lock().await; + if Arc::strong_count(&flight) == 2 + && flights + .get(path) + .is_some_and(|current| Arc::ptr_eq(current, &flight)) + { + flights.remove(path); + } + result + } + fn schedule_trim(&self) { + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |duration| duration.as_millis() as u64); + let last = self.last_trim_ms.load(Ordering::Relaxed); + if now.saturating_sub(last) < 5_000 + || self + .last_trim_ms + .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed) + .is_err() + { + return; + } + let cache = self.clone(); + tokio::spawn(async move { cache.trim().await }); + } + pub async fn trim(&self) { + let Ok(mut entries) = tokio::fs::read_dir(&self.config.root).await else { + return; + }; + let mut files = Vec::new(); + let mut total = 0; + while let Ok(Some(entry)) = entries.next_entry().await { + if let Ok(meta) = entry.metadata().await { + if meta.is_file() { + total += meta.len(); + files.push((meta.modified().ok(), meta.len(), entry.path())); + } + } + } + files.sort_by_key(|(mtime, _, _)| *mtime); + for (_, size, path) in files { + if total <= self.config.capacity_bytes { + break; + } + if tokio::fs::remove_file(path).await.is_ok() { + total -= size; + self.counters.evictions.fetch_add(1, Ordering::Relaxed); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[tokio::test] + async fn hit_miss_and_corrupt_are_observable() { + let dir = tempfile::tempdir().unwrap(); + let cache = BlockCache::new(CacheConfig::new(dir.path().into(), 100, 4)); + let p = dir.path().join("x"); + assert_eq!( + cache + .get_or_fetch(&p, 4, async { Ok(Bytes::from_static(b"abcd")) }) + .await + .unwrap(), + Bytes::from_static(b"abcd") + ); + assert_eq!( + cache + .get_or_fetch(&p, 4, async { panic!("cache miss") }) + .await + .unwrap(), + Bytes::from_static(b"abcd") + ); + tokio::fs::write(&p, b"bad").await.unwrap(); + assert_eq!( + cache + .get_or_fetch(&p, 4, async { Ok(Bytes::from_static(b"efgh")) }) + .await + .unwrap(), + Bytes::from_static(b"efgh") + ); + assert_eq!( + cache.stats(), + CacheStats { + hits: 1, + misses: 2, + evictions: 0 + } + ); + } +} diff --git a/crates/persisting-pchronicle/src/store/blockcache/config.rs b/crates/persisting-pchronicle/src/store/blockcache/config.rs new file mode 100644 index 000000000..6e7ac3c01 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/config.rs @@ -0,0 +1,51 @@ +use std::path::PathBuf; + +pub const DEFAULT_CAPACITY_BYTES: u64 = 512 * 1024 * 1024; +pub const SERVE_CAPACITY_BYTES: u64 = 8 * DEFAULT_CAPACITY_BYTES; +pub const DEFAULT_BLOCK_SIZE_BYTES: u64 = 8 * 1024 * 1024; + +#[derive(Debug, Clone)] +pub struct CacheConfig { + pub root: PathBuf, + pub capacity_bytes: u64, + pub block_size_bytes: u64, +} + +impl CacheConfig { + pub fn new(root: PathBuf, capacity_bytes: u64, block_size_bytes: u64) -> Self { + assert!(capacity_bytes > 0 && block_size_bytes > 0); + Self { + root, + capacity_bytes, + block_size_bytes, + } + } + + pub fn from_env() -> Self { + let capacity = std::env::var("PCHRONICLE_LANCE_CACHE_CAPACITY_BYTES") + .ok() + .and_then(|v| v.parse().ok()) + .filter(|v| *v > 0) + .unwrap_or(DEFAULT_CAPACITY_BYTES); + Self::new(default_cache_dir(), capacity, DEFAULT_BLOCK_SIZE_BYTES) + } +} + +pub fn default_cache_dir() -> PathBuf { + std::env::var("PCHRONICLE_LANCE_CACHE_DIR") + .map(PathBuf::from) + .unwrap_or_else(|_| { + dirs::cache_dir() + .unwrap_or_else(|| PathBuf::from(".")) + .join("pchronicle") + .join("blocks") + }) +} + +pub const fn capacity_for_serve(is_serve: bool) -> u64 { + if is_serve { + SERVE_CAPACITY_BYTES + } else { + DEFAULT_CAPACITY_BYTES + } +} diff --git a/crates/persisting-pchronicle/src/store/blockcache/mod.rs b/crates/persisting-pchronicle/src/store/blockcache/mod.rs new file mode 100644 index 000000000..8c78722d3 --- /dev/null +++ b/crates/persisting-pchronicle/src/store/blockcache/mod.rs @@ -0,0 +1,14 @@ +mod adapter; +mod block; +mod config; + +pub use adapter::{CachedObjectStore, LanceCacheWrapper, lance_store_params}; +pub use block::{BlockCache, CacheStats}; +pub use config::{ + CacheConfig, DEFAULT_BLOCK_SIZE_BYTES, DEFAULT_CAPACITY_BYTES, SERVE_CAPACITY_BYTES, + capacity_for_serve, default_cache_dir, +}; + +pub fn configured_capacity_bytes() -> u64 { + CacheConfig::from_env().capacity_bytes +} diff --git a/crates/persisting-pchronicle/src/store/catalog/discovery.rs b/crates/persisting-pchronicle/src/store/catalog/discovery.rs index 8ce877df1..757db7aec 100644 --- a/crates/persisting-pchronicle/src/store/catalog/discovery.rs +++ b/crates/persisting-pchronicle/src/store/catalog/discovery.rs @@ -931,7 +931,7 @@ async fn is_compact_jsonl_directory(path: &Path) -> Result { if let Some(manifest) = try_load_manifest(path) { return Ok(manifest.is_compact_jsonl_leaf()); } - let dataset = match lance::Dataset::open(path.to_string_lossy().as_ref()).await { + let dataset = match crate::storage::open_lance_dataset(path.to_string_lossy().as_ref()).await { Ok(dataset) => dataset, Err(_) => return Ok(false), }; diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index 0c1734365..8172d36c0 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -8,11 +8,14 @@ use std::path::PathBuf; use std::sync::Arc; use std::time::Duration; -use anyhow::Result; +use anyhow::{Context, Result}; use serde::{Deserialize, Serialize}; -use tokio::sync::Mutex; use tokio::sync::RwLock; +const MAX_REFRESH_DIRECTORIES: usize = 10_000; +const REFRESH_MOUNT_DEADLINE: Duration = Duration::from_secs(120); +const REFRESH_DIRECTORY_TIMEOUT: Duration = Duration::from_secs(20); + use crate::store::{DatasetLocation, PathListEntry, PersistentCache}; #[derive(Debug, Clone, Serialize, Deserialize)] @@ -40,7 +43,6 @@ pub enum ManifestReadMode { pub struct ManifestCache { disk: Arc>, values: Arc>>, - refresh_gate: Arc>, } impl ManifestCache { @@ -53,11 +55,7 @@ impl ManifestCache { .filter_map(|(key, value)| serde_json::from_value(value).ok().map(|v| (key, v))) .collect(), )); - Self { - disk, - values, - refresh_gate: Arc::new(Mutex::new(())), - } + Self { disk, values } } pub async fn get(&self, key: &str) -> Option { @@ -91,8 +89,10 @@ impl ManifestCache { location: &DatasetLocation, prefix: &str, ) -> Result { - let _guard = self.refresh_gate.lock().await; let key = key.into(); + let refresh_gate = + crate::store::root_write_lock::for_root(&serde_json::to_string(&(self.path(), &key))?); + let _guard = refresh_gate.lock().await; let listing = ManifestListing { entries: location.list(prefix).await?, observed_at: chrono::Utc::now().timestamp(), @@ -101,9 +101,14 @@ impl ManifestCache { .write() .await .insert(key.clone(), listing.clone()); - self.disk + if let Err(error) = self + .disk .upsert(&key, &serde_json::to_value(&listing)?, &[]) - .await?; + .await + { + tracing::warn!(target: "pchronicle.serve", %key, %error, + "manifest cache persistence failed; using memory"); + } tracing::info!(target: "pchronicle.serve", %key, prefix, entries = listing.entries.len(), "manifest cache updated"); Ok(listing) } @@ -178,30 +183,30 @@ impl ManifestCache { }) } - /// Breadth-first refresh of a mount. Only one refresh runs at a time; - /// shallow paths are published before deeper paths. + /// Breadth-first refresh of a mount. Refreshes serialize per observation + /// key, allowing unrelated foreground directories to load concurrently. pub async fn refresh_mount(&self, key_prefix: &str, location: &DatasetLocation) -> Result<()> { - let _guard = self.refresh_gate.lock().await; tracing::info!(target: "pchronicle.serve", %key_prefix, "manifest cache refresh started"); let mut queue = VecDeque::from([String::new()]); let mut refreshed = 0usize; + let deadline = tokio::time::Instant::now() + REFRESH_MOUNT_DEADLINE; while let Some(prefix) = queue.pop_front() { - let listing = ManifestListing { - entries: location.list(&prefix).await?, - observed_at: chrono::Utc::now().timestamp(), - }; + if refreshed >= MAX_REFRESH_DIRECTORIES || tokio::time::Instant::now() >= deadline { + tracing::warn!(target: "pchronicle.serve", %key_prefix, refreshed, + queued = queue.len(), "manifest cache refresh reached safety bound; keeping partial view"); + break; + } let key = if prefix.is_empty() { key_prefix.to_owned() } else { format!("{key_prefix}\0{prefix}") }; - self.values - .write() - .await - .insert(key.clone(), listing.clone()); - self.disk - .upsert(&key, &serde_json::to_value(&listing)?, &[]) - .await?; + let listing = tokio::time::timeout( + REFRESH_DIRECTORY_TIMEOUT, + self.refresh(key, location, &prefix), + ) + .await + .context("manifest directory refresh timed out")??; refreshed += 1; for child in listing .entries @@ -240,6 +245,49 @@ mod tests { use super::*; use crate::store::{PathListEntry, PathListKind}; + #[tokio::test] + async fn refresh_survives_disk_failure_and_unrelated_refresh_lock() { + let dir = tempfile::tempdir().unwrap(); + let source = dir.path().join("source"); + std::fs::create_dir_all(source.join("nested")).unwrap(); + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + let path = dir.path().join("manifest.lance"); + let cache = ManifestCache::open(path.clone()).await; + // Inject a persistent disk failure after opening a writable cache. + std::fs::write(&path, "not a Lance directory").unwrap(); + assert!( + cache + .disk + .upsert(&"probe".into(), &serde_json::Value::Null, &[]) + .await + .is_err() + ); + let gate = crate::store::root_write_lock::for_root( + &serde_json::to_string(&(cache.path(), "blocked")).unwrap(), + ); + let _guard = gate.lock().await; + let listing = tokio::time::timeout( + Duration::from_secs(5), + cache.refresh("healthy", &location, ""), + ) + .await + .unwrap() + .unwrap(); + assert!(!listing.entries.is_empty()); + assert_eq!( + cache.get("healthy").await.unwrap().entries.len(), + listing.entries.len() + ); + cache.refresh_mount("mount", &location).await.unwrap(); + assert!(cache.get("mount").await.is_some()); + assert!( + cache + .refresh("missing", &location, "../escape") + .await + .is_err() + ); + } + #[tokio::test] async fn summary_counts_cached_manifest_entries() { let dir = tempfile::tempdir().unwrap(); diff --git a/crates/persisting-pchronicle/src/store/compact_jsonl.rs b/crates/persisting-pchronicle/src/store/compact_jsonl.rs index d0ab8f109..ea0e10307 100644 --- a/crates/persisting-pchronicle/src/store/compact_jsonl.rs +++ b/crates/persisting-pchronicle/src/store/compact_jsonl.rs @@ -168,7 +168,7 @@ impl CompactJsonlStore { /// here so catalog/CLI paths cannot skip `chronicle.manifest`. pub async fn publish_manifest(root: impl AsRef) -> Result { let root = root.as_ref(); - let dataset = Dataset::open(root.to_string_lossy().as_ref()) + let dataset = crate::storage::open_lance_dataset(root.to_string_lossy().as_ref()) .await .with_context(|| { format!( @@ -188,10 +188,11 @@ impl CompactJsonlStore { /// manifests are rewritten; compatible old datasets are upgraded in place. pub async fn ensure_manifest(root: impl AsRef) -> Result> { let root = root.as_ref(); - let dataset = match Dataset::open(root.to_string_lossy().as_ref()).await { - Ok(dataset) => dataset, - Err(_) => return Ok(None), - }; + let dataset = + match crate::storage::open_lance_dataset(root.to_string_lossy().as_ref()).await { + Ok(dataset) => dataset, + Err(_) => return Ok(None), + }; if validate_dataset_schema(&dataset).is_err() { return Ok(None); } @@ -229,7 +230,7 @@ impl CompactJsonlStore { let input = input.as_ref(); let limit = limit.max(1); let manifest = Self::ensure_manifest(input).await?; - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; validate_dataset_schema(&dataset)?; let total = if let Some(count) = manifest .as_ref() @@ -283,7 +284,7 @@ impl CompactJsonlStore { pub async fn records(input: impl AsRef) -> Result> { let input = input.as_ref(); let _ = Self::ensure_manifest(input).await?; - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; validate_dataset_schema(&dataset)?; let stream = dataset.scan().scan_in_order(true).try_into_stream().await?; let mut stream = stream; @@ -320,7 +321,7 @@ impl CompactJsonlStore { pub async fn read_record(input: impl AsRef, id: &str) -> Result> { let input = input.as_ref(); let _ = Self::ensure_manifest(input).await?; - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; let (_, offload_idx) = validate_dataset_schema(&dataset)?; let stream = dataset.scan().scan_in_order(true).try_into_stream().await?; let mut stream = stream; @@ -670,7 +671,7 @@ impl CompactJsonlStore { output_root != input_root && !output_root.starts_with(&input_root), "compact JSONL export output must be outside the input dataset" ); - let dataset = Dataset::open(input.to_string_lossy().as_ref()).await?; + let dataset = crate::storage::open_lance_dataset(input.to_string_lossy().as_ref()).await?; let (raw_idx, offload_idx) = validate_dataset_schema(&dataset)?; if output.exists() { fs::remove_dir_all(output) diff --git a/crates/persisting-pchronicle/src/store/events/mod.rs b/crates/persisting-pchronicle/src/store/events/mod.rs index 888b3b39b..fb388e05d 100644 --- a/crates/persisting-pchronicle/src/store/events/mod.rs +++ b/crates/persisting-pchronicle/src/store/events/mod.rs @@ -421,7 +421,7 @@ pub(crate) async fn compact_sealed_event_segment( ); let _guard = dataset_write_lock::acquire(&sealed.root_uri).await?; let segment_uri = raw_event_manifest::segment_uri(&sealed.root_uri, &sealed.segment.id); - let mut dataset = Dataset::open(&segment_uri) + let mut dataset = crate::storage::open_lance_dataset(&segment_uri) .await .with_context(|| format!("open sealed event segment {segment_uri}"))?; let metrics = compact_files( @@ -514,7 +514,7 @@ pub(super) fn validate_event_schema(dataset: &Dataset, uri: &str) -> Result<()> async fn open_visible_segment(root_uri: &str, segment: &EventSegment) -> Result { let uri = raw_event_manifest::segment_uri(root_uri, &segment.id); - let latest = Dataset::open(&uri) + let latest = crate::storage::open_lance_dataset(&uri) .await .with_context(|| format!("open event segment {uri}"))?; let dataset = if latest.version_id() == segment.version { diff --git a/crates/persisting-pchronicle/src/store/inspect.rs b/crates/persisting-pchronicle/src/store/inspect.rs index 426464508..453a9e809 100644 --- a/crates/persisting-pchronicle/src/store/inspect.rs +++ b/crates/persisting-pchronicle/src/store/inspect.rs @@ -245,7 +245,7 @@ pub async fn inspect_physical_file( }) .collect::>(); if let Ok(uri) = table_uri(snapshot, dataset, file, table).await - && let Ok(lance) = Dataset::open(&uri).await + && let Ok(lance) = crate::storage::open_lance_dataset(&uri).await { enrich_column_stats(&lance, fragment_id, &mut columns).await; } @@ -304,7 +304,7 @@ pub async fn inspect_physical_page( anyhow::ensure!(!columns.is_empty(), "physical data file has no columns"); let limit = query.limit.clamp(1, DEFAULT_PHYSICAL_PAGE_LIMIT); let uri = table_uri(snapshot, query.dataset, query.file, query.table).await?; - let lance = Dataset::open(&uri) + let lance = crate::storage::open_lance_dataset(&uri) .await .with_context(|| format!("open Lance table {uri}"))?; preview_rows(&lance, &columns, query.offset, limit).await @@ -402,7 +402,7 @@ fn storyline_table_path<'a>(paths: &'a StorylineTablePaths, table: &str) -> Opti async fn open_table_version(path: &Path, version: u64) -> Result { let uri = path.to_string_lossy(); - let dataset = Dataset::open(uri.as_ref()) + let dataset = crate::storage::open_lance_dataset(uri.as_ref()) .await .with_context(|| format!("open Lance table {uri}"))?; dataset diff --git a/crates/persisting-pchronicle/src/store/mod.rs b/crates/persisting-pchronicle/src/store/mod.rs index 5972c9b69..70119379f 100644 --- a/crates/persisting-pchronicle/src/store/mod.rs +++ b/crates/persisting-pchronicle/src/store/mod.rs @@ -12,6 +12,8 @@ mod agenticmd_datafusion; #[cfg(feature = "lance-store")] mod attempt_registry; #[cfg(feature = "lance-store")] +pub(crate) mod blockcache; +#[cfg(feature = "lance-store")] mod cas_store; #[cfg(feature = "lance-store")] mod catalog; diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 15c1a0c84..7e4c436b4 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -9,6 +9,7 @@ //! //! Local `file://` paths bypass the gate entirely. +use std::collections::HashMap; use std::sync::{Arc, Mutex, OnceLock}; use std::time::{Duration, Instant}; @@ -16,6 +17,8 @@ use tokio::sync::{OwnedSemaphorePermit, Semaphore}; const DEFAULT_REMOTE_CONCURRENCY: usize = 1; const MAX_REMOTE_CONCURRENCY: usize = 2; +const MAX_RETAINED_SCOPES: usize = 1024; +const SCOPE_IDLE_TTL: Duration = Duration::from_secs(300); const MAX_DELAY_MS: u64 = 30_000; const SUCCESS_STREAK_TO_DECAY: u32 = 4; @@ -117,8 +120,9 @@ fn emit_throttle(event: ObjectStoreThrottleEvent) { } } -#[derive(Debug)] +#[derive(Debug, Clone)] struct AimdState { + last_used: Instant, /// Extra sleep applied before each remote acquire while degraded. delay_ms: u64, /// No new remote op starts until this instant. @@ -134,6 +138,7 @@ struct AimdState { impl Default for AimdState { fn default() -> Self { Self { + last_used: Instant::now(), delay_ms: 0, cooldown_until: None, successes_since_backoff: 0, @@ -147,7 +152,29 @@ impl Default for AimdState { struct Gate { semaphore: Arc, concurrency: usize, - state: Mutex, + states: Mutex>, +} + +fn state_for<'a>(states: &'a mut HashMap, key: &str) -> &'a mut AimdState { + let now = Instant::now(); + if !states.contains_key(key) { + states.retain(|_, state| state.active_waiters > 0 + || state.cooldown_until.is_some_and(|until| until > now) + || now.duration_since(state.last_used) < SCOPE_IDLE_TTL); + if states.len() >= MAX_RETAINED_SCOPES { + let oldest = states.iter() + .filter(|(_, state)| state.active_waiters == 0 + && state.cooldown_until.is_none_or(|until| until <= now)) + .min_by_key(|(_, state)| state.last_used) + .map(|(key, _)| key.clone()); + if let Some(oldest) = oldest { states.remove(&oldest); } + } + } + // Live waits/cooldowns may temporarily exceed the retention limit. Evicting + // them would let backend overload bypass AIMD; reclaim after they finish. + let state = states.entry(key.to_owned()).or_default(); + state.last_used = now; + state } fn gate() -> &'static Gate { @@ -161,11 +188,48 @@ fn gate() -> &'static Gate { Gate { semaphore: Arc::new(Semaphore::new(concurrency)), concurrency, - state: Mutex::new(AimdState::default()), + states: Mutex::new(HashMap::new()), } }) } +/// Canonical admission scope, independent of object path and cache credentials. +/// Adapters capture this at construction so later environment changes cannot +/// move feedback for an existing store into another endpoint's state. +pub(crate) fn scope_key(uri: &str) -> String { + if uri + .split_once('#') + .is_some_and(|(_, fragment)| fragment.starts_with("endpoint=")) + { + return uri.to_owned(); + } + let endpoint_vars: &[&str] = if uri.starts_with("s3") { + &["AWS_ENDPOINT_URL_S3", "AWS_ENDPOINT", "AWS_ENDPOINT_URL"] + } else if uri.starts_with("az") { + &["AZURE_STORAGE_ENDPOINT"] + } else { + &["GOOGLE_STORAGE_BASE_URL"] + }; + let endpoint = endpoint_vars + .iter() + .find_map(|name| std::env::var(name).ok().filter(|v| !v.is_empty())) + .unwrap_or_default(); + scope_for_endpoint(uri, &endpoint) +} + +fn scope_for_endpoint(uri: &str, endpoint: &str) -> String { + let uri = uri.split('#').next().unwrap_or(uri); + let Some((scheme, rest)) = uri.split_once("://") else { + return uri.to_owned(); + }; + let bucket = rest.split('/').next().unwrap_or(rest); + let endpoint = url::Url::parse(endpoint) + .map(|url| url.to_string()) + .unwrap_or_else(|_| endpoint.to_owned()); + let identity = blake3::hash(endpoint.trim_end_matches('/').as_bytes()); + format!("{scheme}://{bucket}#endpoint={}", identity.to_hex()) +} + /// True for s3/gs/az (and similar) URIs; false for local paths / file://. pub(crate) fn is_remote_uri(uri: &str) -> bool { let Some((scheme, _)) = uri.split_once("://") else { @@ -174,12 +238,59 @@ pub(crate) fn is_remote_uri(uri: &str) -> bool { !matches!(scheme, "file" | "file+uring" | "memory" | "shared-memory") } +/// Classify errors for shared AIMD feedback. Permanent object identity, +/// authorization, and precondition failures must never increase cooldown. +pub(crate) fn is_transient_error(error: &object_store::Error) -> bool { + if matches!( + error, + object_store::Error::NotFound { .. } + | object_store::Error::InvalidPath { .. } + | object_store::Error::NotSupported { .. } + | object_store::Error::AlreadyExists { .. } + | object_store::Error::Precondition { .. } + | object_store::Error::NotModified { .. } + | object_store::Error::PermissionDenied { .. } + | object_store::Error::Unauthenticated { .. } + ) { + return false; + } + let text = error.to_string().to_ascii_lowercase(); + [ + "timeout", + "timed out", + "connection", + "broken pipe", + "temporarily", + "slowdown", + "throttl", + "503", + "429", + "reset", + ] + .iter() + .any(|needle| text.contains(needle)) +} + /// Snapshot AIMD / cooldown state for progress UI. pub fn snapshot() -> ObjectStoreGateSnapshot { let g = gate(); let available_permits = g.semaphore.available_permits(); let max_permits = g.concurrency; - let Ok(state) = g.state.lock() else { + let Ok(states) = g.states.lock() else { + return ObjectStoreGateSnapshot { + kind: IoKind::Read, + delay_ms: 0, + cooldown_remaining_ms: 0, + failures: 0, + success_streak: 0, + success_streak_target: SUCCESS_STREAK_TO_DECAY, + active_waiters: 0, + available_permits, + max_permits, + }; + }; + let state = states.values().max_by_key(|state| state.failures); + let Some(state) = state else { return ObjectStoreGateSnapshot { kind: IoKind::Read, delay_ms: 0, @@ -250,37 +361,73 @@ pub(crate) async fn acquire(uri: &str, kind: IoKind) -> Permit { if !is_remote_uri(uri) { return Permit { _permit: None }; } - if let Ok(mut state) = gate().state.lock() { - state.last_kind = kind; + acquire_scoped(gate(), &scope_key(uri), kind).await +} + +async fn acquire_scoped(g: &Gate, key: &str, kind: IoKind) -> Permit { + if let Ok(mut states) = g.states.lock() { + state_for(&mut states, key).last_kind = kind; } - wait_out_degradation(kind).await; - let permit = match gate().semaphore.clone().try_acquire_owned() { - Ok(permit) => permit, - Err(_) => { - enter_wait(kind, "admit", 0); - let permit = match gate().semaphore.clone().acquire_owned().await { - Ok(permit) => permit, - Err(error) => { - leave_wait(kind); - tracing::error!(?error, "object-store I/O semaphore closed unexpectedly"); - return Permit { _permit: None }; + loop { + wait_out_degradation(g, key, kind).await; + let permit = match g.semaphore.clone().try_acquire_owned() { + Ok(permit) => permit, + Err(_) => { + let _wait = WaitGuard::new(g, key, kind, "admit", 0); + match g.semaphore.clone().acquire_owned().await { + Ok(permit) => permit, + Err(error) => { + tracing::error!(?error, "object-store I/O semaphore closed unexpectedly"); + return Permit { _permit: None }; + } } - }; - leave_wait(kind); - permit + } + }; + // A failure may have started a new cooldown while admission was + // queued. Release capacity before sleeping, then compete again. + if !cooldown_remaining(g, key).is_zero() { + drop(permit); + continue; } - }; - wait_out_degradation(kind).await; - Permit { - _permit: Some(permit), + return Permit { + _permit: Some(permit), + }; + } +} + +fn cooldown_remaining(g: &Gate, key: &str) -> Duration { + g.states + .lock() + .ok() + .and_then(|states| states.get(key).and_then(|state| state.cooldown_until)) + .and_then(|until| until.checked_duration_since(Instant::now())) + .unwrap_or_default() +} + +// Cancellation of an HTTP request must also release progress wait accounting. +struct WaitGuard<'a> { + gate: &'a Gate, + key: &'a str, + kind: IoKind, +} +impl<'a> WaitGuard<'a> { + fn new(gate: &'a Gate, key: &'a str, kind: IoKind, reason: &'static str, wait_ms: u64) -> Self { + enter_wait(gate, key, kind, reason, wait_ms); + Self { gate, key, kind } + } +} +impl Drop for WaitGuard<'_> { + fn drop(&mut self) { + leave_wait(self.gate, self.key, self.kind); } } -fn enter_wait(kind: IoKind, reason: &'static str, wait_ms: u64) { +fn enter_wait(g: &Gate, key: &str, kind: IoKind, reason: &'static str, wait_ms: u64) { let (delay_ms, failures) = { - let Ok(mut state) = gate().state.lock() else { + let Ok(mut states) = g.states.lock() else { return; }; + let state = state_for(&mut states, key); state.last_kind = kind; state.active_waiters = state.active_waiters.saturating_add(1); (state.delay_ms, state.failures) @@ -294,18 +441,20 @@ fn enter_wait(kind: IoKind, reason: &'static str, wait_ms: u64) { }); } -fn leave_wait(kind: IoKind) { - if let Ok(mut state) = gate().state.lock() { +fn leave_wait(g: &Gate, key: &str, kind: IoKind) { + if let Ok(mut states) = g.states.lock() { + let state = state_for(&mut states, key); state.active_waiters = state.active_waiters.saturating_sub(1); } emit_throttle(ObjectStoreThrottleEvent::Leave { kind }); } -fn emit_update(kind: IoKind, reason: &'static str, wait_ms: u64) { +fn emit_update(g: &Gate, key: &str, kind: IoKind, reason: &'static str, wait_ms: u64) { let (delay_ms, failures) = { - let Ok(state) = gate().state.lock() else { + let Ok(states) = g.states.lock() else { return; }; + let state = states.get(key).cloned().unwrap_or_default(); (state.delay_ms, state.failures) }; emit_throttle(ObjectStoreThrottleEvent::Update { @@ -317,9 +466,12 @@ fn emit_update(kind: IoKind, reason: &'static str, wait_ms: u64) { }); } -async fn wait_out_degradation(kind: IoKind) { +async fn wait_out_degradation(g: &Gate, key: &str, kind: IoKind) { let (sleep_for, delay_ms, failures) = { - let Ok(state) = gate().state.lock() else { + let Ok(states) = g.states.lock() else { + return; + }; + let Some(state) = states.get(key) else { return; }; let cooldown = state @@ -332,7 +484,7 @@ async fn wait_out_degradation(kind: IoKind) { return; } let wait_ms = sleep_for.as_millis() as u64; - enter_wait(kind, "throttle", wait_ms); + let _wait = WaitGuard::new(g, key, kind, "throttle", wait_ms); crate::store::index_build_progress::note(format!( "s3 {} throttle wait {:.1}s (failures={failures}, delay={delay_ms}ms)", kind.as_str(), @@ -347,24 +499,23 @@ async fn wait_out_degradation(kind: IoKind) { "object-store I/O gate cooling down before next remote op" ); // Tick the progress UI while cooling down so `cd=` counts down live. - let deadline = Instant::now() + sleep_for; const TICK: Duration = Duration::from_millis(250); loop { - let now = Instant::now(); - if now >= deadline { + let remaining = cooldown_remaining(g, key); + if remaining.is_zero() { break; } - let remaining = deadline - now; - emit_update(kind, "throttle", remaining.as_millis() as u64); + emit_update(g, key, kind, "throttle", remaining.as_millis() as u64); tokio::time::sleep(remaining.min(TICK)).await; } - leave_wait(kind); } /// Publish the current I/O phase for progress UI without taking a permit. /// Used around Lance writes that do not go through [`acquire`]. -pub(crate) fn mark_kind(kind: IoKind) { - if let Ok(mut state) = gate().state.lock() { +pub(crate) fn mark_kind(uri: &str, kind: IoKind) { + if !is_remote_uri(uri) { return; } + if let Ok(mut states) = gate().states.lock() { + let state = state_for(&mut states, &scope_key(uri)); state.last_kind = kind; } } @@ -375,9 +526,11 @@ pub(crate) fn note_success(uri: &str) { return; } let (kind, changed, delay_ms, failures) = { - let Ok(mut state) = gate().state.lock() else { + let key = scope_key(uri); + let Ok(mut states) = gate().states.lock() else { return; }; + let state = state_for(&mut states, &key); let kind = state.last_kind; state.successes_since_backoff = state.successes_since_backoff.saturating_add(1); if state.delay_ms == 0 { @@ -421,9 +574,11 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { return; } let (delay_ms, failures, wait_ms) = { - let Ok(mut state) = gate().state.lock() else { + let key = scope_key(uri); + let Ok(mut states) = gate().states.lock() else { return; }; + let state = state_for(&mut states, &key); state.last_kind = kind; state.failures = state.failures.saturating_add(1); state.successes_since_backoff = 0; @@ -460,6 +615,95 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { mod tests { use super::*; + #[test] + fn registry_reclaims_idle_scopes_but_preserves_waits_and_cooldowns() { + let mut states = HashMap::new(); + state_for(&mut states, "waiting").active_waiters = 1; + state_for(&mut states, "cooling").cooldown_until = Some(Instant::now() + Duration::from_secs(60)); + for n in 0..MAX_RETAINED_SCOPES * 2 { + state_for(&mut states, &n.to_string()); + } + assert_eq!(states.len(), MAX_RETAINED_SCOPES); + assert!(states.contains_key("waiting")); + assert!(states.contains_key("cooling")); + assert!(!states.contains_key("0")); + for state in states.values_mut() { state.last_used = Instant::now() - SCOPE_IDLE_TTL; } + state_for(&mut states, "new"); + assert_eq!(states.len(), 3); + } + + #[test] + fn scopes_share_bucket_paths_and_isolate_endpoints_and_buckets() { + let scope = scope_for_endpoint( + "s3://bucket/a/table#old-cache-credentials", + "HTTP://LOCALHOST:9000/", + ); + assert_eq!( + scope, + scope_for_endpoint("s3://bucket/b", "http://localhost:9000") + ); + assert_eq!(scope_key(&scope), scope); + assert_ne!( + scope, + scope_for_endpoint("s3://bucket/a", "http://localhost:9001") + ); + assert_ne!( + scope, + scope_for_endpoint("s3://other/a", "http://localhost:9000") + ); + } + + #[tokio::test] + async fn cooldown_after_queued_admission_releases_capacity_and_cancellation_clears_waiters() { + let g = Arc::new(Gate { + semaphore: Arc::new(Semaphore::new(1)), + concurrency: 1, + states: Mutex::new(HashMap::new()), + }); + let held = g.semaphore.clone().acquire_owned().await.unwrap(); + let bad = scope_for_endpoint("s3://bucket/a", "http://slow"); + let healthy = scope_for_endpoint("s3://bucket/a", "http://healthy"); + let waiter = tokio::spawn({ + let g = g.clone(); + let bad = bad.clone(); + async move { acquire_scoped(&g, &bad, IoKind::Read).await } + }); + tokio::time::timeout(Duration::from_secs(2), async { + while g + .states + .lock() + .unwrap() + .get(&bad) + .is_none_or(|s| s.active_waiters == 0) + { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + g.states + .lock() + .unwrap() + .get_mut(&bad) + .unwrap() + .cooldown_until = Some(Instant::now() + Duration::from_secs(10)); + drop(held); + // The queued unhealthy request is first in the semaphore's FIFO. It + // must give up its slot on discovering the newly imposed cooldown. + let permit = tokio::time::timeout( + Duration::from_millis(500), + acquire_scoped(&g, &healthy, IoKind::Read), + ) + .await + .unwrap(); + assert!(!waiter.is_finished()); + waiter.abort(); + assert!(waiter.await.is_err()); + assert_eq!(g.states.lock().unwrap()[&bad].active_waiters, 0); + drop(permit); + assert_eq!(g.semaphore.available_permits(), 1); + } + #[test] fn aimd_flow_label_healthy_and_degraded() { let healthy = ObjectStoreGateSnapshot { diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 265381c9f..c192a3052 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -4,14 +4,15 @@ //! module keeps pChronicle's own reads, listings and conditional writes on //! OpenDAL so backend differences are handled in one place. +use crate::store::object_store_io_gate::{self as io_gate, IoKind}; use anyhow::{Context, Result, anyhow}; use futures::TryStreamExt; use opendal::layers::RetryLayer; use opendal::{EntryMode, ErrorKind, Metadata, Operator}; -use std::collections::HashMap; +use std::collections::{BTreeMap, HashMap}; use std::sync::Arc; use std::sync::{Mutex, OnceLock}; -use std::time::Duration; +use std::time::{Duration, Instant}; use url::Url; /// Retries for transient object-store failures (DNS blips, connect resets, @@ -54,6 +55,7 @@ impl Version { pub(crate) struct Store { operator: Operator, fallback_lock: Option>>, + io_scope: String, } #[derive(Clone, Debug)] @@ -73,34 +75,70 @@ static SHARED_MEMORY: OnceLock>> = OnceLock::new static SHARED_LOCKS: OnceLock>>>> = OnceLock::new(); +const MAX_CACHED_OPERATORS: usize = 128; +const OPERATOR_IDLE_TTL: Duration = Duration::from_secs(300); + +#[derive(Default)] +struct OperatorRegistry { + entries: HashMap, +} + +impl OperatorRegistry { + fn get(&mut self, key: &str, now: Instant) -> Option { + self.entries.retain(|_, (_, used)| now.duration_since(*used) < OPERATOR_IDLE_TTL); + self.entries.get_mut(key).map(|(operator, used)| { + *used = now; + operator.clone() + }) + } + + fn insert(&mut self, key: String, operator: Operator, now: Instant) { + if !self.entries.contains_key(&key) && self.entries.len() >= MAX_CACHED_OPERATORS { + if let Some(oldest) = self.entries.iter().min_by_key(|(_, (_, used))| *used) + .map(|(key, _)| key.clone()) { + self.entries.remove(&oldest); + } + } + self.entries.insert(key, (operator, now)); + } +} + +static OPERATORS: OnceLock> = OnceLock::new(); + impl Store { pub(crate) async fn from_uri(uri: &str) -> Result { let uri = uri.trim(); let normalized = normalize_uri(uri)?; + let cache_key = operator_cache_key(uri, &normalized); let shared_memory = uri.contains("://") && Url::parse(uri) .map(|parsed| parsed.scheme() == "shared-memory") .unwrap_or(false); - let operator = if shared_memory { - let map = SHARED_MEMORY.get_or_init(|| Mutex::new(HashMap::new())); - let mut map = map - .lock() - .map_err(|_| anyhow!("shared-memory operator registry poisoned"))?; + // Memory operators own the data itself and must not be evicted like clients. + let operator = if normalized.starts_with("memory://") { + let mut map = SHARED_MEMORY.get_or_init(|| Mutex::new(HashMap::new())) + .lock().unwrap_or_else(std::sync::PoisonError::into_inner); if let Some(operator) = map.get(uri) { operator.clone() } else { - let operator = with_object_store_retries( - Operator::from_uri(normalized.as_str()) - .with_context(|| format!("open OpenDAL store {uri}"))?, - ); - map.insert(uri.to_string(), operator.clone()); + let operator = with_object_store_retries(Operator::from_uri(normalized.as_str())?); + map.insert(uri.to_owned(), operator.clone()); operator } } else { - with_object_store_retries( - Operator::from_uri(normalized.as_str()) - .with_context(|| format!("open OpenDAL store {uri}"))?, - ) + let registry = OPERATORS.get_or_init(|| Mutex::new(OperatorRegistry::default())); + let cached = registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&cache_key, Instant::now()); + if let Some(operator) = cached { + operator + } else { + // Construct outside the registry lock so one backend cannot block all others. + let operator = with_object_store_retries(Operator::from_uri(normalized.as_str()) + .with_context(|| format!("open OpenDAL store {uri}"))?); + registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(cache_key, operator.clone(), Instant::now()); + operator + } }; let fallback_lock = if shared_memory { let locks = SHARED_LOCKS.get_or_init(|| Mutex::new(HashMap::new())); @@ -119,26 +157,79 @@ impl Store { Ok(Self { operator, fallback_lock, + io_scope: io_gate::scope_key(uri), }) } + async fn remote(&self, kind: IoKind, request: F) -> Result + where + F: FnOnce(Operator) -> Fut, + Fut: std::future::IntoFuture>, + { + let _permit = io_gate::acquire(&self.io_scope, kind).await; + match request(self.operator.clone()).into_future().await { + Ok(value) => { + io_gate::note_success(&self.io_scope); + Ok(value) + } + Err(error) => { + let text = error.to_string().to_ascii_lowercase(); + if [ + "timeout", + "connection", + "broken pipe", + "temporarily", + "slowdown", + "throttl", + "503", + "429", + "reset", + ] + .iter() + .any(|needle| text.contains(needle)) + { + io_gate::note_failure(&self.io_scope, kind); + } + Err(error.into()) + } + } + } + pub(crate) async fn read(&self, path: &str) -> Result, Version)>> { - let metadata = match self.operator.stat(path).await { + let path_owned = path.to_owned(); + let metadata = match self + .remote(IoKind::Read, |operator| async move { + operator.stat(&path_owned).await + }) + .await + { Ok(metadata) => metadata, - Err(error) if error.kind() == ErrorKind::NotFound => return Ok(None), + Err(error) + if error + .downcast_ref::() + .is_some_and(|error| error.kind() == ErrorKind::NotFound) => + { + return Ok(None); + } Err(error) => return Err(error.into()), }; - let bytes = self.operator.read(path).await?.to_vec(); + let path_owned = path.to_owned(); + let bytes = self + .remote(IoKind::Read, |operator| async move { + operator.read(&path_owned).await + }) + .await? + .to_vec(); Ok(Some((bytes, version(&metadata)))) } pub(crate) async fn write_create(&self, path: &str, bytes: Vec) -> Result<()> { - self.operator - .write_with(path, bytes) - .if_not_exists(true) - .await - .map(|_| ()) - .map_err(Into::into) + let path = path.to_owned(); + self.remote(IoKind::Write, |operator| async move { + operator.write_with(&path, bytes).if_not_exists(true).await + }) + .await + .map(|_| ()) } pub(crate) async fn write_match( @@ -155,44 +246,46 @@ impl Store { let condition = expected.condition().ok_or_else(|| { anyhow!("OpenDAL backend did not return an ETag/version for conditional write") })?; + let path = path.to_owned(); + let condition = condition.to_owned(); + let log_path = path.clone(); + let log_condition = condition.clone(); let result = self - .operator - .write_with(path, bytes.clone()) - .if_match(condition) - .await; - match result { - Ok(_) => Ok(()), - // Some S3-compatible gateways compare the If-Match header against - // their unquoted ETag, so a correctly quoted condition always - // fails with 412. One retry with the unquoted form still proves - // the stored ETag matched; real contention fails both attempts. - Err(error) - if error.kind() == ErrorKind::ConditionNotMatch - && let Some(unquoted) = unquoted_etag(condition) => - { - let retry = self - .operator - .write_with(path, bytes) - .if_match(unquoted) + .remote(IoKind::Write, |operator| async move { + let result = operator + .write_with(&path, bytes.clone()) + .if_match(&condition) .await; - match retry { + match result { Ok(_) => Ok(()), - // Preserve the original conditional conflict when the - // gateway rejects the compatibility form itself. - Err(retry_error) if retry_error.kind() != ErrorKind::ConditionNotMatch => { - Err(error) + // Some S3-compatible gateways compare the If-Match header against + // their unquoted ETag, so retry once with the compatibility form. + Err(error) + if error.kind() == ErrorKind::ConditionNotMatch + && let Some(unquoted) = unquoted_etag(&condition) => + { + match operator.write_with(&path, bytes).if_match(unquoted).await { + Ok(_) => Ok(()), + Err(retry_error) + if retry_error.kind() != ErrorKind::ConditionNotMatch => + { + Err(error) + } + Err(retry_error) => Err(retry_error), + } } - Err(retry_error) => Err(retry_error), + Err(error) => Err(error), } - } - Err(error) => Err(error), - } - .map_err(|error| { - if is_conflict(&error) { + }) + .await; + result.map_err(|error| { + if let Some(error) = error.downcast_ref::() + && is_conflict(error) + { tracing::debug!( target: "pchronicle.opendal", - path, - if_match = condition, + path = log_path, + if_match = log_condition, error = %error, kind = ?error.kind(), "conditional object write conflict (If-Match)" @@ -203,79 +296,103 @@ impl Store { } pub(crate) async fn write_overwrite(&self, path: &str, bytes: Vec) -> Result<()> { - self.operator - .write(path, bytes) - .await - .map(|_| ()) - .map_err(Into::into) + let path = path.to_owned(); + self.remote(IoKind::Write, |operator| async move { + operator.write(&path, bytes).await + }) + .await + .map(|_| ()) } pub(crate) async fn list(&self, prefix: &str) -> Result> { - let mut lister = self.operator.lister_with(prefix).recursive(true).await?; - let mut entries = Vec::new(); - while let Some(entry) = lister.try_next().await? { - if entry.metadata().mode() == EntryMode::FILE { - entries.push(Entry { - path: entry.path().to_string(), - metadata: entry.metadata().clone(), - }); + let prefix = prefix.to_owned(); + self.remote(IoKind::Read, |operator| async move { + let mut lister = operator.lister_with(&prefix).recursive(true).await?; + let mut entries = Vec::new(); + while let Some(entry) = lister.try_next().await? { + if entry.metadata().mode() == EntryMode::FILE { + entries.push(Entry { + path: entry.path().to_string(), + metadata: entry.metadata().clone(), + }); + } } - } - Ok(entries) + Ok(entries) + }) + .await } /// Non-recursive listing of the immediate children under `prefix`. /// Returns both files and directories so callers can navigate lazily. pub(crate) async fn list_shallow(&self, prefix: &str) -> Result> { - let mut lister = self.operator.lister_with(prefix).recursive(false).await?; - let mut entries = Vec::new(); - while let Some(entry) = lister.try_next().await? { - entries.push(ShallowEntry { - path: entry.path().to_string(), - mode: entry.metadata().mode(), - metadata: entry.metadata().clone(), - }); - } - Ok(entries) + let prefix = prefix.to_owned(); + self.remote(IoKind::Read, |operator| async move { + let mut lister = operator.lister_with(&prefix).recursive(false).await?; + let mut entries = Vec::new(); + while let Some(entry) = lister.try_next().await? { + entries.push(ShallowEntry { + path: entry.path().to_string(), + mode: entry.metadata().mode(), + metadata: entry.metadata().clone(), + }); + } + Ok(entries) + }) + .await } pub(crate) async fn stat_file(&self, path: &str) -> Result> { - match self.operator.stat(path).await { + let path_owned = path.to_owned(); + match self + .remote(IoKind::Read, |operator| async move { + operator.stat(&path_owned).await + }) + .await + { Ok(metadata) if metadata.mode() == EntryMode::FILE => Ok(Some(Entry { path: path.to_string(), metadata, })), Ok(_) => Ok(None), - Err(error) if error.kind() == ErrorKind::NotFound => Ok(None), + Err(error) + if error + .downcast_ref::() + .is_some_and(|error| error.kind() == ErrorKind::NotFound) => + { + Ok(None) + } Err(error) => Err(error.into()), } } pub(crate) async fn exists(&self) -> Result { - Ok(self - .operator - .lister_with("") - .recursive(true) - .await? - .try_next() - .await? - .is_some()) + self.remote(IoKind::Read, |operator| async move { + Ok(operator + .lister_with("") + .recursive(true) + .await? + .try_next() + .await? + .is_some()) + }) + .await } pub(crate) async fn remove_all(&self) -> Result<()> { - self.operator - .delete_with("") - .recursive(true) - .await - .map_err(Into::into) + self.remote(IoKind::Write, |operator| async move { + operator.delete_with("").recursive(true).await + }) + .await + .map(|_| ()) } pub(crate) async fn remove(&self, path: &str) -> Result<()> { - self.operator - .delete_with(path) - .recursive(true) - .await - .map_err(Into::into) + let path = path.to_owned(); + self.remote(IoKind::Write, |operator| async move { + operator.delete_with(&path).recursive(true).await + }) + .await + .map(|_| ()) } pub(crate) fn fallback_lock(&self) -> Option>> { @@ -283,6 +400,22 @@ impl Store { } } +fn operator_cache_key(uri: &str, normalized: &str) -> String { + let config: BTreeMap<_, _> = std::env::vars() + .filter(|(key, _)| { + key.starts_with("AWS_") + || key.starts_with("AZURE_") + || key.starts_with("GOOGLE_") + || matches!( + key.as_str(), + "HTTP_PROXY" | "HTTPS_PROXY" | "NO_PROXY" | "ALL_PROXY" + ) + }) + .collect(); + let fingerprint = blake3::hash(&serde_json::to_vec(&config).unwrap_or_default()).to_hex(); + format!("{uri}\0{normalized}\0{fingerprint}") +} + pub(crate) fn is_conflict(error: &opendal::Error) -> bool { matches!( error.kind(), @@ -334,6 +467,25 @@ fn normalize_uri(uri: &str) -> Result { mod tests { use super::*; + #[tokio::test] + async fn operator_registry_bounds_clients_without_invalidating_live_handles() -> Result<()> { + let store = Store::from_uri("shared-memory://registry-test").await?; + let mut registry = OperatorRegistry::default(); + let now = Instant::now(); + registry.insert("first".into(), store.operator.clone(), now); + let live = registry.get("first", now).unwrap(); + for n in 0..MAX_CACHED_OPERATORS { + registry.insert(n.to_string(), store.operator.clone(), now + Duration::from_millis(1)); + } + assert_eq!(registry.entries.len(), MAX_CACHED_OPERATORS); + assert!(!registry.entries.contains_key("first")); + live.write("probe", "still alive").await?; + assert_eq!(live.read("probe").await?.to_vec(), b"still alive"); + assert!(registry.get("0", now + OPERATOR_IDLE_TTL + Duration::from_secs(1)).is_none()); + assert!(registry.entries.is_empty()); + Ok(()) + } + #[test] fn unquoted_etag_strips_one_quote_pair() { assert_eq!(unquoted_etag("\"abc\""), Some("abc")); diff --git a/crates/persisting-pchronicle/src/store/persistent_cache.rs b/crates/persisting-pchronicle/src/store/persistent_cache.rs index e2a43ae38..6fa988282 100644 --- a/crates/persisting-pchronicle/src/store/persistent_cache.rs +++ b/crates/persisting-pchronicle/src/store/persistent_cache.rs @@ -15,6 +15,7 @@ pub struct PersistentCache { path: PathBuf, values: RwLock>, disk_lock: Option, + write_gate: tokio::sync::Mutex<()>, } impl PersistentCache where @@ -37,20 +38,42 @@ where path, values: RwLock::new(HashMap::new()), disk_lock: lock.ok(), + write_gate: tokio::sync::Mutex::new(()), }; - if cache.disk_lock.is_some() { + let writable = cache.writable(); + let attempts = if writable { 1 } else { 5 }; + let mut loaded = None; + let mut last_error = None; + for attempt in 0..attempts { match cache.load().await { - Ok(values) => *cache.values.get_mut() = values, + Ok(values) => { + loaded = Some(values); + break; + } Err(error) => { - tracing::warn!(target: "pchronicle.serve", error = %error, "persistent cache unreadable; rebuilding"); - let _ = if cache.path.is_dir() { - std::fs::remove_dir_all(&cache.path) - } else { - std::fs::remove_file(&cache.path) - }; + last_error = Some(error); + if attempt + 1 < attempts { + tokio::time::sleep(std::time::Duration::from_millis( + 50 * (1u64 << attempt), + )) + .await; + } } } } + if let Some(values) = loaded { + *cache.values.get_mut() = values; + } else if let Some(error) = last_error { + tracing::warn!(target: "pchronicle.serve", error = %error, writable, attempts, + "persistent cache unreadable"); + if writable { + let _ = if cache.path.is_dir() { + std::fs::remove_dir_all(&cache.path) + } else { + std::fs::remove_file(&cache.path) + }; + } + } cache } pub fn path(&self) -> &Path { @@ -70,6 +93,7 @@ where if !self.writable() { return Ok(()); } + let _write = self.write_gate.lock().await; let schema = Arc::new(Schema::new(vec![ Field::new("key", DataType::Utf8, false), Field::new("payload", DataType::Utf8, false), diff --git a/crates/persisting-pchronicle/src/store/storyline/mod.rs b/crates/persisting-pchronicle/src/store/storyline/mod.rs index d9a5dac79..a9146448f 100644 --- a/crates/persisting-pchronicle/src/store/storyline/mod.rs +++ b/crates/persisting-pchronicle/src/store/storyline/mod.rs @@ -49,7 +49,7 @@ use std::io::Write; use std::path::{Path, PathBuf}; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; -use std::time::{Duration, SystemTime, UNIX_EPOCH}; +use std::time::{SystemTime, UNIX_EPOCH}; use anyhow::{Context, Result}; use fs2::FileExt; @@ -1906,25 +1906,13 @@ pub(super) async fn open_dataset_uri(uri: &str) -> Result { let mut attempt = 0u32; loop { attempt += 1; - let _permit = crate::store::object_store_io_gate::acquire( - uri, - crate::store::object_store_io_gate::IoKind::Read, - ) - .await; - match Dataset::open(uri).await { - Ok(dataset) => { - crate::store::object_store_io_gate::note_success(uri); - return Ok(dataset); - } + match crate::storage::open_lance_dataset(uri).await { + Ok(dataset) => return Ok(dataset), Err(error) => { let error = anyhow::Error::from(error); if !is_transient_storage_error(&error) { return Err(error).with_context(|| format!("open Lance dataset {uri}")); } - crate::store::object_store_io_gate::note_failure( - uri, - crate::store::object_store_io_gate::IoKind::Read, - ); if attempt >= DATASET_OPEN_MAX_ATTEMPTS { return Err(error).with_context(|| format!("open Lance dataset {uri}")); } @@ -1939,9 +1927,8 @@ pub(super) async fn open_dataset_uri(uri: &str) -> Result { error = %error, "transient object-store error opening Lance dataset; retrying under I/O gate" ); - // Shared AIMD delay is applied on the next acquire(); keep a - // small per-attempt floor so we never spin. - tokio::time::sleep(Duration::from_millis(100)).await; + // Individual HEAD/range requests already fed the shared AIMD + // gate; do not add a second fixed backoff here. } } } @@ -2123,6 +2110,7 @@ async fn maintain_table_layout( } if options.optimize_indices { crate::store::object_store_io_gate::mark_kind( + path.to_string_lossy().as_ref(), crate::store::object_store_io_gate::IoKind::Write, ); crate::store::index_build_progress::note(format!( diff --git a/crates/persisting-pchronicle/src/store/storyline/mutation.rs b/crates/persisting-pchronicle/src/store/storyline/mutation.rs index 6d9f3cfc6..bf5da626a 100644 --- a/crates/persisting-pchronicle/src/store/storyline/mutation.rs +++ b/crates/persisting-pchronicle/src/store/storyline/mutation.rs @@ -359,6 +359,7 @@ async fn write_record_batch_reader( ) -> Result { let uri = path.to_string_lossy().into_owned(); crate::store::object_store_io_gate::mark_kind( + &uri, crate::store::object_store_io_gate::IoKind::Write, ); let mut dataset = InsertBuilder::new(&uri) From e6bb07cfb3d9393c9700945824503125505b381a Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 17:24:06 +0800 Subject: [PATCH 03/22] feat(catalog): enhance manifest caching and dataset browsing capabilities - Introduced `ManifestRefreshReport` to track the status of mount refresh operations, including partial refreshes. - Updated `refresh_mount` method to return a report on the number of directories refreshed and whether the refresh was complete. - Enhanced `BrowseStatus` to include a `partial` field, indicating if the dataset view is incomplete. - Modified `QueryDatasetSummary` to include browsing status, improving dataset visibility in the UI. - Updated various modules to reflect changes in manifest caching and browsing, ensuring consistency across the application. This commit aims to improve the efficiency and user experience of dataset management and browsing in the pChronicle application. --- .../src/server/mod.rs | 5 + .../src/server/ui_cache.rs | 131 ++++++++++++++--- crates/persisting-pchronicle/src/storage.rs | 14 +- .../src/store/blockcache/block.rs | 44 ++++-- .../src/store/catalog/manifest_cache.rs | 137 ++++++++++++++++-- .../src/store/catalog/mod.rs | 4 +- .../src/store/catalog/status.rs | 1 + crates/persisting-pchronicle/src/store/mod.rs | 4 +- .../src/store/object_store_io_gate.rs | 36 +++-- .../src/store/opendal_store.rs | 43 ++++-- pchronicle-web/src/analysis_agent.rs | 1 + pchronicle-web/src/catalog.rs | 12 +- pchronicle-web/src/model.rs | 41 ++++++ pchronicle-web/src/workspace.rs | 8 +- 14 files changed, 396 insertions(+), 85 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 767f9ba4e..831ca209b 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -2346,6 +2346,8 @@ struct QueryCatalog { #[derive(Debug, Serialize)] struct QueryDatasetSummary { + #[serde(skip_serializing_if = "Option::is_none")] + browse: Option, name: String, uri: String, ready_sources: usize, @@ -2703,6 +2705,7 @@ async fn ui_query_catalog(state: &AppState) -> Result, ApiErr for mount in &state.config.datasets { let tree = browse.cached_dataset(mount).await; let (ready_sources, error_sources) = tree + .as_ref() .map(|tree| { tree.children.iter().fold((0, 0), |(ready, failed), child| { ( @@ -2713,6 +2716,7 @@ async fn ui_query_catalog(state: &AppState) -> Result, ApiErr }) .unwrap_or_default(); datasets.push(QueryDatasetSummary { + browse: None, name: mount.name.clone(), uri: mount.uri.clone(), ready_sources, @@ -2771,6 +2775,7 @@ async fn query_tables( .datasets() .iter() .map(|dataset| QueryDatasetSummary { + browse: None, name: dataset.mount.name.clone(), uri: dataset.mount.uri.clone(), ready_sources: dataset.ready_source_count(), diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index 61316a478..d800893fa 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -84,12 +84,13 @@ struct IndexEntry { pub(crate) struct BrowseSnapshot { // Preserve existing Tree wire fields for old Web clients. #[serde(flatten)] - tree: CatalogTree, - browse: BrowseStatus, + pub(super) tree: CatalogTree, + pub(super) browse: BrowseStatus, } #[derive(Clone, Debug, Serialize)] -struct BrowseStatus { +pub(super) struct BrowseStatus { + partial: bool, consistency: CatalogConsistency, state: CatalogState, generation: String, @@ -182,16 +183,6 @@ impl BrowseCoordinator { } } - pub(crate) async fn cached_dataset(&self, mount: &DatasetMount) -> Option { - let key = TreeKey::new(mount, "").ok()?; - self.index - .values - .read() - .await - .get(&key) - .map(|entry| entry.tree.clone()) - } - pub(crate) async fn cached_source_paths(&self, mount: &DatasetMount) -> Vec { let values = self.index.values.read().await; let fingerprint = mount_fingerprint(mount); @@ -207,11 +198,22 @@ impl BrowseCoordinator { paths } + pub(crate) async fn cached_dataset(&self, mount: &DatasetMount) -> Option { + let key = TreeKey::new(mount, "").ok()?; + self.index + .values + .read() + .await + .get(&key) + .map(|entry| entry.tree.clone()) + } + pub(crate) async fn roots(&self, mounts: &[DatasetMount]) -> BrowseSnapshot { let mut tree = catalog_tree_from_mount_specs(mounts); let values = self.index.values.read().await; let mut observed_at = now(); let mut complete = true; + let mut partial = false; let mut refreshing = false; let mut error = None; for mount in mounts { @@ -224,6 +226,7 @@ impl BrowseCoordinator { { error = Some(failure); } + partial |= projection_is_partial(&values, &key); if let Some(entry) = values.get(&key) { observed_at = observed_at.min(entry.observed_at); if let Some(child) = tree.children.iter_mut().find(|c| c.name == mount.name) { @@ -237,6 +240,7 @@ impl BrowseCoordinator { .manifests .summary(&format!("{}\0{}", mount.name, mount_fingerprint(mount))) .await; + partial |= summary.partial; if let Some(child) = tree.children.iter_mut().find(|c| c.name == mount.name) { child.run_count = summary.trajectories as usize; child.dataset_count = Some(summary.datasets as usize); @@ -260,9 +264,12 @@ impl BrowseCoordinator { .fold(0usize, |sum, c| sum.saturating_add(c.failed_count)); BrowseSnapshot { browse: BrowseStatus { + partial, consistency: CatalogConsistency::BestEffort, state: if refreshing { CatalogState::Refreshing + } else if partial { + CatalogState::Partial } else if !complete && error.is_some() { CatalogState::Unavailable } else if !complete || error.is_some() || now() - observed_at >= 30 { @@ -295,20 +302,26 @@ impl BrowseCoordinator { let _ = self.enqueue(&key, None); return Ok(self.snapshot_with_summary(&key, entry).await); } - self.enqueue(&key, None)?; + let error = self.enqueue(&key, None).err().map(|e| e.to_string()); + let refreshing = lock_recover(&self.pending).contains_key(&key); // Cold browse requests return an empty, explicitly refreshing view; // the worker fills the index asynchronously instead of blocking HTTP. let tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]); Ok(BrowseSnapshot { tree, browse: BrowseStatus { + partial: false, consistency: CatalogConsistency::BestEffort, - state: CatalogState::Refreshing, + state: if error.is_some() { + CatalogState::Unavailable + } else { + CatalogState::Refreshing + }, generation: String::new(), observed_at: 0, stale: true, - refreshing: true, - last_error: None, + refreshing, + last_error: error, }, }) } @@ -347,6 +360,7 @@ impl BrowseCoordinator { BrowseSnapshot { tree: entry.tree, browse: BrowseStatus { + partial: false, consistency: CatalogConsistency::BestEffort, state: if lock_recover(&self.pending).contains_key(key) { CatalogState::Refreshing @@ -463,10 +477,44 @@ impl BrowseCoordinator { entry.tree.dataset_count = Some((summary.datasets as usize).max(child_summary.0)); entry.tree.trajectory_count = Some((summary.trajectories as usize).max(child_summary.1)); entry.tree.run_count = entry.tree.trajectory_count.unwrap_or_default(); - self.snapshot(key, entry) + let partial = + summary.partial || projection_is_partial(&*self.index.values.read().await, key); + let mut snapshot = self.snapshot(key, entry); + snapshot.browse.partial = partial; + if partial && !snapshot.browse.refreshing { + snapshot.browse.state = CatalogState::Partial; + } + snapshot } } +// A root observation is not a complete descendant inventory. This also +// exposes the worker's traversal limit without treating unseen directories as empty. +fn projection_is_partial(values: &HashMap, key: &TreeKey) -> bool { + values + .iter() + .filter(|(other, _)| { + other.dataset == key.dataset + && other.uri_fingerprint == key.uri_fingerprint + && (key.prefix.is_empty() + || other.prefix == key.prefix + || other.prefix.starts_with(&format!("{}/", key.prefix))) + }) + .any(|(_, entry)| { + entry + .tree + .children + .iter() + .filter(|child| child.kind == "dir") + .any(|child| { + !values.contains_key(&TreeKey { + prefix: child.path.clone(), + ..key.clone() + }) + }) + }) +} + fn manifest_key(key: &TreeKey, child_prefix: Option<&str>) -> String { let prefix = child_prefix.unwrap_or(&key.prefix); if prefix.is_empty() { @@ -631,9 +679,6 @@ async fn refresh_tree( None }; let location = DatasetLocation::parse(&mount.uri)?; - if location.local_path().is_none() { - tokio::time::sleep(Duration::from_millis(100)).await; - } if let Some(root) = location.local_path() { tokio::fs::metadata(root.join(&key.prefix)) .await @@ -774,6 +819,50 @@ mod tests { DatasetMount::new("test", path.to_string_lossy()).unwrap() } + #[tokio::test] + async fn cold_tree_returns_loading_and_incomplete_descendants_are_partial() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + std::fs::create_dir_all(source.join("nested")).unwrap(); + let mount = mount(&source); + let coordinator = + BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; + let gate = BROWSE_IO.acquire_many(2).await.unwrap(); + let cold = tokio::time::timeout(Duration::from_secs(1), coordinator.tree(&mount, "")) + .await + .unwrap() + .unwrap(); + assert_eq!(cold.tree.dataset.as_deref(), Some("test")); + assert_eq!(cold.browse.state, CatalogState::Refreshing); + assert_eq!(cold.browse.observed_at, 0); + let key = TreeKey::new(&mount, "").unwrap(); + let mut entry = IndexEntry { + tree: CatalogTree::default(), + generation: String::new(), + observed_at: now(), + }; + entry + .tree + .children + .push(super::super::explorer::CatalogTreeChild { + kind: "dir".into(), + path: "nested".into(), + ..Default::default() + }); + let mut values = HashMap::from([(key.clone(), entry.clone())]); + assert!(projection_is_partial(&values, &key)); + entry.tree.children.clear(); + values.insert( + TreeKey { + prefix: "nested".into(), + ..key.clone() + }, + entry, + ); + assert!(!projection_is_partial(&values, &key)); + drop(gate); + } + #[tokio::test] async fn foreground_finishes_while_background_is_blocked_and_drop_cancels_work() { let temp = tempfile::tempdir().unwrap(); diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 7a0133461..832a3b55f 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -92,13 +92,13 @@ pub use crate::store::{ DatasetLocation, DatasetLocationKind, DatasetMount, DatasetResolver, DiscoveredSource, EventFactSnapshot, EventLogLayoutStats, EventWriterFence, ExportOutcome, ImportableObjectEvent, LanceMaintenanceOptions, LanceMaintenanceReport, LeaseAcquireOutcome, LocationSummary, - ManifestCache, ManifestKind, ManifestListing, ManifestReadMode, ManifestStats, NamespacePath, - PathListEntry, PathListKind, PersistentCache, PhysicalColumn, PhysicalDataFile, - PhysicalFileLayout, PhysicalFragment, PhysicalLayout, PhysicalPage, PhysicalPagePreview, - PhysicalPageQuery, PhysicalSource, PhysicalTable, ProjectionSourceSnapshot, QueryScope, - RawEventLanceAppender, RawEventLanceStore, ReplayOutcome, ResolveMode, ResolveTarget, - RunControlStore, ShallowNavEntry, StorylineContentOptions, StorylineContentReadMode, - StorylineDataSource, StorylineDataSourceOptions, StorylineLanceStore, + ManifestCache, ManifestKind, ManifestListing, ManifestReadMode, ManifestRefreshReport, + ManifestStats, NamespacePath, PathListEntry, PathListKind, PersistentCache, PhysicalColumn, + PhysicalDataFile, PhysicalFileLayout, PhysicalFragment, PhysicalLayout, PhysicalPage, + PhysicalPagePreview, PhysicalPageQuery, PhysicalSource, PhysicalTable, + ProjectionSourceSnapshot, QueryScope, RawEventLanceAppender, RawEventLanceStore, ReplayOutcome, + ResolveMode, ResolveTarget, RunControlStore, ShallowNavEntry, StorylineContentOptions, + StorylineContentReadMode, StorylineDataSource, StorylineDataSourceOptions, StorylineLanceStore, StorylineMaintenanceReport, StorylineProjectionLineage, StorylineSearchIndexSuppressGuard, StorylineStreamImportReport, StorylineStreamOptions, StorylineTablePaths, TrajectoryStats, attempt_registry_now_ms, distinct_session_ids_in_run, export_source_dirs, export_story_bundle, diff --git a/crates/persisting-pchronicle/src/store/blockcache/block.rs b/crates/persisting-pchronicle/src/store/blockcache/block.rs index 82321cc71..a13722ed4 100644 --- a/crates/persisting-pchronicle/src/store/blockcache/block.rs +++ b/crates/persisting-pchronicle/src/store/blockcache/block.rs @@ -49,11 +49,16 @@ impl BlockCache { } } pub fn block_path(&self, key: &str, block: u64) -> PathBuf { - self.config.root.join( - blake3::hash(format!("{key}:{block}").as_bytes()) - .to_hex() - .to_string(), - ) + let hash = blake3::hash(format!("{key}:{block}").as_bytes()) + .to_hex() + .to_string(); + // Keep directory fan-out bounded; a single flat cache directory becomes + // expensive once a service has browsed many datasets and versions. + self.config + .root + .join(&hash[..2]) + .join(&hash[2..4]) + .join(hash) } pub async fn get_or_fetch( &self, @@ -64,6 +69,8 @@ impl BlockCache { where F: Future>, { + // Reclaim stale files after a restart as well as after new downloads. + self.schedule_trim(); match tokio::fs::read(path).await { Ok(bytes) if bytes.len() == expected => { self.counters.hits.fetch_add(1, Ordering::Relaxed); @@ -95,7 +102,10 @@ impl BlockCache { source: format!("block length {}, expected {expected}", bytes.len()).into(), }); } - if tokio::fs::create_dir_all(&self.config.root).await.is_ok() { + if tokio::fs::create_dir_all(path.parent().unwrap_or(&self.config.root)) + .await + .is_ok() + { let tmp = path.with_extension(format!("{}.tmp", std::process::id())); if tokio::fs::write(&tmp, &bytes).await.is_ok() { let _ = tokio::fs::rename(&tmp, path).await; @@ -133,16 +143,22 @@ impl BlockCache { tokio::spawn(async move { cache.trim().await }); } pub async fn trim(&self) { - let Ok(mut entries) = tokio::fs::read_dir(&self.config.root).await else { - return; - }; let mut files = Vec::new(); let mut total = 0; - while let Ok(Some(entry)) = entries.next_entry().await { - if let Ok(meta) = entry.metadata().await { - if meta.is_file() { - total += meta.len(); - files.push((meta.modified().ok(), meta.len(), entry.path())); + let mut dirs = vec![self.config.root.clone()]; + while let Some(dir) = dirs.pop() { + let Ok(mut entries) = tokio::fs::read_dir(dir).await else { + continue; + }; + while let Ok(Some(entry)) = entries.next_entry().await { + let path = entry.path(); + if let Ok(meta) = entry.metadata().await { + if meta.is_dir() { + dirs.push(path); + } else if meta.is_file() && path.extension().is_none_or(|ext| ext != "tmp") { + total += meta.len(); + files.push((meta.modified().ok(), meta.len(), path)); + } } } } diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index 8172d36c0..08dbbd099 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -8,7 +8,7 @@ use std::path::PathBuf; use std::sync::Arc; use std::time::Duration; -use anyhow::{Context, Result}; +use anyhow::Result; use serde::{Deserialize, Serialize}; use tokio::sync::RwLock; @@ -20,16 +20,27 @@ use crate::store::{DatasetLocation, PathListEntry, PersistentCache}; #[derive(Debug, Clone, Serialize, Deserialize)] pub struct ManifestListing { + #[serde(default)] + pub partial: bool, pub entries: Vec, pub observed_at: i64, } #[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct LocationSummary { + #[serde(default)] + pub partial: bool, pub datasets: u64, pub trajectories: u64, } +/// Outcome of one bounded mount walk. A partial walk is usable, but not complete. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct ManifestRefreshReport { + pub refreshed_directories: usize, + pub partial: bool, +} + #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ManifestReadMode { Cached, @@ -94,6 +105,7 @@ impl ManifestCache { crate::store::root_write_lock::for_root(&serde_json::to_string(&(self.path(), &key))?); let _guard = refresh_gate.lock().await; let listing = ManifestListing { + partial: false, entries: location.list(prefix).await?, observed_at: chrono::Utc::now().timestamp(), }; @@ -168,6 +180,7 @@ impl ManifestCache { }) }) .fold(LocationSummary::default(), |mut total, (_, listing)| { + total.partial |= listing.partial; for entry in &listing.entries { let is_dataset = matches!(entry.kind, crate::store::PathListKind::Dataset) || (matches!(entry.kind, crate::store::PathListKind::File) @@ -185,15 +198,36 @@ impl ManifestCache { /// Breadth-first refresh of a mount. Refreshes serialize per observation /// key, allowing unrelated foreground directories to load concurrently. - pub async fn refresh_mount(&self, key_prefix: &str, location: &DatasetLocation) -> Result<()> { - tracing::info!(target: "pchronicle.serve", %key_prefix, "manifest cache refresh started"); + pub async fn refresh_mount( + &self, + key_prefix: &str, + location: &DatasetLocation, + ) -> Result { + self.refresh_mount_bounded( + key_prefix, + location, + MAX_REFRESH_DIRECTORIES, + REFRESH_MOUNT_DEADLINE, + ) + .await + } + + async fn refresh_mount_bounded( + &self, + key_prefix: &str, + location: &DatasetLocation, + max_directories: usize, + budget: Duration, + ) -> Result { let mut queue = VecDeque::from([String::new()]); - let mut refreshed = 0usize; - let deadline = tokio::time::Instant::now() + REFRESH_MOUNT_DEADLINE; + let mut seen = HashSet::from([String::new()]); + let mut report = ManifestRefreshReport::default(); + let deadline = tokio::time::Instant::now() + budget; while let Some(prefix) = queue.pop_front() { - if refreshed >= MAX_REFRESH_DIRECTORIES || tokio::time::Instant::now() >= deadline { - tracing::warn!(target: "pchronicle.serve", %key_prefix, refreshed, - queued = queue.len(), "manifest cache refresh reached safety bound; keeping partial view"); + if report.refreshed_directories >= max_directories + || tokio::time::Instant::now() >= deadline + { + report.partial = true; break; } let key = if prefix.is_empty() { @@ -201,24 +235,60 @@ impl ManifestCache { } else { format!("{key_prefix}\0{prefix}") }; - let listing = tokio::time::timeout( - REFRESH_DIRECTORY_TIMEOUT, + let directory_deadline = + deadline.min(tokio::time::Instant::now() + REFRESH_DIRECTORY_TIMEOUT); + let listing = match tokio::time::timeout_at( + directory_deadline, self.refresh(key, location, &prefix), ) .await - .context("manifest directory refresh timed out")??; - refreshed += 1; + { + Ok(result) => result?, + Err(_) => { + report.partial = true; + break; + } + }; + report.refreshed_directories += 1; for child in listing .entries .iter() .filter(|e| matches!(e.kind, crate::store::PathListKind::Directory)) { + if seen.contains(&child.path) { + continue; + } + if seen.len() >= max_directories { + report.partial = true; + continue; + } + seen.insert(child.path.clone()); queue.push_back(child.path.clone()); } tokio::task::yield_now().await; } - tracing::info!(target: "pchronicle.serve", %key_prefix, refreshed, "manifest cache refresh finished"); - Ok(()) + // Publish completeness alongside the cached root, including across restarts. + // Do not hold the values lock over persistence. + let root = { + let mut values = self.values.write().await; + values.get_mut(key_prefix).map(|root| { + root.partial = report.partial; + root.clone() + }) + }; + if let Some(root) = root { + if let Err(error) = self + .disk + .upsert(&key_prefix.to_owned(), &serde_json::to_value(root)?, &[]) + .await + { + tracing::warn!(target: "pchronicle.serve", %error, "mount completeness persistence failed"); + } + } + tracing::info!(target: "pchronicle.serve", %key_prefix, + refreshed = report.refreshed_directories, partial = report.partial, + "manifest cache refresh finished"); + Ok(report) } pub fn spawn_periodic_refresh( @@ -245,6 +315,40 @@ mod tests { use super::*; use crate::store::{PathListEntry, PathListKind}; + #[tokio::test] + async fn bounded_walk_reports_and_persists_partial_then_recovers() { + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().join("source"); + std::fs::create_dir_all(root.join("child")).unwrap(); + let location = DatasetLocation::parse(root.to_str().unwrap()).unwrap(); + let path = dir.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let partial = cache + .refresh_mount_bounded("mount", &location, 1, Duration::from_secs(20)) + .await + .unwrap(); + assert_eq!( + partial, + ManifestRefreshReport { + refreshed_directories: 1, + partial: true + } + ); + assert!(cache.summary("mount").await.partial); + drop(cache); + let cache = ManifestCache::open(path).await; + assert!(cache.summary("mount").await.partial); + let complete = cache.refresh_mount("mount", &location).await.unwrap(); + assert!(!complete.partial); + assert!(!cache.summary("mount").await.partial); + let timed_out = cache + .refresh_mount_bounded("mount", &location, 10, Duration::ZERO) + .await + .unwrap(); + assert!(timed_out.partial); + assert_eq!(timed_out.refreshed_directories, 0); + } + #[tokio::test] async fn refresh_survives_disk_failure_and_unrelated_refresh_lock() { let dir = tempfile::tempdir().unwrap(); @@ -295,6 +399,7 @@ mod tests { cache.values.write().await.insert( "mount".into(), ManifestListing { + partial: false, entries: vec![PathListEntry { name: "a".into(), path: "a".into(), @@ -309,6 +414,7 @@ mod tests { assert_eq!( cache.summary("mount").await, LocationSummary { + partial: false, datasets: 1, trajectories: 3 } @@ -316,6 +422,7 @@ mod tests { cache.values.write().await.insert( "mount\0nested".into(), ManifestListing { + partial: false, entries: vec![PathListEntry { name: "b".into(), path: "nested/b".into(), @@ -369,6 +476,7 @@ mod tests { assert_eq!( cache.summary_under(prefix).await, LocationSummary { + partial: false, datasets, trajectories }, @@ -393,6 +501,7 @@ mod tests { assert_eq!( cache.summary_under("rfs\0fingerprint\0nested").await, LocationSummary { + partial: false, datasets: 2, trajectories: 708 } diff --git a/crates/persisting-pchronicle/src/store/catalog/mod.rs b/crates/persisting-pchronicle/src/store/catalog/mod.rs index accd2888f..41842a3c3 100644 --- a/crates/persisting-pchronicle/src/store/catalog/mod.rs +++ b/crates/persisting-pchronicle/src/store/catalog/mod.rs @@ -30,7 +30,9 @@ pub use manifest::{ load_manifest_at_uri, try_load_manifest, write_compact_jsonl_manifest, write_storyline_manifest, write_storyline_manifest_at_uri, }; -pub use manifest_cache::{LocationSummary, ManifestCache, ManifestListing, ManifestReadMode}; +pub use manifest_cache::{ + LocationSummary, ManifestCache, ManifestListing, ManifestReadMode, ManifestRefreshReport, +}; pub use namespace::{CatalogNamespace, CatalogPage, CatalogSourceDescription}; use provider::*; pub use resolver::{CachedDataset, Dataset, DatasetResolver, ResolveMode, ResolveTarget}; diff --git a/crates/persisting-pchronicle/src/store/catalog/status.rs b/crates/persisting-pchronicle/src/store/catalog/status.rs index 5c5456efc..365ea84b8 100644 --- a/crates/persisting-pchronicle/src/store/catalog/status.rs +++ b/crates/persisting-pchronicle/src/store/catalog/status.rs @@ -15,6 +15,7 @@ pub enum CatalogConsistency { #[serde(rename_all = "snake_case")] pub enum CatalogState { Ready, + Partial, Stale, Refreshing, Unavailable, diff --git a/crates/persisting-pchronicle/src/store/mod.rs b/crates/persisting-pchronicle/src/store/mod.rs index 70119379f..6e7c77d1e 100644 --- a/crates/persisting-pchronicle/src/store/mod.rs +++ b/crates/persisting-pchronicle/src/store/mod.rs @@ -93,8 +93,8 @@ pub use catalog::{ CatalogStorylineKey, CatalogTrajectoryBundle, DEFAULT_DATASET_NAME, DEFAULT_MAX_EVENT_FALLBACK_BYTES, DEFAULT_MAX_EVENT_FALLBACK_ROWS, Dataset, DatasetCatalogSnapshot, DatasetMount, DatasetResolver, DiscoveredSource, LocationSummary, - ManifestCache, ManifestListing, ManifestReadMode, NamespacePath, QueryScope, ResolveMode, - ResolveTarget, + ManifestCache, ManifestListing, ManifestReadMode, ManifestRefreshReport, NamespacePath, + QueryScope, ResolveMode, ResolveTarget, }; #[cfg(feature = "lance-store")] pub use compact_jsonl::{ diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 7e4c436b4..44cd23dfc 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -158,16 +158,25 @@ struct Gate { fn state_for<'a>(states: &'a mut HashMap, key: &str) -> &'a mut AimdState { let now = Instant::now(); if !states.contains_key(key) { - states.retain(|_, state| state.active_waiters > 0 - || state.cooldown_until.is_some_and(|until| until > now) - || now.duration_since(state.last_used) < SCOPE_IDLE_TTL); - if states.len() >= MAX_RETAINED_SCOPES { - let oldest = states.iter() - .filter(|(_, state)| state.active_waiters == 0 - && state.cooldown_until.is_none_or(|until| until <= now)) + states.retain(|_, state| { + state.active_waiters > 0 + || state.cooldown_until.is_some_and(|until| until > now) + || now.duration_since(state.last_used) < SCOPE_IDLE_TTL + }); + while states.len() >= MAX_RETAINED_SCOPES { + let oldest = states + .iter() + .filter(|(_, state)| { + state.active_waiters == 0 + && state.cooldown_until.is_none_or(|until| until <= now) + }) .min_by_key(|(_, state)| state.last_used) .map(|(key, _)| key.clone()); - if let Some(oldest) = oldest { states.remove(&oldest); } + if let Some(oldest) = oldest { + states.remove(&oldest); + } else { + break; + } } } // Live waits/cooldowns may temporarily exceed the retention limit. Evicting @@ -513,7 +522,9 @@ async fn wait_out_degradation(g: &Gate, key: &str, kind: IoKind) { /// Publish the current I/O phase for progress UI without taking a permit. /// Used around Lance writes that do not go through [`acquire`]. pub(crate) fn mark_kind(uri: &str, kind: IoKind) { - if !is_remote_uri(uri) { return; } + if !is_remote_uri(uri) { + return; + } if let Ok(mut states) = gate().states.lock() { let state = state_for(&mut states, &scope_key(uri)); state.last_kind = kind; @@ -619,7 +630,8 @@ mod tests { fn registry_reclaims_idle_scopes_but_preserves_waits_and_cooldowns() { let mut states = HashMap::new(); state_for(&mut states, "waiting").active_waiters = 1; - state_for(&mut states, "cooling").cooldown_until = Some(Instant::now() + Duration::from_secs(60)); + state_for(&mut states, "cooling").cooldown_until = + Some(Instant::now() + Duration::from_secs(60)); for n in 0..MAX_RETAINED_SCOPES * 2 { state_for(&mut states, &n.to_string()); } @@ -627,7 +639,9 @@ mod tests { assert!(states.contains_key("waiting")); assert!(states.contains_key("cooling")); assert!(!states.contains_key("0")); - for state in states.values_mut() { state.last_used = Instant::now() - SCOPE_IDLE_TTL; } + for state in states.values_mut() { + state.last_used = Instant::now() - SCOPE_IDLE_TTL; + } state_for(&mut states, "new"); assert_eq!(states.len(), 3); } diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index c192a3052..53e69ed2f 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -85,7 +85,8 @@ struct OperatorRegistry { impl OperatorRegistry { fn get(&mut self, key: &str, now: Instant) -> Option { - self.entries.retain(|_, (_, used)| now.duration_since(*used) < OPERATOR_IDLE_TTL); + self.entries + .retain(|_, (_, used)| now.duration_since(*used) < OPERATOR_IDLE_TTL); self.entries.get_mut(key).map(|(operator, used)| { *used = now; operator.clone() @@ -94,8 +95,12 @@ impl OperatorRegistry { fn insert(&mut self, key: String, operator: Operator, now: Instant) { if !self.entries.contains_key(&key) && self.entries.len() >= MAX_CACHED_OPERATORS { - if let Some(oldest) = self.entries.iter().min_by_key(|(_, (_, used))| *used) - .map(|(key, _)| key.clone()) { + if let Some(oldest) = self + .entries + .iter() + .min_by_key(|(_, (_, used))| *used) + .map(|(key, _)| key.clone()) + { self.entries.remove(&oldest); } } @@ -116,8 +121,10 @@ impl Store { .unwrap_or(false); // Memory operators own the data itself and must not be evicted like clients. let operator = if normalized.starts_with("memory://") { - let mut map = SHARED_MEMORY.get_or_init(|| Mutex::new(HashMap::new())) - .lock().unwrap_or_else(std::sync::PoisonError::into_inner); + let mut map = SHARED_MEMORY + .get_or_init(|| Mutex::new(HashMap::new())) + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); if let Some(operator) = map.get(uri) { operator.clone() } else { @@ -127,15 +134,21 @@ impl Store { } } else { let registry = OPERATORS.get_or_init(|| Mutex::new(OperatorRegistry::default())); - let cached = registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner) + let cached = registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) .get(&cache_key, Instant::now()); if let Some(operator) = cached { operator } else { // Construct outside the registry lock so one backend cannot block all others. - let operator = with_object_store_retries(Operator::from_uri(normalized.as_str()) - .with_context(|| format!("open OpenDAL store {uri}"))?); - registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner) + let operator = with_object_store_retries( + Operator::from_uri(normalized.as_str()) + .with_context(|| format!("open OpenDAL store {uri}"))?, + ); + registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) .insert(cache_key, operator.clone(), Instant::now()); operator } @@ -475,13 +488,21 @@ mod tests { registry.insert("first".into(), store.operator.clone(), now); let live = registry.get("first", now).unwrap(); for n in 0..MAX_CACHED_OPERATORS { - registry.insert(n.to_string(), store.operator.clone(), now + Duration::from_millis(1)); + registry.insert( + n.to_string(), + store.operator.clone(), + now + Duration::from_millis(1), + ); } assert_eq!(registry.entries.len(), MAX_CACHED_OPERATORS); assert!(!registry.entries.contains_key("first")); live.write("probe", "still alive").await?; assert_eq!(live.read("probe").await?.to_vec(), b"still alive"); - assert!(registry.get("0", now + OPERATOR_IDLE_TTL + Duration::from_secs(1)).is_none()); + assert!( + registry + .get("0", now + OPERATOR_IDLE_TTL + Duration::from_secs(1)) + .is_none() + ); assert!(registry.entries.is_empty()); Ok(()) } diff --git a/pchronicle-web/src/analysis_agent.rs b/pchronicle-web/src/analysis_agent.rs index 3549fea53..656557820 100644 --- a/pchronicle-web/src/analysis_agent.rs +++ b/pchronicle-web/src/analysis_agent.rs @@ -1287,6 +1287,7 @@ mod tests { fn plan_prompt_sends_only_approved_catalog_and_scope_context() { let mut catalog = catalog(); catalog.datasets = vec![QueryDatasetSummary { + browse: None, name: "private-dataset".into(), uri: "s3://secret-bucket/?token=private".into(), ready_sources: 17, diff --git a/pchronicle-web/src/catalog.rs b/pchronicle-web/src/catalog.rs index dacbb1d4e..ba69d065d 100644 --- a/pchronicle-web/src/catalog.rs +++ b/pchronicle-web/src/catalog.rs @@ -34,7 +34,11 @@ pub fn CatalogExplorer( p { "{catalog_subtitle(tree.as_ref())}" } if let Some(status) = tree.as_ref().and_then(|tree| tree.browse.as_ref()) { p { role: "status", - if status.last_error.is_some() { + if status.partial { + "Partial view · Some directories have not been loaded" + } else if status.observed_at == 0 && status.refreshing { + "Loading directory contents…" + } else if status.last_error.is_some() { "Showing cached view · Refresh failed; retrying automatically" } else if status.refreshing { "Showing cached view · Refreshing in background" @@ -55,7 +59,8 @@ pub fn CatalogExplorer( CatalogStats { tree: tree.clone() } } div { class: "pc-catalog-mosaic", - if loading && tree.is_none() { + if (loading && tree.is_none()) || tree.as_ref().and_then(|tree| tree.browse.as_ref()) + .is_some_and(|status| status.observed_at == 0 && status.refreshing) { div { class: "pc-catalog-empty", span { class: "spinner" } "Loading datasets…" } } else if auth_required { div { class: "pc-catalog-empty", @@ -63,6 +68,9 @@ pub fn CatalogExplorer( span { "Add an access key and secret key to browse this catalog." } button { class: "button primary", onclick: on_settings, "Open Keys" } } + } else if tree.as_ref().and_then(|tree| tree.browse.as_ref()) + .is_some_and(|status| status.observed_at == 0 && status.last_error.is_some()) { + div { class: "pc-catalog-empty", strong { "Directory unavailable" } span { "Retrying automatically. Check the storage connection if this persists." } } } else if tree.as_ref().is_none_or(|tree| tree.children.is_empty() && tree.run_count == 0) { div { class: "pc-catalog-empty", strong { "No datasets" } span { "Add a dataset, then refresh this page." } } } else if tree.as_ref().is_some_and(|tree| tree.children.is_empty()) { diff --git a/pchronicle-web/src/model.rs b/pchronicle-web/src/model.rs index 55fa79f34..83ba66899 100644 --- a/pchronicle-web/src/model.rs +++ b/pchronicle-web/src/model.rs @@ -116,6 +116,8 @@ pub struct QueryCatalog { #[derive(Clone, Debug, PartialEq, Eq, Deserialize)] pub struct QueryDatasetSummary { + #[serde(default)] + pub browse: Option, pub name: String, pub uri: String, pub ready_sources: usize, @@ -773,12 +775,14 @@ mod tests { path_column: "_file_".into(), datasets: vec![ QueryDatasetSummary { + browse: None, name: "atif".into(), uri: "atif".into(), ready_sources: 1, error_sources: 0, }, QueryDatasetSummary { + browse: None, name: "actf".into(), uri: "actf".into(), ready_sources: 1, @@ -844,8 +848,45 @@ mod tests { /// Directory observations are independent of the query's pinned revisions. #[derive(Clone, Debug, PartialEq, Eq, Deserialize)] pub struct BrowseStatus { + #[serde(default)] + pub partial: bool, + #[serde(default)] + pub state: String, pub observed_at: i64, pub stale: bool, pub refreshing: bool, pub last_error: Option, } + +impl QueryDatasetSummary { + pub fn label(&self) -> String { + let suffix = self.browse.as_ref().map(|status| { + if status.observed_at == 0 && status.last_error.is_some() { " · Unavailable" } + else if status.observed_at == 0 { " · Loading…" } + else if status.partial { " · Partial" } + else if status.stale { " · Cached" } + else { "" } + }).unwrap_or(""); + format!("{}{suffix}", self.name) + } +} + +#[cfg(test)] +mod browse_status_tests { + use super::*; + #[test] + fn distinguish_unknown_counts_from_an_empty_loaded_dataset() { + let mut dataset: QueryDatasetSummary = serde_json::from_value(serde_json::json!({ + "name": "mount", "uri": "s3://bucket", "ready_sources": 0, "error_sources": 0, + "browse": {"observed_at": 0, "refreshing": true, "stale": true, "last_error": null} + })).unwrap(); + assert_eq!(dataset.label(), "mount · Loading…"); + dataset.browse.as_mut().unwrap().observed_at = 1; + dataset.browse.as_mut().unwrap().stale = false; + assert_eq!(dataset.label(), "mount"); + dataset.browse.as_mut().unwrap().partial = true; + assert_eq!(dataset.label(), "mount · Partial"); + dataset.browse = None; + assert_eq!(dataset.label(), "mount"); + } +} diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 9598a9148..9c77f82a9 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -490,8 +490,12 @@ pub fn App() -> Element { if page() == "home" { return; } - if catalog().is_none() { + let initial = catalog().is_none(); + let waiting = catalog().as_ref().is_some_and(|catalog| catalog.datasets.iter() + .any(|dataset| dataset.browse.as_ref().is_some_and(|status| status.observed_at == 0))); + if initial || waiting { spawn(async move { + if !initial { TimeoutFuture::new(CATALOG_REFRESH_MS).await; } match api::query_catalog().await { Ok(value) => { if selected_table().is_empty() { @@ -1536,7 +1540,7 @@ fn RunsExplorer( label { class: "pc2-filter-search", span { "⌕" } input { value: "{query}", placeholder: "{search_placeholder}", aria_label: "Search runs and content", oninput: move |event| on_query.call(event.value()), onkeydown: move |event| { if event.key() == Key::Enter { event.prevent_default(); on_apply_query.call(query.clone()); } } } if !query.is_empty() { button { r#type: "button", class: "pc2-filter-clear", aria_label: "Clear run search", title: "Clear search", onclick: move |event| { event.prevent_default(); on_apply_query.call(String::new()); }, "×" } } } select { value: "{dataset}", aria_label: "Filter by Dataset", onchange: move |event| on_dataset.call(event.value()), option { value: "all", "All Datasets" } - for mounted in datasets { option { value: "{mounted.name}", "{mounted.name}" } } + for mounted in datasets { option { value: "{mounted.name}", "{mounted.label()}" } } } select { value: "{status}", aria_label: "Filter by run status", onchange: move |event| on_status.call(event.value()), option { value: "all", "All statuses" } option { value: "active", "Active" } option { value: "completed", "Completed" } option { value: "failed", "Failed" } } select { value: "{sort}", aria_label: "Sort runs", onchange: move |event| on_sort.call(event.value()), option { value: "session", "Session" } option { value: "events", "Events" } option { value: "status", "Status" } option { value: "agent", "Agent" } } From c7ecd70fc716abb2ffb2a11f128136fd67e3229b Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 17:25:31 +0800 Subject: [PATCH 04/22] fix(workspace): improve error handling in dataset loading - Refactored the dataset loading logic to handle errors from the catalog query more gracefully. - Added error reporting to notify users when catalog discovery fails, ensuring a better user experience. - Preserved the functionality for all-dataset requests, allowing for continued usability even when some scans fail. This commit aims to enhance the robustness of dataset loading and improve user feedback during errors. --- pchronicle-web/src/workspace.rs | 35 ++++++++++++++++++++++----------- 1 file changed, 23 insertions(+), 12 deletions(-) diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 9c77f82a9..28b350f15 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -936,16 +936,18 @@ fn load_runs( spawn(async move { let all_datasets = filters.dataset.trim().is_empty() || filters.dataset == "all"; let dataset_names = if all_datasets { - api::query_catalog() - .await - .map(|catalog| { - catalog - .datasets - .into_iter() - .map(|dataset| dataset.name) - .collect::>() - }) - .unwrap_or_default() + match api::query_catalog().await { + Ok(catalog) => catalog + .datasets + .into_iter() + .map(|dataset| dataset.name) + .collect::>(), + Err(failure) => { + error.set(Some(workspace_notice(&failure))); + loading.set(false); + return; + } + } } else { vec![filters.dataset.clone()] }; @@ -988,8 +990,17 @@ fn load_runs( } } if partials.is_empty() { - // Preserve the previous all-Dataset behavior if catalog discovery - // failed before fan-out could be started. + // Keep the selected dataset path usable even when one scoped scan + // fails; for an all-dataset request the catalog error was already + // surfaced above, so an empty page is preferable to a second + // unscoped remote scan. + if all_datasets { + loading.set(false); + if let Some(failure) = first_error { + error.set(Some(workspace_notice(&failure))); + } + return; + } match api::explorer_runs( &filters.query, &filters.dataset, From af441ceb7be73796f68a6ffb8f4b3e2f27b93e4b Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 17:45:56 +0800 Subject: [PATCH 05/22] feat(server): implement timeout for run scans and enhance error handling - Introduced a `RUNS_SCAN_TIMEOUT` constant to enforce a timeout on run scan operations, improving responsiveness and preventing long-running queries from blocking. - Updated the `load_run_summaries` function to utilize the timeout, ensuring that both catalog building and run summary retrieval are subject to this limit. - Enhanced error handling to provide clearer feedback when timeouts occur, improving user experience during dataset loading. This commit aims to improve the robustness and reliability of the server's handling of run scans. --- .../src/server/mod.rs | 89 ++++++++++++------- pchronicle-web/src/workspace.rs | 9 +- 2 files changed, 65 insertions(+), 33 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 831ca209b..f82fd4df1 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -73,6 +73,7 @@ struct AppState { } const DEFAULT_CATALOG_REFRESH_INTERVAL: Duration = Duration::from_secs(30); +const RUNS_SCAN_TIMEOUT: Duration = Duration::from_secs(60); #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct HomeLink { @@ -718,29 +719,34 @@ async fn load_run_summaries( }; let query_metrics = metrics.cloned(); let execute = move |background: bool| async move { - // Background work must not be attributed to the triggering HTTP request. - let metrics = query_metrics.filter(|_| !background); - let phase = Instant::now(); - let runtime = - build_scoped_query_runtime(&config, query_scope.clone(), cached_files.clone()) + let summaries = tokio::time::timeout(RUNS_SCAN_TIMEOUT, async { + // Background work must not be attributed to the triggering HTTP request. + let metrics = query_metrics.filter(|_| !background); + let phase = Instant::now(); + let runtime = + build_scoped_query_runtime(&config, query_scope.clone(), cached_files.clone()) + .await?; + if let Some(metrics) = &metrics { + metrics.record("summary_catalog", phase); + } + let phase = Instant::now(); + let summaries = runtime + .acceleration + .scoped_run_summaries( + &runtime.snapshot, + &runtime.engine, + Some(&query_scope.dataset), + query_scope.source_file.as_deref(), + ) .await?; - if let Some(metrics) = &metrics { - metrics.record("summary_catalog", phase); - } - let phase = Instant::now(); - let summaries = runtime - .acceleration - .scoped_run_summaries( - &runtime.snapshot, - &runtime.engine, - Some(&query_scope.dataset), - query_scope.source_file.as_deref(), - ) - .await; - if let Some(metrics) = &metrics { - metrics.record("summary_sql", phase); - } - summaries + if let Some(metrics) = &metrics { + metrics.record("summary_sql", phase); + } + anyhow::Ok(summaries) + }) + .await + .context("runs scan timed out")??; + Ok(summaries) }; let started = Instant::now(); let result = if state.live_reads { @@ -766,21 +772,44 @@ async fn load_run_summaries( return result; } let phase = Instant::now(); - let runtime = current_catalog_for_runs(state, request_id).await?; + let runtime = tokio::time::timeout( + RUNS_SCAN_TIMEOUT, + current_catalog_for_runs(state, request_id), + ) + .await + .map_err(|error| { + fail( + request_id, + "load_run_summaries", + anyhow::anyhow!("catalog build timed out: {error}"), + ) + })??; if let Some(metrics) = metrics { metrics.record("summary_catalog", phase); } let phase = Instant::now(); - let summaries = runtime - .acceleration - .scoped_run_summaries(&runtime.snapshot, &runtime.engine, dataset, file) - .await; + let summaries = tokio::time::timeout( + RUNS_SCAN_TIMEOUT, + runtime.acceleration.scoped_run_summaries( + &runtime.snapshot, + &runtime.engine, + dataset, + file, + ), + ) + .await + .map_err(|error| { + fail( + request_id, + "load_run_summaries", + anyhow::anyhow!("runs scan timed out: {error}"), + ) + })? + .map_err(|error| fail(request_id, "load_run_summaries", error))?; if let Some(metrics) = metrics { metrics.record("summary_sql", phase); } - summaries - .map(|summaries| summaries.as_ref().clone()) - .map_err(|error| fail(request_id, "load_run_summaries", error)) + Ok(summaries.as_ref().clone()) } fn api_query(query: Result, QueryRejection>) -> Result { diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 28b350f15..8c2b3f2f3 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -933,6 +933,7 @@ fn load_runs( ) { page.set(None); loading.set(true); + error.set(None); spawn(async move { let all_datasets = filters.dataset.trim().is_empty() || filters.dataset == "all"; let dataset_names = if all_datasets { @@ -989,6 +990,11 @@ fn load_runs( } } } + if let Some(failure) = &first_error { + // Keep successful datasets visible, but never present a partial + // all-dataset result as complete. + error.set(Some(workspace_notice(failure))); + } if partials.is_empty() { // Keep the selected dataset path usable even when one scoped scan // fails; for an all-dataset request the catalog error was already @@ -996,9 +1002,6 @@ fn load_runs( // unscoped remote scan. if all_datasets { loading.set(false); - if let Some(failure) = first_error { - error.set(Some(workspace_notice(&failure))); - } return; } match api::explorer_runs( From 575a4d22f54afc69e654a2b14458978cc37b5317 Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 17:49:47 +0800 Subject: [PATCH 06/22] feat(server): implement structured timeout handling for run operations - Introduced a new `with_runs_deadline` function to manage timeouts for run-related operations, enhancing responsiveness and error handling. - Updated the `RUNS_SCAN_TIMEOUT` constant to 50 seconds, ensuring timely cancellation of long-running tasks. - Enhanced the `ApiError` struct with a `runs_timeout` method to provide clearer error messages when timeouts occur. - Added tests to verify the timeout behavior and error reporting for run scans, improving reliability and user feedback. This commit aims to improve the server's handling of run operations by enforcing timeouts and providing structured error responses. --- .../src/server/mod.rs | 104 ++++++++++-------- .../src/server/problem.rs | 8 ++ .../src/server/tests.rs | 60 ++++++++++ 3 files changed, 128 insertions(+), 44 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index f82fd4df1..50feca9f1 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -73,7 +73,22 @@ struct AppState { } const DEFAULT_CATALOG_REFRESH_INTERVAL: Duration = Duration::from_secs(30); -const RUNS_SCAN_TIMEOUT: Duration = Duration::from_secs(60); +// Return a structured timeout before the catalog worker's 60s transport cutoff. +const RUNS_SCAN_TIMEOUT: Duration = Duration::from_secs(50); + +async fn with_runs_deadline( + timeout: Duration, + request_id: &RequestId, + operation: impl std::future::Future>, +) -> Result { + tokio::time::timeout(timeout, operation) + .await + .map_err(|_| { + ApiError::runs_timeout() + .with_request_id(request_id.as_str()) + .with_stage(ExecutionStage::Query) + })? +} #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct HomeLink { @@ -683,7 +698,12 @@ async fn runs( metrics: RequestMetrics, ) -> Result>, ApiError> { let started = Instant::now(); - let summaries = load_run_summaries(&state, None, None, &request_id, Some(&metrics)).await?; + let summaries = with_runs_deadline( + RUNS_SCAN_TIMEOUT, + &request_id, + load_run_summaries(&state, None, None, &request_id, Some(&metrics)), + ) + .await?; metrics.record("summary_total", started); Ok(Json(summaries)) } @@ -703,17 +723,19 @@ async fn load_run_summaries( }; let config = state.config.clone(); let query_scope = scope.clone(); + let mount = config + .datasets + .iter() + .find(|mount| mount.name == dataset) + .ok_or_else(|| { + ApiError::not_found("dataset was not found").with_request_id(request_id.as_str()) + })?; + // Browse initialization and its disk projection are optional for Runs. let cached_files = if file.is_none() && !state.live_reads { - browse_coordinator(state) - .await - .cached_source_paths( - config - .datasets - .iter() - .find(|mount| mount.name == dataset) - .expect("validated dataset mount"), - ) - .await + match state.browse.get() { + Some(browse) => browse.cached_source_paths(mount).await, + None => Vec::new(), + } } else { Vec::new() }; @@ -772,40 +794,16 @@ async fn load_run_summaries( return result; } let phase = Instant::now(); - let runtime = tokio::time::timeout( - RUNS_SCAN_TIMEOUT, - current_catalog_for_runs(state, request_id), - ) - .await - .map_err(|error| { - fail( - request_id, - "load_run_summaries", - anyhow::anyhow!("catalog build timed out: {error}"), - ) - })??; + let runtime = current_catalog_for_runs(state, request_id).await?; if let Some(metrics) = metrics { metrics.record("summary_catalog", phase); } let phase = Instant::now(); - let summaries = tokio::time::timeout( - RUNS_SCAN_TIMEOUT, - runtime.acceleration.scoped_run_summaries( - &runtime.snapshot, - &runtime.engine, - dataset, - file, - ), - ) - .await - .map_err(|error| { - fail( - request_id, - "load_run_summaries", - anyhow::anyhow!("runs scan timed out: {error}"), - ) - })? - .map_err(|error| fail(request_id, "load_run_summaries", error))?; + let summaries = runtime + .acceleration + .scoped_run_summaries(&runtime.snapshot, &runtime.engine, dataset, file) + .await + .map_err(|error| fail(request_id, "load_run_summaries", error))?; if let Some(metrics) = metrics { metrics.record("summary_sql", phase); } @@ -1142,6 +1140,21 @@ async fn explorer_runs( metrics: RequestMetrics, fts: FtsDiagnostics, query: Result, QueryRejection>, +) -> Result, ApiError> { + with_runs_deadline( + RUNS_SCAN_TIMEOUT, + &request_id, + explorer_runs_inner(State(state), request_id.clone(), metrics, fts, query), + ) + .await +} + +async fn explorer_runs_inner( + State(state): State, + request_id: RequestId, + metrics: RequestMetrics, + fts: FtsDiagnostics, + query: Result, QueryRejection>, ) -> Result, ApiError> { let query = api_query(query)?; if let Some(page) = try_compact_jsonl_runs_page(&state, &query, &request_id).await? { @@ -2710,7 +2723,7 @@ struct QueryTablesQuery { } async fn ui_query_catalog(state: &AppState) -> Result, ApiError> { - let browse = browse_coordinator(state).await; + let browse = state.browse.get(); let default_name = state .config .default_dataset @@ -2732,7 +2745,10 @@ async fn ui_query_catalog(state: &AppState) -> Result, ApiErr .unwrap_or_default(); let mut datasets = Vec::with_capacity(state.config.datasets.len()); for mount in &state.config.datasets { - let tree = browse.cached_dataset(mount).await; + let tree = match browse { + Some(browse) => browse.cached_dataset(mount).await, + None => None, + }; let (ready_sources, error_sources) = tree .as_ref() .map(|tree| { diff --git a/crates/persisting-pchronicle-cli/src/server/problem.rs b/crates/persisting-pchronicle-cli/src/server/problem.rs index 7d519b4c8..f3f65b5a4 100644 --- a/crates/persisting-pchronicle-cli/src/server/problem.rs +++ b/crates/persisting-pchronicle-cli/src/server/problem.rs @@ -221,6 +221,14 @@ impl ApiError { ) } + pub(super) fn runs_timeout() -> Self { + Self::public( + StatusCode::GATEWAY_TIMEOUT, + BoundaryCode::Unavailable, + "Runs request timed out; narrow the dataset or file scope and retry", + ) + } + pub(super) fn input(issue: InputIssue) -> Self { let message = issue.message().to_owned(); match issue.kind() { diff --git a/crates/persisting-pchronicle-cli/src/server/tests.rs b/crates/persisting-pchronicle-cli/src/server/tests.rs index 71c44a081..d56681ced 100644 --- a/crates/persisting-pchronicle-cli/src/server/tests.rs +++ b/crates/persisting-pchronicle-cli/src/server/tests.rs @@ -2676,3 +2676,63 @@ async fn browse_tree_does_not_build_query_runtime() -> anyhow::Result<()> { ); Ok(()) } + +#[tokio::test] +async fn runs_metadata_does_not_initialize_browse_or_resolve_sources() { + let root = tempfile::tempdir().unwrap(); + let config = ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.path().join("missing").to_string_lossy().to_string()).unwrap(), + ]) + .unwrap(); + let state = app_state(config); + let Json(catalog) = ui_query_catalog(&state).await.unwrap(); + assert_eq!(catalog.datasets.len(), 1); + assert_eq!(catalog.datasets[0].name, "dataset"); + assert!(state.browse.get().is_none()); + assert!(state.catalog.read().await.is_none()); +} + +#[tokio::test] +async fn runs_unknown_dataset_is_a_structured_error_without_panicking() { + let root = tempfile::tempdir().unwrap(); + let app = router(root.path().to_string_lossy().to_string()); + let (status, response) = get_json(&app, "/api/explorer/runs?dataset=missing").await; + assert_eq!(status, StatusCode::NOT_FOUND, "{response}"); + assert_eq!(response["code"], "not_found"); + let (status, _) = get_json(&app, "/api/query/tables?ui=true").await; + assert_eq!(status, StatusCode::OK); +} + +#[tokio::test] +async fn runs_deadline_includes_catalog_lock_and_releases_cancelled_work() { + let root = tempfile::tempdir().unwrap(); + write_gateway_fixture(root.path(), "run.json", "session", "job"); + let config = ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.path().to_string_lossy().to_string()).unwrap(), + ]) + .unwrap(); + let state = app_state(config); + let held = state.catalog_refresh.lock().await; + let request_id = RequestId("runs-timeout-test".into()); + let error = with_runs_deadline( + Duration::from_millis(20), + &request_id, + load_run_summaries(&state, None, None, &request_id, None), + ) + .await + .unwrap_err(); + assert_eq!(error.status, StatusCode::GATEWAY_TIMEOUT); + let body = serde_json::to_value(error).unwrap(); + assert_eq!(body["code"], "unavailable"); + assert_eq!(body["request_id"], "runs-timeout-test"); + assert_eq!(body["stage"], "query"); + drop(held); + let summaries = with_runs_deadline( + Duration::from_secs(5), + &request_id, + load_run_summaries(&state, None, None, &request_id, None), + ) + .await + .unwrap(); + assert_eq!(summaries.len(), 1); +} From 0a2c85547795159bd64cc485885516b30d01e06d Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 17:55:41 +0800 Subject: [PATCH 07/22] feat(workspace): add run generation tracking to load_runs function - Introduced a new signal for tracking the generation of run requests, allowing for better management of concurrent requests. - Updated the load_runs function to utilize the generation signal, ensuring that only the latest request's results are processed. - Enhanced error handling and loading state management to prevent outdated responses from interfering with the current request. This commit aims to improve the responsiveness and accuracy of run data loading in the workspace component. --- pchronicle-web/src/workspace.rs | 33 ++++++++++++++++++++++++++------- 1 file changed, 26 insertions(+), 7 deletions(-) diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 8c2b3f2f3..fd01ec3e5 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -245,6 +245,7 @@ pub fn App() -> Element { let mut page = use_signal(move || initial_page.to_string()); let runs = use_signal(|| None::); let runs_loading = use_signal(|| true); + let runs_generation = use_signal(|| 0u64); let initial_query = url_param("q").unwrap_or_default(); let mut query = use_signal({ let initial_query = initial_query.clone(); @@ -338,6 +339,7 @@ pub fn App() -> Element { }, runs, runs_loading, + runs_generation, error, ); }); @@ -828,7 +830,7 @@ pub fn App() -> Element { if let Ok(value) = api::query_catalog().await { catalog.set(Some(value)); } - load_runs(filters, runs, runs_loading, error); + load_runs(filters, runs, runs_loading, runs_generation, error); }); }, on_page: move |value| offset.set(value), @@ -929,8 +931,11 @@ fn load_runs( filters: RunFilters, mut page: Signal>, mut loading: Signal, + mut generation: Signal, mut error: Signal>, ) { + let request_generation = generation() + 1; + generation.set(request_generation); page.set(None); loading.set(true); error.set(None); @@ -944,6 +949,9 @@ fn load_runs( .map(|dataset| dataset.name) .collect::>(), Err(failure) => { + if generation() != request_generation { + return; + } error.set(Some(workspace_notice(&failure))); loading.set(false); return; @@ -983,7 +991,9 @@ fn load_runs( match result { Ok(value) => { partials.push(value); - page.set(Some(merge_run_pages(&partials, &filters))); + if generation() == request_generation { + page.set(Some(merge_run_pages(&partials, &filters))); + } } Err(message) => { first_error.get_or_insert(message); @@ -993,7 +1003,9 @@ fn load_runs( if let Some(failure) = &first_error { // Keep successful datasets visible, but never present a partial // all-dataset result as complete. - error.set(Some(workspace_notice(failure))); + if generation() == request_generation { + error.set(Some(workspace_notice(failure))); + } } if partials.is_empty() { // Keep the selected dataset path usable even when one scoped scan @@ -1001,7 +1013,9 @@ fn load_runs( // surfaced above, so an empty page is preferable to a second // unscoped remote scan. if all_datasets { - loading.set(false); + if generation() == request_generation { + loading.set(false); + } return; } match api::explorer_runs( @@ -1017,11 +1031,16 @@ fn load_runs( ) .await { - Ok(value) => page.set(Some(value)), - Err(failure) => error.set(Some(workspace_notice(&first_error.unwrap_or(failure)))), + Ok(value) if generation() == request_generation => page.set(Some(value)), + Err(failure) if generation() == request_generation => { + error.set(Some(workspace_notice(&first_error.unwrap_or(failure)))) + } + _ => {} } } - loading.set(false); + if generation() == request_generation { + loading.set(false); + } }); } From 6e2f1abd4622b29dfd21ee2f251b6aea8ef8d092 Mon Sep 17 00:00:00 2001 From: Reiase Date: Tue, 15 Sep 2026 19:10:08 +0800 Subject: [PATCH 08/22] feat(catalog): enhance dataset browsing and manifest refresh capabilities - Introduced `list_for_browse` and `refresh_for_browse` methods to improve the efficiency of browsing datasets without probing every child immediately. - Updated `list_impl` and `list_nav_children` methods to support conditional probing of child datasets, optimizing performance during directory navigation. - Enhanced the `BrowseCoordinator` to leverage the new browsing capabilities, allowing for quicker access to dataset summaries without unnecessary requests. This commit aims to improve the user experience and performance of dataset browsing in the pChronicle application. --- .../src/server/ui_cache.rs | 27 ++++++++++++++---- .../src/store/catalog/location.rs | 26 ++++++++++++++--- .../src/store/catalog/manifest_cache.rs | 28 +++++++++++++++++-- 3 files changed, 69 insertions(+), 12 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index d800893fa..c89e246a2 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -425,10 +425,8 @@ impl BrowseCoordinator { uri_fingerprint: key.uri_fingerprint.clone(), prefix: child_prefix.trim_matches('/').to_owned(), }; - // A directory can be discovered after the background walk started. - // Schedule it here so its descendant manifests become available to - // the next render without making the request wait on storage I/O. - let _ = self.enqueue(&child_key, None); + // Child discovery belongs to the bounded background walk. Rendering + // a wide directory must not promote every child to foreground work. let cached = self.index.values.read().await.get(&child_key).cloned(); tracing::info!( target: "pchronicle.serve", @@ -441,6 +439,19 @@ impl BrowseCoordinator { "catalog directory summary" ); if let Some(cached) = cached { + // Remote parents initially contain names only. Reuse the child's + // own leaf observation without another request to S3. + if let Some(leaf) = cached + .tree + .children + .iter() + .find(|leaf| leaf.kind == "file" && leaf.path == child_prefix) + { + let name = child.name.clone(); + *child = leaf.clone(); + child.name = name; + continue; + } if let Some(count) = cached.tree.dataset_count.filter(|count| *count > 0) { child.dataset_count = Some(count); } @@ -556,6 +567,8 @@ async fn run_worker( let mut visited = HashSet::new(); let mut background_budget = 0usize; let mut background_running = false; + // FIFO is intentional: children are appended only after their parent + // completes, so the background walk is breadth-first (shallow to deep). // JoinSet aborts outstanding work when the coordinator is dropped. let mut jobs = tokio::task::JoinSet::<(TreeKey, bool, Result)>::new(); loop { @@ -689,9 +702,11 @@ async fn refresh_tree( } else { format!("{}\0{}\0{}", key.dataset, key.uri_fingerprint, key.prefix) }; + // Only inspect this prefix: child marker probes and statistics are deferred + // to the bounded background walk, so a wide directory can be cached promptly. let listing = tokio::time::timeout( - Duration::from_secs(20), - manifests.refresh(manifest_key, &location, &key.prefix), + Duration::from_secs(60), + manifests.refresh_for_browse(manifest_key, &location, &key.prefix), ) .await .context("browse list timed out")??; diff --git a/crates/persisting-pchronicle/src/store/catalog/location.rs b/crates/persisting-pchronicle/src/store/catalog/location.rs index e0e929ff7..8fdfe9065 100644 --- a/crates/persisting-pchronicle/src/store/catalog/location.rs +++ b/crates/persisting-pchronicle/src/store/catalog/location.rs @@ -353,6 +353,16 @@ impl DatasetLocation { /// directories, leaf Datasets (with sidecar preview when present), and /// JSON / JSONL / NDJSON files. pub async fn list(&self, relative: &str) -> Result> { + self.list_impl(relative, true).await + } + + /// Browse one remote level without probing every child. A child is shown + /// as a directory until its own observation identifies it as a Dataset. + pub(crate) async fn list_for_browse(&self, relative: &str) -> Result> { + self.list_impl(relative, !self.is_object_store()).await + } + + async fn list_impl(&self, relative: &str, probe_children: bool) -> Result> { let relative = relative.trim().trim_matches('/'); anyhow::ensure!( !relative.split('/').any(|part| part == ".."), @@ -379,7 +389,7 @@ impl DatasetLocation { }]); } - let nav = self.list_nav_children(relative).await?; + let nav = self.list_nav_children(relative, probe_children).await?; let mut out = Vec::with_capacity(nav.len()); for entry in nav { let path = if relative.is_empty() { @@ -491,11 +501,15 @@ impl DatasetLocation { if self.probe_nav_dataset_kind(relative).await?.is_some() { return Ok(Vec::new()); } - self.list_nav_children(relative).await + self.list_nav_children(relative, true).await } // Caller has validated the path and established that it is not a Dataset leaf. - async fn list_nav_children(&self, relative: &str) -> Result> { + async fn list_nav_children( + &self, + relative: &str, + probe_children: bool, + ) -> Result> { if let Some(root) = &self.local_path { let dir = if relative.is_empty() { root.clone() @@ -599,7 +613,11 @@ impl DatasetLocation { } else { format!("{relative}/{name}") }; - let kind = self.probe_nav_dataset_kind(&child_rel).await?; + let kind = if probe_children { + self.probe_nav_dataset_kind(&child_rel).await? + } else { + None + }; Ok::<_, anyhow::Error>(ShallowNavEntry { name, is_dir: kind.is_none(), diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index 08dbbd099..da6e3bb83 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -100,13 +100,37 @@ impl ManifestCache { location: &DatasetLocation, prefix: &str, ) -> Result { - let key = key.into(); + self.refresh_impl(key.into(), location, prefix, false).await + } + + /// Observe remote child names now; the browse worker resolves each child's + /// type and statistics when visiting that prefix, without delaying its parent. + pub async fn refresh_for_browse( + &self, + key: impl Into, + location: &DatasetLocation, + prefix: &str, + ) -> Result { + self.refresh_impl(key.into(), location, prefix, true).await + } + + async fn refresh_impl( + &self, + key: String, + location: &DatasetLocation, + prefix: &str, + browse: bool, + ) -> Result { let refresh_gate = crate::store::root_write_lock::for_root(&serde_json::to_string(&(self.path(), &key))?); let _guard = refresh_gate.lock().await; let listing = ManifestListing { partial: false, - entries: location.list(prefix).await?, + entries: if browse { + location.list_for_browse(prefix).await? + } else { + location.list(prefix).await? + }, observed_at: chrono::Utc::now().timestamp(), }; self.values From 07fb8da9d87170803c870085d543287c650890d7 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 05:49:32 +0800 Subject: [PATCH 09/22] feat(object_store): enhance concurrency control and error handling - Updated the `AimdState` structure to manage concurrency per endpoint and bucket, improving remote operation handling. - Refactored the `state_for` function to incorporate concurrency limits, ensuring better resource management during I/O operations. - Introduced a new `is_transient_error` function to classify errors more accurately, enhancing the robustness of error handling in remote operations. - Improved logging for manifest cache persistence and dataset browsing, providing clearer insights into operation statuses. This commit aims to optimize the performance and reliability of object store interactions in the pChronicle application. --- .../src/server/catalog.rs | 16 ++- .../src/server/mod.rs | 3 + .../src/server/ui_cache.rs | 66 +++++---- .../src/store/catalog/manifest_cache.rs | 6 +- .../src/store/object_store_io_gate.rs | 126 ++++++++++++++---- .../src/store/opendal_store.rs | 74 +++++++--- 6 files changed, 218 insertions(+), 73 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/catalog.rs b/crates/persisting-pchronicle-cli/src/server/catalog.rs index a9945e032..7a94ee97b 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog.rs @@ -141,6 +141,16 @@ impl CatalogAcl { Some(user) } + pub(crate) fn apply_public_backend_env(&self) { + if let Some(library) = self.public_datasets.iter().find_map(|name| { + self.libraries + .get(name) + .filter(|library| library.uri.starts_with("s3://")) + }) { + apply_library_env(library); + } + } + fn authenticate_headers( &self, headers: &axum::http::HeaderMap, @@ -702,7 +712,11 @@ pub(crate) fn apply_library_env(library: &CatalogLibrary) { } } } - if let Some(region) = library.region.as_deref() { + if let Some(region) = library + .region + .as_deref() + .or_else(|| library.uri.starts_with("s3://").then_some("us-west-2")) + { unsafe { std::env::set_var("AWS_REGION", region); std::env::set_var("AWS_DEFAULT_REGION", region); diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 50feca9f1..26bff53fe 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -279,6 +279,9 @@ impl PreparedWarehouse { acl: catalog::CatalogAcl, mut config: ChronicleServerConfig, ) -> anyhow::Result { + // The browse worker starts before any HTTP request can select a + // dataset. Seed its shared OpenDAL S3 configuration from the catalog. + acl.apply_public_backend_env(); let browse_mounts = acl .public_for_all() .into_iter() diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index c89e246a2..d60893452 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -16,6 +16,7 @@ use tokio::sync::{RwLock, mpsc, oneshot}; use super::explorer::{CatalogTree, catalog_tree_from_mount_specs, catalog_tree_from_path_list}; const REFRESH_INTERVAL: Duration = Duration::from_secs(30); +const FOREGROUND_REFRESH_TIMEOUT: Duration = Duration::from_secs(5); const QUEUE_CAPACITY: usize = 128; // Keep browse work bounded while allowing a foreground request to run beside // one background walk. @@ -302,28 +303,40 @@ impl BrowseCoordinator { let _ = self.enqueue(&key, None); return Ok(self.snapshot_with_summary(&key, entry).await); } - let error = self.enqueue(&key, None).err().map(|e| e.to_string()); - let refreshing = lock_recover(&self.pending).contains_key(&key); - // Cold browse requests return an empty, explicitly refreshing view; - // the worker fills the index asynchronously instead of blocking HTTP. - let tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]); - Ok(BrowseSnapshot { - tree, - browse: BrowseStatus { - partial: false, - consistency: CatalogConsistency::BestEffort, - state: if error.is_some() { - CatalogState::Unavailable - } else { - CatalogState::Refreshing - }, - generation: String::new(), - observed_at: 0, - stale: true, - refreshing, - last_error: error, - }, - }) + // A cold page is user-visible work: attach to the same single-flight + // refresh as the background walker and return once this prefix exists. + let (reply, wait) = oneshot::channel(); + self.enqueue(&key, Some(reply))?; + match tokio::time::timeout(FOREGROUND_REFRESH_TIMEOUT, wait).await { + Ok(result) => result + .context("browse refresh task stopped")? + .map_err(anyhow::Error::msg)?, + Err(_) => { + let tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]); + return Ok(BrowseSnapshot { + tree, + browse: BrowseStatus { + partial: true, + consistency: CatalogConsistency::BestEffort, + state: CatalogState::Refreshing, + generation: String::new(), + observed_at: 0, + stale: true, + refreshing: true, + last_error: Some("remote browse is still loading".into()), + }, + }); + } + } + let entry = self + .index + .values + .read() + .await + .get(&key) + .cloned() + .context("browse refresh completed without a cached view")?; + Ok(self.snapshot_with_summary(&key, entry).await) } fn enqueue(&self, key: &TreeKey, reply: Option) -> Result<()> { @@ -842,14 +855,16 @@ mod tests { let mount = mount(&source); let coordinator = BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; - let gate = BROWSE_IO.acquire_many(2).await.unwrap(); let cold = tokio::time::timeout(Duration::from_secs(1), coordinator.tree(&mount, "")) .await .unwrap() .unwrap(); assert_eq!(cold.tree.dataset.as_deref(), Some("test")); - assert_eq!(cold.browse.state, CatalogState::Refreshing); - assert_eq!(cold.browse.observed_at, 0); + assert!(matches!( + cold.browse.state, + CatalogState::Ready | CatalogState::Partial + )); + assert!(cold.browse.observed_at > 0); let key = TreeKey::new(&mount, "").unwrap(); let mut entry = IndexEntry { tree: CatalogTree::default(), @@ -875,7 +890,6 @@ mod tests { entry, ); assert!(!projection_is_partial(&values, &key)); - drop(gate); } #[tokio::test] diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index da6e3bb83..7e8241325 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -142,10 +142,12 @@ impl ManifestCache { .upsert(&key, &serde_json::to_value(&listing)?, &[]) .await { - tracing::warn!(target: "pchronicle.serve", %key, %error, + tracing::warn!(target: "pchronicle.serve", prefix, %error, "manifest cache persistence failed; using memory"); } - tracing::info!(target: "pchronicle.serve", %key, prefix, entries = listing.entries.len(), "manifest cache updated"); + // `key` contains NUL separators and is an internal cache identity; + // logging it makes journald truncate the record at the dataset name. + tracing::debug!(target: "pchronicle.serve", prefix, entries = listing.entries.len(), browse, "manifest cache updated"); Ok(listing) } diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 44cd23dfc..c0c5dc489 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -3,7 +3,7 @@ //! Lance opens and table writes against flaky S3-compatible gateways amplify //! timeouts when several datasets race (list `_versions/`, retries, AIMD inside //! object_store). This gate: -//! 1. caps concurrent remote Lance ops (default 1); +//! 1. caps concurrent remote ops per endpoint + bucket (default 1); //! 2. after a transient failure, forces a shared cooldown + growing delay; //! 3. decays the delay after a streak of successes. //! @@ -122,6 +122,7 @@ fn emit_throttle(event: ObjectStoreThrottleEvent) { #[derive(Debug, Clone)] struct AimdState { + semaphore: Arc, last_used: Instant, /// Extra sleep applied before each remote acquire while degraded. delay_ms: u64, @@ -138,6 +139,7 @@ struct AimdState { impl Default for AimdState { fn default() -> Self { Self { + semaphore: Arc::new(Semaphore::new(DEFAULT_REMOTE_CONCURRENCY)), last_used: Instant::now(), delay_ms: 0, cooldown_until: None, @@ -150,16 +152,20 @@ impl Default for AimdState { } struct Gate { - semaphore: Arc, concurrency: usize, states: Mutex>, } -fn state_for<'a>(states: &'a mut HashMap, key: &str) -> &'a mut AimdState { +fn state_for<'a>( + states: &'a mut HashMap, + key: &str, + concurrency: usize, +) -> &'a mut AimdState { let now = Instant::now(); if !states.contains_key(key) { states.retain(|_, state| { - state.active_waiters > 0 + Arc::strong_count(&state.semaphore) > 1 + || state.active_waiters > 0 || state.cooldown_until.is_some_and(|until| until > now) || now.duration_since(state.last_used) < SCOPE_IDLE_TTL }); @@ -167,7 +173,8 @@ fn state_for<'a>(states: &'a mut HashMap, key: &str) -> &'a m let oldest = states .iter() .filter(|(_, state)| { - state.active_waiters == 0 + Arc::strong_count(&state.semaphore) == 1 + && state.active_waiters == 0 && state.cooldown_until.is_none_or(|until| until <= now) }) .min_by_key(|(_, state)| state.last_used) @@ -181,7 +188,10 @@ fn state_for<'a>(states: &'a mut HashMap, key: &str) -> &'a m } // Live waits/cooldowns may temporarily exceed the retention limit. Evicting // them would let backend overload bypass AIMD; reclaim after they finish. - let state = states.entry(key.to_owned()).or_default(); + let state = states.entry(key.to_owned()).or_insert_with(|| AimdState { + semaphore: Arc::new(Semaphore::new(concurrency)), + ..Default::default() + }); state.last_used = now; state } @@ -195,7 +205,6 @@ fn gate() -> &'static Gate { .unwrap_or(DEFAULT_REMOTE_CONCURRENCY) .clamp(1, MAX_REMOTE_CONCURRENCY); Gate { - semaphore: Arc::new(Semaphore::new(concurrency)), concurrency, states: Mutex::new(HashMap::new()), } @@ -283,7 +292,7 @@ pub(crate) fn is_transient_error(error: &object_store::Error) -> bool { /// Snapshot AIMD / cooldown state for progress UI. pub fn snapshot() -> ObjectStoreGateSnapshot { let g = gate(); - let available_permits = g.semaphore.available_permits(); + let available_permits = g.concurrency; let max_permits = g.concurrency; let Ok(states) = g.states.lock() else { return ObjectStoreGateSnapshot { @@ -325,7 +334,7 @@ pub fn snapshot() -> ObjectStoreGateSnapshot { success_streak: state.successes_since_backoff, success_streak_target: SUCCESS_STREAK_TO_DECAY, active_waiters: state.active_waiters, - available_permits, + available_permits: state.semaphore.available_permits(), max_permits, } } @@ -374,16 +383,22 @@ pub(crate) async fn acquire(uri: &str, kind: IoKind) -> Permit { } async fn acquire_scoped(g: &Gate, key: &str, kind: IoKind) -> Permit { - if let Ok(mut states) = g.states.lock() { - state_for(&mut states, key).last_kind = kind; - } + let semaphore = { + let mut states = g + .states + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let state = state_for(&mut states, key, g.concurrency); + state.last_kind = kind; + Arc::clone(&state.semaphore) + }; loop { wait_out_degradation(g, key, kind).await; - let permit = match g.semaphore.clone().try_acquire_owned() { + let permit = match semaphore.clone().try_acquire_owned() { Ok(permit) => permit, Err(_) => { let _wait = WaitGuard::new(g, key, kind, "admit", 0); - match g.semaphore.clone().acquire_owned().await { + match semaphore.clone().acquire_owned().await { Ok(permit) => permit, Err(error) => { tracing::error!(?error, "object-store I/O semaphore closed unexpectedly"); @@ -436,7 +451,7 @@ fn enter_wait(g: &Gate, key: &str, kind: IoKind, reason: &'static str, wait_ms: let Ok(mut states) = g.states.lock() else { return; }; - let state = state_for(&mut states, key); + let state = state_for(&mut states, key, g.concurrency); state.last_kind = kind; state.active_waiters = state.active_waiters.saturating_add(1); (state.delay_ms, state.failures) @@ -452,7 +467,7 @@ fn enter_wait(g: &Gate, key: &str, kind: IoKind, reason: &'static str, wait_ms: fn leave_wait(g: &Gate, key: &str, kind: IoKind) { if let Ok(mut states) = g.states.lock() { - let state = state_for(&mut states, key); + let state = state_for(&mut states, key, g.concurrency); state.active_waiters = state.active_waiters.saturating_sub(1); } emit_throttle(ObjectStoreThrottleEvent::Leave { kind }); @@ -499,7 +514,9 @@ async fn wait_out_degradation(g: &Gate, key: &str, kind: IoKind) { kind.as_str(), sleep_for.as_secs_f32() )); - tracing::warn!( + // A cooldown can have several queued callers; the failure itself is + // already logged by `note_failure`, so each waiter need not be WARN noise. + tracing::debug!( target: "pchronicle.object_store_gate", kind = kind.as_str(), wait_ms, @@ -526,7 +543,7 @@ pub(crate) fn mark_kind(uri: &str, kind: IoKind) { return; } if let Ok(mut states) = gate().states.lock() { - let state = state_for(&mut states, &scope_key(uri)); + let state = state_for(&mut states, &scope_key(uri), gate().concurrency); state.last_kind = kind; } } @@ -541,7 +558,7 @@ pub(crate) fn note_success(uri: &str) { let Ok(mut states) = gate().states.lock() else { return; }; - let state = state_for(&mut states, &key); + let state = state_for(&mut states, &key, gate().concurrency); let kind = state.last_kind; state.successes_since_backoff = state.successes_since_backoff.saturating_add(1); if state.delay_ms == 0 { @@ -589,7 +606,7 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { let Ok(mut states) = gate().states.lock() else { return; }; - let state = state_for(&mut states, &key); + let state = state_for(&mut states, &key, gate().concurrency); state.last_kind = kind; state.failures = state.failures.saturating_add(1); state.successes_since_backoff = 0; @@ -601,6 +618,7 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { state.cooldown_until = Some(Instant::now() + Duration::from_millis(state.delay_ms)); tracing::warn!( target: "pchronicle.object_store_gate", + scope = %key, kind = kind.as_str(), delay_ms = state.delay_ms, failures = state.failures, @@ -629,11 +647,11 @@ mod tests { #[test] fn registry_reclaims_idle_scopes_but_preserves_waits_and_cooldowns() { let mut states = HashMap::new(); - state_for(&mut states, "waiting").active_waiters = 1; - state_for(&mut states, "cooling").cooldown_until = + state_for(&mut states, "waiting", 1).active_waiters = 1; + state_for(&mut states, "cooling", 1).cooldown_until = Some(Instant::now() + Duration::from_secs(60)); for n in 0..MAX_RETAINED_SCOPES * 2 { - state_for(&mut states, &n.to_string()); + state_for(&mut states, &n.to_string(), 1); } assert_eq!(states.len(), MAX_RETAINED_SCOPES); assert!(states.contains_key("waiting")); @@ -642,7 +660,7 @@ mod tests { for state in states.values_mut() { state.last_used = Instant::now() - SCOPE_IDLE_TTL; } - state_for(&mut states, "new"); + state_for(&mut states, "new", 1); assert_eq!(states.len(), 3); } @@ -667,16 +685,62 @@ mod tests { ); } + #[tokio::test] + async fn busy_backend_does_not_block_other_endpoints_or_buckets() { + let g = Gate { + concurrency: 1, + states: Mutex::new(HashMap::new()), + }; + let busy = scope_for_endpoint("s3://bucket/a", "http://slow"); + let held = acquire_scoped(&g, &busy, IoKind::Read).await; + for independent in [ + scope_for_endpoint("s3://bucket/a", "http://healthy"), + scope_for_endpoint("s3://other/a", "http://slow"), + ] { + let permit = tokio::time::timeout( + Duration::from_millis(200), + acquire_scoped(&g, &independent, IoKind::Read), + ) + .await + .unwrap(); + drop(permit); + } + // Registry reclamation must not replace a semaphore with a live permit. + { + let mut states = g.states.lock().unwrap(); + states.get_mut(&busy).unwrap().last_used = Instant::now() - SCOPE_IDLE_TTL; + for n in 0..MAX_RETAINED_SCOPES + 1 { + state_for(&mut states, &n.to_string(), g.concurrency); + } + assert!(states.contains_key(&busy)); + } + assert!( + tokio::time::timeout( + Duration::from_millis(50), + acquire_scoped(&g, &busy, IoKind::Read) + ) + .await + .is_err() + ); + assert_eq!(g.states.lock().unwrap()[&busy].active_waiters, 0); + drop(held); + let _permit = tokio::time::timeout( + Duration::from_millis(200), + acquire_scoped(&g, &busy, IoKind::Read), + ) + .await + .unwrap(); + } + #[tokio::test] async fn cooldown_after_queued_admission_releases_capacity_and_cancellation_clears_waiters() { let g = Arc::new(Gate { - semaphore: Arc::new(Semaphore::new(1)), concurrency: 1, states: Mutex::new(HashMap::new()), }); - let held = g.semaphore.clone().acquire_owned().await.unwrap(); let bad = scope_for_endpoint("s3://bucket/a", "http://slow"); let healthy = scope_for_endpoint("s3://bucket/a", "http://healthy"); + let held = acquire_scoped(&g, &bad, IoKind::Read).await; let waiter = tokio::spawn({ let g = g.clone(); let bad = bad.clone(); @@ -702,8 +766,7 @@ mod tests { .unwrap() .cooldown_until = Some(Instant::now() + Duration::from_secs(10)); drop(held); - // The queued unhealthy request is first in the semaphore's FIFO. It - // must give up its slot on discovering the newly imposed cooldown. + // Cooling on one endpoint must not block another endpoint. let permit = tokio::time::timeout( Duration::from_millis(500), acquire_scoped(&g, &healthy, IoKind::Read), @@ -715,7 +778,12 @@ mod tests { assert!(waiter.await.is_err()); assert_eq!(g.states.lock().unwrap()[&bad].active_waiters, 0); drop(permit); - assert_eq!(g.semaphore.available_permits(), 1); + assert_eq!( + g.states.lock().unwrap()[&healthy] + .semaphore + .available_permits(), + 1 + ); } #[test] diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 53e69ed2f..2462c5cfe 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -186,21 +186,10 @@ impl Store { Ok(value) } Err(error) => { - let text = error.to_string().to_ascii_lowercase(); - if [ - "timeout", - "connection", - "broken pipe", - "temporarily", - "slowdown", - "throttl", - "503", - "429", - "reset", - ] - .iter() - .any(|needle| text.contains(needle)) - { + if is_transient_error(&error) { + tracing::warn!(target: "pchronicle.opendal", + scope = %self.io_scope, kind = kind.as_str(), error = %error, + "remote operation failed; applying shared cooldown"); io_gate::note_failure(&self.io_scope, kind); } Err(error.into()) @@ -450,6 +439,13 @@ pub(crate) fn version(metadata: &Metadata) -> Version { } } +fn is_transient_error(error: &opendal::Error) -> bool { + // RetryLayer marks all returned errors persistent, even 404/403/412. + // Never classify by response headers or request IDs in the display text. + matches!(error.kind(), ErrorKind::Unexpected | ErrorKind::RateLimited) + && (error.is_temporary() || error.is_persistent()) +} + fn normalize_uri(uri: &str) -> Result { if !uri.contains("://") { let path = std::path::Path::new(uri); @@ -480,6 +476,54 @@ fn normalize_uri(uri: &str) -> Result { mod tests { use super::*; + #[test] + fn temporary_and_exhausted_transport_errors_still_trigger_backoff() { + for kind in [ErrorKind::Unexpected, ErrorKind::RateLimited] { + assert!(is_transient_error( + &opendal::Error::new(kind, "backend failure").set_temporary() + )); + assert!(is_transient_error( + &opendal::Error::new(kind, "backend failure").set_persistent() + )); + } + assert!(!is_transient_error( + &opendal::Error::new(ErrorKind::Unexpected, "invalid response").set_permanent() + )); + assert!(!is_transient_error( + &opendal::Error::new(ErrorKind::ConfigInvalid, "connection").set_persistent() + )); + } + + #[tokio::test] + async fn missing_markers_with_connection_headers_do_not_trigger_cooldown() -> Result<()> { + // RetryLayer marks even non-retryable errors persistent. S3 includes + // response headers in the error context, including `connection`. + let mut store = Store::from_uri("shared-memory://missing-marker-gate-test").await?; + store.io_scope = io_gate::scope_key("s3://missing-marker-gate-test"); + for kind in [ + ErrorKind::NotFound, + ErrorKind::PermissionDenied, + ErrorKind::ConditionNotMatch, + ] { + let error = store + .remote::<(), _, _>(IoKind::Read, |_| async move { + Err(opendal::Error::new(kind, "S3 response") + .with_context("response", "connection: keep-alive; request-id: 503429") + .set_persistent()) + }) + .await + .unwrap_err(); + assert_eq!(error.downcast_ref::().unwrap().kind(), kind); + tokio::time::timeout( + Duration::from_millis(200), + store.remote(IoKind::Read, |_| async { Ok(()) }), + ) + .await + .context("non-transient response started AIMD cooldown")??; + } + Ok(()) + } + #[tokio::test] async fn operator_registry_bounds_clients_without_invalidating_live_handles() -> Result<()> { let store = Store::from_uri("shared-memory://registry-test").await?; From 43a9502aef3646a8cfce4ed61c7e4c280311ce13 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 06:13:53 +0800 Subject: [PATCH 10/22] feat(model): enhance dataset browsing and status management - Updated `CatalogTree` and `BrowseStatus` structs to include default values, improving initialization consistency. - Refactored the `label` method in `QueryDatasetSummary` for better readability and maintainability. - Introduced a new signal for tracking catalog generation, enhancing concurrency control during dataset loading. - Improved error handling and loading state management in the `load_catalog_tree` function to ensure accurate responses based on request generation. This commit aims to optimize dataset browsing and improve the user experience in the pChronicle application. --- pchronicle-web/src/model.rs | 31 +++++++++++----- pchronicle-web/src/workspace.rs | 65 ++++++++++++++++++++++++++------- 2 files changed, 73 insertions(+), 23 deletions(-) diff --git a/pchronicle-web/src/model.rs b/pchronicle-web/src/model.rs index 83ba66899..1b5ad7822 100644 --- a/pchronicle-web/src/model.rs +++ b/pchronicle-web/src/model.rs @@ -124,7 +124,7 @@ pub struct QueryDatasetSummary { pub error_sources: usize, } -#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize)] pub struct CatalogTree { #[serde(default)] pub browse: Option, @@ -846,7 +846,7 @@ mod tests { } /// Directory observations are independent of the query's pinned revisions. -#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize)] pub struct BrowseStatus { #[serde(default)] pub partial: bool, @@ -860,13 +860,23 @@ pub struct BrowseStatus { impl QueryDatasetSummary { pub fn label(&self) -> String { - let suffix = self.browse.as_ref().map(|status| { - if status.observed_at == 0 && status.last_error.is_some() { " · Unavailable" } - else if status.observed_at == 0 { " · Loading…" } - else if status.partial { " · Partial" } - else if status.stale { " · Cached" } - else { "" } - }).unwrap_or(""); + let suffix = self + .browse + .as_ref() + .map(|status| { + if status.observed_at == 0 && status.last_error.is_some() { + " · Unavailable" + } else if status.observed_at == 0 { + " · Loading…" + } else if status.partial { + " · Partial" + } else if status.stale { + " · Cached" + } else { + "" + } + }) + .unwrap_or(""); format!("{}{suffix}", self.name) } } @@ -879,7 +889,8 @@ mod browse_status_tests { let mut dataset: QueryDatasetSummary = serde_json::from_value(serde_json::json!({ "name": "mount", "uri": "s3://bucket", "ready_sources": 0, "error_sources": 0, "browse": {"observed_at": 0, "refreshing": true, "stale": true, "last_error": null} - })).unwrap(); + })) + .unwrap(); assert_eq!(dataset.label(), "mount · Loading…"); dataset.browse.as_mut().unwrap().observed_at = 1; dataset.browse.as_mut().unwrap().stale = false; diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index fd01ec3e5..5b80151e2 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -28,9 +28,9 @@ use crate::llm_settings::LlmSettings; #[cfg(test)] use crate::model::RunSearchStatus; use crate::model::{ - CatalogTree, CompactRecordDetail, DimensionAggregate, HistogramBucket, PageSnapshot, - QueryCatalog, QueryDatasetSummary, RunAnalysis, RunExplorerItem, RunPage, RunSummary, - ToolAggregate, TurnDetail, TurnSearchStatus, TurnSummary, + BrowseStatus, CatalogTree, CompactRecordDetail, DimensionAggregate, HistogramBucket, + PageSnapshot, QueryCatalog, QueryDatasetSummary, RunAnalysis, RunExplorerItem, RunPage, + RunSummary, ToolAggregate, TurnDetail, TurnSearchStatus, TurnSummary, }; use crate::notice::{ErrorNotice, WorkspaceNotice, workspace_notice}; use crate::terminology::{ANALYSIS, ASSISTANT, DATASETS, RUNS, STEPS, STORAGE, TIMELINE}; @@ -281,8 +281,9 @@ pub fn App() -> Element { let last_place = use_signal(String::new); let history_ready = use_signal(|| false); let mut history_seq = use_signal(|| 0i32); - let catalog_tree = use_signal(|| None::); - let catalog_loading = use_signal(|| false); + let mut catalog_tree = use_signal(|| None::); + let mut catalog_loading = use_signal(|| false); + let mut catalog_generation = use_signal(|| 0u64); let mut offset = use_signal(|| 0usize); let mut error = use_signal(|| None::); let mut catalog_auth_configured = use_signal(|| catalog_auth::load().is_configured()); @@ -356,6 +357,8 @@ pub fn App() -> Element { catalog_tree, catalog_loading, error, + catalog_generation, + catalog_generation(), ); spawn(async move { loop { @@ -370,6 +373,8 @@ pub fn App() -> Element { catalog_tree, catalog_loading, error, + catalog_generation, + catalog_generation(), ); } }); @@ -493,11 +498,19 @@ pub fn App() -> Element { return; } let initial = catalog().is_none(); - let waiting = catalog().as_ref().is_some_and(|catalog| catalog.datasets.iter() - .any(|dataset| dataset.browse.as_ref().is_some_and(|status| status.observed_at == 0))); + let waiting = catalog().as_ref().is_some_and(|catalog| { + catalog.datasets.iter().any(|dataset| { + dataset + .browse + .as_ref() + .is_some_and(|status| status.observed_at == 0) + }) + }); if initial || waiting { spawn(async move { - if !initial { TimeoutFuture::new(CATALOG_REFRESH_MS).await; } + if !initial { + TimeoutFuture::new(CATALOG_REFRESH_MS).await; + } match api::query_catalog().await { Ok(value) => { if selected_table().is_empty() { @@ -583,6 +596,22 @@ pub fn App() -> Element { auth_required: !catalog_auth_configured() && catalog_tree().is_none(), on_settings: move |_| settings_open.set(true), on_open: move |(dataset, prefix): (String, String)| { + // Move immediately. The old tree must not remain visible while + // a slow remote prefix is loading. + catalog_tree.set(Some(CatalogTree { + dataset: (!dataset.is_empty()).then_some(dataset.clone()), + prefix: prefix.clone(), + browse: Some(BrowseStatus { + state: "refreshing".into(), + refreshing: true, + observed_at: 0, + stale: true, + ..Default::default() + }), + ..Default::default() + })); + catalog_loading.set(true); + catalog_generation.set(catalog_generation().saturating_add(1)); catalog_dataset.set(dataset); catalog_prefix.set(prefix); }, @@ -1122,24 +1151,34 @@ fn load_catalog_tree( mut tree: Signal>, mut loading: Signal, mut error: Signal>, + generation: Signal, + requested_generation: u64, ) { loading.set(true); spawn(async move { match api::explorer_tree(&dataset, &prefix).await { - Ok(value) => tree.set(Some(value)), + Ok(value) if generation() == requested_generation => tree.set(Some(value)), Err(failure) if matches!(failure.status, 400 | 401) && dataset.is_empty() && prefix.is_empty() => { match api::explorer_tree_anonymous(&dataset, &prefix).await { - Ok(value) => tree.set(Some(value)), - Err(failure) => error.set(Some(workspace_notice(&failure))), + Ok(value) if generation() == requested_generation => tree.set(Some(value)), + Err(failure) if generation() == requested_generation => { + error.set(Some(workspace_notice(&failure))) + } + _ => {} } } - Err(failure) => error.set(Some(workspace_notice(&failure))), + Err(failure) if generation() == requested_generation => { + error.set(Some(workspace_notice(&failure))) + } + _ => {} + } + if generation() == requested_generation { + loading.set(false); } - loading.set(false); }); } From 5fab76249d54ac7ab469f29ea9a7390cd17f66b5 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 06:29:17 +0800 Subject: [PATCH 11/22] fix(workspace): prevent request loop in load_runs function - Updated the `load_runs` function to use `generation.peek()` instead of directly reading the signal, avoiding potential subscription to its own writes and preventing infinite request loops. - This change enhances the stability of the reactive effect by ensuring proper management of request generations. This commit aims to improve the reliability of run data loading in the workspace component. --- pchronicle-web/src/workspace.rs | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 5b80151e2..f0e54a68e 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -963,7 +963,10 @@ fn load_runs( mut generation: Signal, mut error: Signal>, ) { - let request_generation = generation() + 1; + // This function is called from a reactive effect. Reading the signal here + // would subscribe that effect to its own generation writes and create a + // request loop. + let request_generation = *generation.peek() + 1; generation.set(request_generation); page.set(None); loading.set(true); From c87c983c44fc9f62b83cec84a4827ac006cb405f Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 15:52:25 +0800 Subject: [PATCH 12/22] feat(store): enhance object store configuration and background I/O management - Introduced `StoreConfig` struct to encapsulate S3 configuration parameters, improving clarity and usability. - Added `with_background_object_store_io` function to manage background I/O operations independently from foreground tasks, enhancing concurrency control. - Updated `DatasetLocation` and `DatasetMount` to support backend configuration, allowing for more flexible dataset management. - Refactored object store interaction methods to utilize the new configuration, ensuring consistent behavior across different storage backends. This commit aims to improve the configurability and performance of object store interactions in the pChronicle application. --- .../src/server/catalog.rs | 169 +++-- .../src/server/mod.rs | 162 +++-- .../src/server/tests.rs | 30 + .../src/server/ui_cache.rs | 625 ++++++++++++------ crates/persisting-pchronicle/src/storage.rs | 4 +- .../src/store/catalog/identity.rs | 13 + .../src/store/catalog/location.rs | 68 +- .../src/store/catalog/manifest_cache.rs | 108 ++- .../src/store/object_store_io_gate.rs | 50 +- .../src/store/opendal_store.rs | 120 ++++ ...pchronicle-catalog-resolution-and-cache.md | 38 +- ...pchronicle-catalog-resolution-and-cache.md | 38 +- pchronicle-web/src/catalog.rs | 10 +- pchronicle-web/src/workspace.rs | 74 +-- 14 files changed, 1079 insertions(+), 430 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/catalog.rs b/crates/persisting-pchronicle-cli/src/server/catalog.rs index 7a94ee97b..3b9db94f3 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog.rs @@ -141,16 +141,6 @@ impl CatalogAcl { Some(user) } - pub(crate) fn apply_public_backend_env(&self) { - if let Some(library) = self.public_datasets.iter().find_map(|name| { - self.libraries - .get(name) - .filter(|library| library.uri.starts_with("s3://")) - }) { - apply_library_env(library); - } - } - fn authenticate_headers( &self, headers: &axum::http::HeaderMap, @@ -169,6 +159,32 @@ impl CatalogAcl { .collect() } + pub(crate) fn libraries_for_public(&self) -> Vec { + self.public_datasets + .iter() + .filter_map(|name| self.libraries.get(name)) + .cloned() + .collect() + } + + fn visible_for_headers( + &self, + headers: &axum::http::HeaderMap, + ) -> Result, ApiError> { + match credentials_from_headers(headers) { + Some((access, secret)) => self + .authenticate(&access, &secret) + .map(|user| self.list_for(user)) + .ok_or_else(catalog_unauthorized), + None if !headers.contains_key(ACCESS_KEY_HEADER) + && !headers.contains_key(SECRET_KEY_HEADER) => + { + Ok(self.public_for_all()) + } + None => Err(catalog_unauthorized()), + } + } + fn credentials_for_public(&self, dataset: &str) -> Option<(&str, &str)> { self.users_by_access_key .iter() @@ -829,16 +845,7 @@ pub(super) async fn list_datasets( .catalog_acl .as_ref() .ok_or_else(|| ApiError::not_found("catalog is not enabled"))?; - let has_credential_headers = - headers.contains_key(ACCESS_KEY_HEADER) || headers.contains_key(SECRET_KEY_HEADER); - let libraries = match credentials_from_headers(&headers) { - Some((access, secret)) => acl - .authenticate(&access, &secret) - .map(|user| acl.list_for(user)) - .ok_or_else(catalog_unauthorized)?, - None if !has_credential_headers => acl.public_for_all(), - None => return Err(catalog_unauthorized()), - }; + let libraries = acl.visible_for_headers(&headers)?; Ok(axum::Json(libraries)) } @@ -872,20 +879,35 @@ pub(super) async fn catalog_data_plane_layer( if !path.starts_with("/api/") || parent_handles_path(&path) { return next.run(request).await; } - // Anonymous browsing is limited to wildcard-granted datasets. - if credentials_from_headers(request.headers()).is_none() - && path.ends_with("/query/tables") + // UI metadata is ACL-derived, not an all-dataset storage query. In + // particular, authenticated callers may have incompatible S3 backends. + let api_path = path + .strip_prefix("/api/v1") + .or_else(|| path.strip_prefix("/api")) + .unwrap_or(&path); + let ui_tables = api_path == "/query/tables" && url::form_urlencoded::parse(request.uri().query().unwrap_or("").as_bytes()) - .any(|(key, value)| key == "ui" && (value == "true" || value == "1")) - { - if let Some(library) = state.catalog_acl.as_ref().and_then(|acl| { - acl.libraries.values().find(|library| { - acl.public_for_all() - .iter() - .any(|public| public.name == library.name) - }) - }) { - apply_library_env(library); + .any(|(key, value)| key == "ui" && (value == "true" || value == "1")); + let root_tree = api_path == "/explorer/tree" + && !url::form_urlencoded::parse(request.uri().query().unwrap_or("").as_bytes()) + .any(|(key, value)| (key == "dataset" || key == "prefix") && !value.is_empty()); + if request.method() == axum::http::Method::GET && (ui_tables || root_tree) { + let libraries = match state + .catalog_acl + .as_ref() + .unwrap() + .visible_for_headers(request.headers()) + { + Ok(libraries) => libraries, + Err(error) => return error.into_response(), + }; + if root_tree { + let mounts = libraries + .iter() + .filter_map(|library| DatasetMount::new(&library.name, &library.uri).ok()) + .collect::>(); + return axum::Json(super::explorer::catalog_tree_from_mount_specs(&mounts)) + .into_response(); } let catalog = super::QueryCatalog { snapshot_id: String::new(), @@ -893,7 +915,16 @@ pub(super) async fn catalog_data_plane_layer( database: String::new(), storage_path: String::new(), path_column: "_file_", - datasets: Vec::new(), + datasets: libraries + .into_iter() + .map(|library| super::QueryDatasetSummary { + browse: None, + name: library.name, + uri: library.uri, + ready_sources: 0, + error_sources: 0, + }) + .collect(), tables: super::query_table_summaries(), }; return axum::Json(catalog).into_response(); @@ -926,13 +957,6 @@ pub(super) async fn catalog_data_plane_layer( .unwrap_or_else(|error| error.into_response()); } if !dataset.is_empty() { - if let Some(library) = state.catalog_acl.as_ref().and_then(|acl| { - acl.libraries - .values() - .find(|library| library.name == dataset) - }) { - apply_library_env(library); - } if let Some((access_key, secret_key)) = state .catalog_acl .as_ref() @@ -1488,6 +1512,69 @@ uri = "{}" builder.body(axum::body::Body::empty()).unwrap() } + #[tokio::test] + async fn ui_metadata_handles_mixed_backends_without_mounting_workers() { + use tower::ServiceExt; + let acl = + CatalogAcl::parse(&SAMPLE.replacen("127.0.0.1:9000", "127.0.0.1:9001", 1)).unwrap(); + assert!( + super::super::catalog_worker::validate_backends( + &acl.libraries.values().cloned().collect::>() + ) + .is_err() + ); + let mut state = super::super::app_state(super::super::ChronicleServerConfig::front_only()); + state.catalog_acl = Some(std::sync::Arc::new(acl)); + let warehouse = super::super::PreparedWarehouse { state }; + let app = warehouse.router(); + for path in [ + "/api/query/tables?ui=true", + "/api/v1/query/tables?ui=1", + "/api/explorer/tree?dataset=&prefix=", + "/api/v1/explorer/tree", + ] { + for (access, secret, expected) in [ + (Some("USER_AK"), Some("USER_SK"), 2), + (Some("BOB_AK"), Some("BOB_SK"), 1), + (None, None, 0), + ] { + let (status, body) = catalog_body( + app.clone() + .oneshot(catalog_request(path, access, secret)) + .await + .unwrap(), + ) + .await; + assert_eq!(status, axum::http::StatusCode::OK, "{path}: {body}"); + let json: serde_json::Value = serde_json::from_str(&body).unwrap(); + let entries = json[if path.contains("query/tables") { + "datasets" + } else { + "children" + }] + .as_array() + .unwrap(); + assert_eq!(entries.len(), expected, "{body}"); + if expected == 1 { + assert_eq!(entries[0]["name"], "evals"); + } + assert!(!body.contains("BACKEND_AK") && !body.contains("BACKEND_SK")); + } + for secret in [Some("wrong"), None] { + let (status, _) = catalog_body( + app.clone() + .oneshot(catalog_request(path, Some("USER_AK"), secret)) + .await + .unwrap(), + ) + .await; + assert_eq!(status, axum::http::StatusCode::UNAUTHORIZED); + } + } + assert!(warehouse.state.browse.get().is_none()); + assert!(warehouse.state.catalog.read().await.is_none()); + } + #[tokio::test] async fn catalog_list_allows_anonymous_public_datasets_and_rejects_invalid_credentials() { use tower::ServiceExt; diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 26bff53fe..2ac7b19f1 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -279,13 +279,21 @@ impl PreparedWarehouse { acl: catalog::CatalogAcl, mut config: ChronicleServerConfig, ) -> anyhow::Result { - // The browse worker starts before any HTTP request can select a - // dataset. Seed its shared OpenDAL S3 configuration from the catalog. - acl.apply_public_backend_env(); let browse_mounts = acl - .public_for_all() + .libraries_for_public() .into_iter() - .filter_map(|library| DatasetMount::new(&library.name, &library.uri).ok()) + .filter_map(|library| { + DatasetMount::new(&library.name, &library.uri) + .ok() + .map(|m| { + m.with_backend(persisting_pchronicle::storage::StoreConfig { + endpoint: library.endpoint.clone(), + region: library.region.clone(), + access_key: library.access_key.clone(), + secret_key: library.secret_key.clone(), + }) + }) + }) .collect::>(); config.datasets.clear(); config.default_dataset = None; @@ -583,6 +591,28 @@ async fn current_catalog( Ok(runtime) } +// An exact file request must not wait for discovery of unrelated sources. +async fn catalog_for_source( + state: &AppState, + dataset: Option<&str>, + file: &str, + request_id: &RequestId, +) -> Result, ApiError> { + let Some(dataset) = dataset else { + return current_catalog(state, request_id).await; + }; + build_scoped_query_runtime( + &state.config, + persisting_pchronicle::storage::QueryScope { + dataset: dataset.to_owned(), + source_file: Some(file.to_owned()), + }, + Vec::new(), + ) + .await + .map_err(|error| fail(request_id, "catalog_for_source", error)) +} + async fn current_catalog_for_runs( state: &AppState, request_id: &RequestId, @@ -886,7 +916,7 @@ async fn try_compact_jsonl_runs_page( .map(str::trim) .filter(|value| !value.is_empty() && *value != "all"); - let runtime = current_catalog(state, request_id).await?; + let runtime = catalog_for_source(state, dataset_filter, file_filter, request_id).await?; let mut matched: Option<(String, String, Option)> = None; for dataset in runtime.snapshot.datasets() { if dataset_filter.is_some_and(|filter| dataset.mount.name != filter) { @@ -989,6 +1019,44 @@ async fn try_compact_jsonl_runs_page( })) } +async fn open_storyline_source( + state: &AppState, + dataset: &str, + file: &str, + request_id: &RequestId, +) -> Result, ApiError> { + let Some(mount) = state + .config + .datasets + .iter() + .find(|mount| mount.name == dataset) + else { + return Ok(None); + }; + let location = persisting_pchronicle::storage::DatasetLocation::parse_with_backend( + &mount.uri, + mount.backend().cloned(), + ) + .map_err(|error| fail(request_id, "storyline_source", error))?; + if location + .probe_nav_dataset_kind(file) + .await + .map_err(|error| fail(request_id, "storyline_source", error))? + != Some("storyline") + { + return Ok(None); + } + let uri = format!( + "{}/{}", + mount.uri.trim_end_matches('/'), + file.trim_matches('/') + ); + persisting_pchronicle::storage::StorylineLanceStore::open_uri(&uri) + .await + .map(Some) + .map_err(|error| fail(request_id, "storyline_source", error)) +} + /// Directory mounts only expose immediate children in the catalog. Nested /// Storyline leaves reached via explorer navigation are therefore absent from /// SQL acceleration. When the client asks for an exact `file=` that is a @@ -1022,42 +1090,9 @@ async fn try_on_demand_storyline_runs_page( return Ok(None); }; - let runtime = current_catalog(state, request_id).await?; - let Some(dataset) = runtime.snapshot.dataset(dataset_name) else { + let Some(store) = open_storyline_source(state, dataset_name, file, request_id).await? else { return Ok(None); }; - // Prefer catalog-backed sources; only fall through for nested Directory paths. - if dataset.sources.iter().any(|source| { - source.kind != persisting_pchronicle::storage::CatalogSourceKind::Directory - && (source.file == file || source.file.starts_with(&format!("{file}/"))) - }) { - return Ok(None); - } - let under_directory = dataset.sources.iter().any(|source| { - source.kind == persisting_pchronicle::storage::CatalogSourceKind::Directory - && (file == source.file || file.starts_with(&format!("{}/", source.file))) - }); - if !under_directory && !dataset.sources.is_empty() { - return Ok(None); - } - - let location = persisting_pchronicle::storage::DatasetLocation::parse(&dataset.mount.uri) - .map_err(|error| fail(request_id, "explorer_runs", error))?; - let kind = location - .probe_nav_dataset_kind(file) - .await - .map_err(|error| fail(request_id, "explorer_runs", error))?; - if kind != Some("storyline") { - return Ok(None); - } - let uri = format!( - "{}/{}", - dataset.mount.uri.trim_end_matches('/'), - file.trim_matches('/') - ); - let store = persisting_pchronicle::storage::StorylineLanceStore::open_uri(&uri) - .await - .map_err(|error| fail(request_id, "explorer_runs", error))?; let Some((_generation, ids)) = store .document_ids_snapshot() .await @@ -1160,10 +1195,10 @@ async fn explorer_runs_inner( query: Result, QueryRejection>, ) -> Result, ApiError> { let query = api_query(query)?; - if let Some(page) = try_compact_jsonl_runs_page(&state, &query, &request_id).await? { + if let Some(page) = try_on_demand_storyline_runs_page(&state, &query, &request_id).await? { return Ok(Json(page)); } - if let Some(page) = try_on_demand_storyline_runs_page(&state, &query, &request_id).await? { + if let Some(page) = try_compact_jsonl_runs_page(&state, &query, &request_id).await? { return Ok(Json(page)); } let dataset_filter = query @@ -1577,46 +1612,9 @@ async fn try_resolve_on_demand_storyline_run( return Ok(None); } - let runtime = current_catalog(state, request_id).await?; - let Some(dataset) = runtime.snapshot.dataset(dataset_name) else { + let Some(store) = open_storyline_source(state, dataset_name, file, request_id).await? else { return Ok(None); }; - let exact_source = dataset.sources.iter().any(|source| { - source.kind != persisting_pchronicle::storage::CatalogSourceKind::Directory - && source.file == file - }); - if dataset.sources.iter().any(|source| { - source.kind != persisting_pchronicle::storage::CatalogSourceKind::Directory - && source.file.starts_with(&format!("{file}/")) - }) { - return Ok(None); - } - let under_directory = dataset.sources.iter().any(|source| { - source.kind == persisting_pchronicle::storage::CatalogSourceKind::Directory - && (file == source.file || file.starts_with(&format!("{}/", source.file))) - }); - if !under_directory && !exact_source && !dataset.sources.is_empty() { - return Ok(None); - } - - let location = persisting_pchronicle::storage::DatasetLocation::parse(&dataset.mount.uri) - .map_err(|error| fail(request_id, "resolve_run", error))?; - if location - .probe_nav_dataset_kind(file) - .await - .map_err(|error| fail(request_id, "resolve_run", error))? - != Some("storyline") - { - return Ok(None); - } - let uri = format!( - "{}/{}", - dataset.mount.uri.trim_end_matches('/'), - file.trim_matches('/') - ); - let store = persisting_pchronicle::storage::StorylineLanceStore::open_uri(&uri) - .await - .map_err(|error| fail(request_id, "resolve_run", error))?; let Some((_generation, ids)) = store .document_ids_snapshot() .await @@ -1655,7 +1653,7 @@ async fn load_on_demand_storyline_bundle( request_id: &RequestId, op: &'static str, ) -> Result, ApiError> { - let runtime = current_catalog(state, request_id).await?; + let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; let Some(dataset) = runtime.snapshot.dataset(&run.dataset) else { return Ok(None); }; @@ -1712,7 +1710,7 @@ async fn catalog_or_on_demand_trajectory_bundle( request_id: &RequestId, op: &'static str, ) -> Result { - let runtime = current_catalog(state, request_id).await?; + let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; let key = catalog_storyline_key(run); let catalog_result = if state.live_reads { runtime.snapshot.load_live_trajectory_bundle(&key).await diff --git a/crates/persisting-pchronicle-cli/src/server/tests.rs b/crates/persisting-pchronicle-cli/src/server/tests.rs index d56681ced..1b0c9641e 100644 --- a/crates/persisting-pchronicle-cli/src/server/tests.rs +++ b/crates/persisting-pchronicle-cli/src/server/tests.rs @@ -2640,6 +2640,36 @@ async fn physical_api_inspects_storyline_lance_layout_file_and_page() { ); } +#[tokio::test] +async fn exact_runs_request_does_not_wait_for_global_catalog() -> anyhow::Result<()> { + let root = tempfile::tempdir()?; + let store = + persisting_pchronicle::storage::StorylineLanceStore::open(root.path().join("nested/story")) + .await?; + store + .replace_storyline(&storyline_document("session-a", "run-a")) + .await?; + let state = app_state(ChronicleServerConfig::mounted(vec![DatasetMount::new( + "prod2", + root.path().to_string_lossy(), + )?])?); + // A slow unrelated catalog refresh must not block an exact source request. + let _refresh = state.catalog_refresh.lock().await; + let app = finish_routes(state.clone()); + let (status, page) = tokio::time::timeout( + Duration::from_secs(5), + get_json( + &app, + "/api/explorer/runs?dataset=prod2&file=nested/story&limit=50", + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{page}"); + assert_eq!(page["snapshot"]["total"], 1, "{page}"); + assert!(state.catalog.read().await.is_none()); + Ok(()) +} + #[tokio::test] async fn browse_tree_does_not_build_query_runtime() -> anyhow::Result<()> { use http_body_util::BodyExt; diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index d60893452..fa5e1991b 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -8,7 +8,8 @@ use std::time::{Duration, Instant}; use anyhow::{Context, Result}; use persisting_pchronicle::storage::{ - CatalogConsistency, CatalogState, DatasetLocation, DatasetMount, ManifestCache, + CatalogConsistency, CatalogState, DatasetLocation, DatasetMount, ManifestCache, PathListKind, + with_background_object_store_io, }; use serde::{Deserialize, Serialize}; use tokio::sync::{RwLock, mpsc, oneshot}; @@ -16,7 +17,7 @@ use tokio::sync::{RwLock, mpsc, oneshot}; use super::explorer::{CatalogTree, catalog_tree_from_mount_specs, catalog_tree_from_path_list}; const REFRESH_INTERVAL: Duration = Duration::from_secs(30); -const FOREGROUND_REFRESH_TIMEOUT: Duration = Duration::from_secs(5); +const FOREGROUND_REFRESH_TIMEOUT: Duration = Duration::from_millis(250); const QUEUE_CAPACITY: usize = 128; // Keep browse work bounded while allowing a foreground request to run beside // one background walk. @@ -48,7 +49,7 @@ impl TreeKey { } fn mount_fingerprint(mount: &DatasetMount) -> String { - let location = DatasetLocation::parse(&mount.uri).ok(); + let location = DatasetLocation::parse_with_backend(&mount.uri, mount.backend().cloned()).ok(); let identity = location .as_ref() .and_then(|l| l.local_path()) @@ -61,16 +62,28 @@ fn mount_fingerprint(mount: &DatasetMount) -> String { .unwrap_or_else(|| mount.uri.trim_end_matches('/').to_owned()); // S3-compatible endpoints can expose different data under the same URI. let endpoint = if mount.uri.starts_with("s3://") { - std::env::var("AWS_ENDPOINT_URL_S3") - .or_else(|_| std::env::var("AWS_ENDPOINT_URL")) - .or_else(|_| std::env::var("AWS_ENDPOINT")) - .unwrap_or_default() + match mount.backend() { + Some(config) => config.endpoint.clone().unwrap_or_default(), + None => std::env::var("AWS_ENDPOINT_URL_S3") + .or_else(|_| std::env::var("AWS_ENDPOINT")) + .or_else(|_| std::env::var("AWS_ENDPOINT_URL")) + .unwrap_or_default(), + } } else { String::new() }; - blake3::hash(&serde_json::to_vec(&(identity, endpoint)).unwrap()) - .to_hex() - .to_string() + blake3::hash( + &serde_json::to_vec(&( + identity, + endpoint, + mount + .backend() + .map(|b| (&b.region, &b.access_key, &b.secret_key)), + )) + .unwrap(), + ) + .to_hex() + .to_string() } #[derive(Clone, Debug, Serialize, Deserialize)] @@ -219,7 +232,8 @@ impl BrowseCoordinator { let mut error = None; for mount in mounts { let key = TreeKey::new(mount, "").expect("root prefix"); - let _ = self.enqueue(&key, None); + // Mount names are local configuration; the root page must not + // promote every mount into foreground remote work. refreshing |= lock_recover(&self.pending).contains_key(&key); if let Some(failure) = lock_recover(&self.refresh) .get(&key) @@ -241,7 +255,11 @@ impl BrowseCoordinator { .manifests .summary(&format!("{}\0{}", mount.name, mount_fingerprint(mount))) .await; - partial |= summary.partial; + partial |= summary.partial + || !self + .manifests + .is_complete_under(&manifest_key(&key, Some("")), "") + .await; if let Some(child) = tree.children.iter_mut().find(|c| c.name == mount.name) { child.run_count = summary.trajectories as usize; child.dataset_count = Some(summary.datasets as usize); @@ -299,44 +317,62 @@ impl BrowseCoordinator { "browse tree request" ); let existing = self.index.values.read().await.get(&key).cloned(); + let existing = match existing { + Some(entry) => Some(entry), + None => self + .manifests + .get(&manifest_key(&key, None)) + .await + .map(|listing| { + let tree = + catalog_tree_from_path_list(&key.dataset, &key.prefix, &listing.entries); + IndexEntry { + generation: blake3::hash(&serde_json::to_vec(&tree).unwrap()) + .to_hex() + .to_string(), + tree, + observed_at: listing.observed_at, + } + }), + }; if let Some(entry) = existing { - let _ = self.enqueue(&key, None); + if now() - entry.observed_at >= REFRESH_INTERVAL.as_secs() as i64 { + let _ = self.enqueue(&key, None); + } return Ok(self.snapshot_with_summary(&key, entry).await); } // A cold page is user-visible work: attach to the same single-flight // refresh as the background walker and return once this prefix exists. let (reply, wait) = oneshot::channel(); self.enqueue(&key, Some(reply))?; - match tokio::time::timeout(FOREGROUND_REFRESH_TIMEOUT, wait).await { - Ok(result) => result - .context("browse refresh task stopped")? - .map_err(anyhow::Error::msg)?, - Err(_) => { - let tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]); - return Ok(BrowseSnapshot { - tree, - browse: BrowseStatus { - partial: true, - consistency: CatalogConsistency::BestEffort, - state: CatalogState::Refreshing, - generation: String::new(), - observed_at: 0, - stale: true, - refreshing: true, - last_error: Some("remote browse is still loading".into()), - }, - }); - } + // enqueue may decline during cooldown; it drops the sender in that + // case. A closed reply is not a new internal error or an empty directory. + let _ = tokio::time::timeout(FOREGROUND_REFRESH_TIMEOUT, wait).await; + let existing = self.index.values.read().await.get(&key).cloned(); + if let Some(entry) = existing { + return Ok(self.snapshot_with_summary(&key, entry).await); } - let entry = self - .index - .values - .read() - .await + let refreshing = lock_recover(&self.pending).contains_key(&key); + let last_error = lock_recover(&self.refresh) .get(&key) - .cloned() - .context("browse refresh completed without a cached view")?; - Ok(self.snapshot_with_summary(&key, entry).await) + .and_then(|s| s.error.clone()); + Ok(BrowseSnapshot { + tree: catalog_tree_from_path_list(&key.dataset, &key.prefix, &[]), + browse: BrowseStatus { + partial: true, + consistency: CatalogConsistency::BestEffort, + state: if refreshing { + CatalogState::Refreshing + } else { + CatalogState::Unavailable + }, + generation: String::new(), + observed_at: 0, + stale: false, + refreshing, + last_error, + }, + }) } fn enqueue(&self, key: &TreeKey, reply: Option) -> Result<()> { @@ -345,9 +381,6 @@ impl BrowseCoordinator { .retry_at .is_some_and(|deadline| deadline > Instant::now()) { - if let Some(error) = &state.error { - anyhow::bail!("{error}"); - } return Ok(()); } let mut pending = lock_recover(&self.pending); @@ -396,116 +429,58 @@ impl BrowseCoordinator { async fn snapshot_with_summary(&self, key: &TreeKey, mut entry: IndexEntry) -> BrowseSnapshot { let current_manifest_key = manifest_key(key, None); - let summary = self.manifests.summary_under(¤t_manifest_key).await; - entry.tree.dataset_count = Some(summary.datasets as usize); - entry.tree.trajectory_count = Some(summary.trajectories as usize); - entry.tree.run_count = summary.trajectories as usize; - let cached_trees = self - .index - .values - .read() - .await - .values() - .map(|entry| entry.tree.clone()) - .collect::>(); - let (cached_datasets, cached_trajectories) = cached_leaf_summary(&cached_trees, ""); - if cached_datasets > 0 { - entry.tree.dataset_count = Some(cached_datasets); - entry.tree.trajectory_count = Some(cached_trajectories); - entry.tree.run_count = cached_trajectories; + let observation = self.manifests.get(¤t_manifest_key).await; + // The directory projection supplies names; only local manifest + // observations supply dataset identity, formats and record counts. + if let Some(listing) = &observation + && listing.entries.iter().any(|leaf| { + leaf.kind == PathListKind::Dataset + && (leaf.path == key.prefix || (key.prefix.is_empty() && leaf.path == ".")) + }) + { + entry.tree = catalog_tree_from_path_list(&key.dataset, &key.prefix, &listing.entries); } for child in &mut entry.tree.children { if child.kind != "dir" { continue; } - let child_prefix = if key.prefix.is_empty() - || child.path == key.prefix - || child.path.starts_with(&format!("{}/", key.prefix)) - { - child.path.clone() + let child_prefix = child.path.clone(); + let child_manifest_key = manifest_key(key, Some(&child_prefix)); + let listing = self.manifests.get(&child_manifest_key).await; + let leaf = observation + .iter() + .flat_map(|listing| &listing.entries) + .chain(listing.iter().flat_map(|listing| &listing.entries)) + .find(|leaf| leaf.kind == PathListKind::Dataset && leaf.path == child_prefix); + if let Some(leaf) = leaf { + let name = child.name.clone(); + *child = catalog_tree_from_path_list( + &key.dataset, + &key.prefix, + std::slice::from_ref(leaf), + ) + .children + .remove(0); + child.name = name; + child.dataset_count = Some(1); + child.trajectory_count = Some(child.run_count); } else { - format!("{}/{}", key.prefix, child.path) - }; - let summary = self - .manifests - .summary_under(&manifest_key(key, Some(&child_prefix))) - .await; - child.dataset_count = (summary.datasets > 0).then_some(summary.datasets as usize); - child.trajectory_count = - (summary.trajectories > 0).then_some(summary.trajectories as usize); - let child_key = TreeKey { - dataset: key.dataset.clone(), - uri_fingerprint: key.uri_fingerprint.clone(), - prefix: child_prefix.trim_matches('/').to_owned(), - }; - // Child discovery belongs to the bounded background walk. Rendering - // a wide directory must not promote every child to foreground work. - let cached = self.index.values.read().await.get(&child_key).cloned(); - tracing::info!( - target: "pchronicle.serve", - dataset = %key.dataset, - parent_prefix = %key.prefix, - child_prefix = %child.path, - manifest_datasets = summary.datasets, - manifest_trajectories = summary.trajectories, - projection_hit = cached.is_some(), - "catalog directory summary" - ); - if let Some(cached) = cached { - // Remote parents initially contain names only. Reuse the child's - // own leaf observation without another request to S3. - if let Some(leaf) = cached - .tree - .children - .iter() - .find(|leaf| leaf.kind == "file" && leaf.path == child_prefix) - { - let name = child.name.clone(); - *child = leaf.clone(); - child.name = name; - continue; - } - if let Some(count) = cached.tree.dataset_count.filter(|count| *count > 0) { - child.dataset_count = Some(count); - } - if let Some(count) = cached - .tree - .trajectory_count - .or(Some(cached.tree.run_count)) - .filter(|count| *count > 0) - { - child.trajectory_count = Some(count); - } - } - let (datasets, trajectories) = cached_leaf_summary(&cached_trees, &child_prefix); - if datasets > 0 { - child.dataset_count = Some(datasets); - child.trajectory_count = Some(trajectories); + let summary = self.manifests.summary_under(&child_manifest_key).await; + child.dataset_count = Some(summary.datasets as usize); + child.trajectory_count = Some(summary.trajectories as usize); } } - let child_summary = - entry - .tree - .children - .iter() - .fold((0usize, 0usize), |(datasets, trajectories), child| { - let datasets = datasets.saturating_add( - child - .dataset_count - .unwrap_or_else(|| (child.kind == "dataset") as usize), - ); - let trajectories = trajectories - .saturating_add(child.trajectory_count.unwrap_or(child.run_count)); - (datasets, trajectories) - }); - entry.tree.dataset_count = Some((summary.datasets as usize).max(child_summary.0)); - entry.tree.trajectory_count = Some((summary.trajectories as usize).max(child_summary.1)); - entry.tree.run_count = entry.tree.trajectory_count.unwrap_or_default(); - let partial = - summary.partial || projection_is_partial(&*self.index.values.read().await, key); + let summary = self.manifests.summary_under(¤t_manifest_key).await; + entry.tree.dataset_count = Some(summary.datasets as usize); + entry.tree.trajectory_count = Some(summary.trajectories as usize); + entry.tree.run_count = summary.trajectories as usize; + let partial = !self + .manifests + .is_complete_under(&manifest_key(key, Some("")), &key.prefix) + .await; let mut snapshot = self.snapshot(key, entry); snapshot.browse.partial = partial; - if partial && !snapshot.browse.refreshing { + if partial && snapshot.browse.state == CatalogState::Ready { snapshot.browse.state = CatalogState::Partial; } snapshot @@ -548,24 +523,6 @@ fn manifest_key(key: &TreeKey, child_prefix: Option<&str>) -> String { } } -fn cached_leaf_summary(trees: &[CatalogTree], prefix: &str) -> (usize, usize) { - let mut leaves = HashMap::::new(); - for tree in trees { - for child in &tree.children { - if child.kind == "file" - && child.data_type != "other" - && (child.path == prefix || child.path.starts_with(&format!("{prefix}/"))) - { - leaves.insert(child.path.clone(), child.run_count); - } - } - } - ( - leaves.len(), - leaves.values().copied().fold(0usize, usize::saturating_add), - ) -} - async fn run_worker( mounts: Vec, index: Arc, @@ -580,6 +537,8 @@ async fn run_worker( let mut visited = HashSet::new(); let mut background_budget = 0usize; let mut background_running = false; + // Manifest failures/backoff never suppress an interactive directory list. + let background_states = Mutex::new(HashMap::::new()); // FIFO is intentional: children are appended only after their parent // completes, so the background walk is breadth-first (shallow to deep). // JoinSet aborts outstanding work when the coordinator is dropped. @@ -601,16 +560,23 @@ async fn run_worker( } }; if was_background { background_running = false; } + let target_states = if was_background { &background_states } else { states.as_ref() }; let outcome = match result { Ok(tree) => { - visited.insert(key.clone()); - for child in &tree.children { - if child.kind == "dir" && visited.len() < 10_000 { - let next = TreeKey { prefix: child.path.clone(), ..key.clone() }; - if visited.insert(next.clone()) { background.push_back(next); } + if was_background { + visited.insert(key.clone()); + for child in &tree.children { + if child.kind == "dir" && visited.len() < 10_000 { + let next = TreeKey { prefix: child.path.clone(), ..key.clone() }; + if visited.insert(next.clone()) { background.push_back(next); } + } } + } else if visited.len() < 10_000 && visited.insert(key.clone()) { + // An opened directory becomes a manifest observation; + // its descendants still enter the FIFO breadth-first. + background.push_back(key.clone()); } - lock_recover(&states).insert(key.clone(), RefreshState { + lock_recover(target_states).insert(key.clone(), RefreshState { retry_at: Some(Instant::now() + Duration::from_secs(2)), ..Default::default() }); @@ -619,18 +585,18 @@ async fn run_worker( Err(error) => { let error = format!("{error:#}"); let cached_view = index.values.read().await.contains_key(&key); - let mut states = lock_recover(&states); + let mut states = lock_recover(target_states); let state = states.entry(key.clone()).or_default(); state.failures = state.failures.saturating_add(1); state.retry_at = Some(Instant::now() + Duration::from_secs( (30u64 * (1u64 << state.failures.min(4))).min(300))); state.error = Some(error.clone()); tracing::warn!(target: "pchronicle.serve", dataset = %key.dataset, - prefix = %key.prefix, cached_view, %error, "browse refresh failed"); + prefix = %key.prefix, background = was_background, cached_view, %error, "browse refresh failed"); Err(error) } }; - finish(&pending, &key, outcome); + if !was_background { finish(&pending, &key, outcome); } continue; } request = receiver.recv(), if jobs.len() < 2 => match request { @@ -658,19 +624,23 @@ async fn run_worker( else { continue; }; - // A queued foreground request or an active job already owns this key. - // Leave its waiters attached; never start a duplicate background scan. - if is_background && lock_recover(&pending).contains_key(&key) { - continue; - } - if lock_recover(&states) + let target_states = if is_background { + &background_states + } else { + states.as_ref() + }; + if lock_recover(target_states) .get(&key) .is_some_and(|s| s.retry_at.is_some_and(|deadline| deadline > Instant::now())) { - finish(&pending, &key, Ok(())); + if !is_background { + finish(&pending, &key, Ok(())); + } continue; } - lock_recover(&pending).entry(key.clone()).or_default(); + if !is_background { + lock_recover(&pending).entry(key.clone()).or_default(); + } if is_background { background_running = true; } @@ -683,7 +653,20 @@ async fn run_worker( } else { None }; - let result = refresh_tree(&mount, &key, &index, &manifests).await; + let work = async { + tokio::time::timeout( + Duration::from_secs(10), + refresh_tree(&mount, &key, &index, &manifests, is_background), + ) + .await + .context("browse list timed out") + .and_then(|result| result) + }; + let result = if is_background { + with_background_object_store_io(work).await + } else { + work.await + }; (key, is_background, result) }); } @@ -694,6 +677,7 @@ async fn refresh_tree( key: &TreeKey, index: &BrowseTreeProjection, manifests: &ManifestCache, + background: bool, ) -> Result { let _permit = BROWSE_IO.acquire().await?; #[cfg(test)] @@ -704,26 +688,23 @@ async fn refresh_tree( } else { None }; - let location = DatasetLocation::parse(&mount.uri)?; + let location = DatasetLocation::parse_with_backend(&mount.uri, mount.backend().cloned())?; if let Some(root) = location.local_path() { tokio::fs::metadata(root.join(&key.prefix)) .await .context("browse path unavailable")?; } - let manifest_key = if key.prefix.is_empty() { - format!("{}\0{}", key.dataset, key.uri_fingerprint) + let entries = if background { + manifests + .refresh_for_browse(manifest_key(key, None), &location, &key.prefix) + .await? + .entries } else { - format!("{}\0{}\0{}", key.dataset, key.uri_fingerprint, key.prefix) + // Foreground LIST never waits for a background manifest refresh lock, + // probes a sidecar, or replaces the cached manifest observation. + location.list_directory(&key.prefix).await? }; - // Only inspect this prefix: child marker probes and statistics are deferred - // to the bounded background walk, so a wide directory can be cached promptly. - let listing = tokio::time::timeout( - Duration::from_secs(60), - manifests.refresh_for_browse(manifest_key, &location, &key.prefix), - ) - .await - .context("browse list timed out")??; - let tree = catalog_tree_from_path_list(&mount.name, &key.prefix, &listing.entries); + let tree = catalog_tree_from_path_list(&mount.name, &key.prefix, &entries); index.put(key.clone(), tree.clone()).await; Ok(tree) } @@ -847,6 +828,241 @@ mod tests { DatasetMount::new("test", path.to_string_lossy()).unwrap() } + #[tokio::test] + async fn failed_cold_browse_stays_unavailable_during_cooldown_then_recovers() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("not-yet-created"); + let mount = mount(&source); + let coordinator = + BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; + let key = TreeKey::new(&mount, "").unwrap(); + let (reply, wait) = oneshot::channel(); + coordinator.enqueue(&key, Some(reply)).unwrap(); + assert!(wait.await.unwrap().is_err()); + for _ in 0..3 { + let view = + tokio::time::timeout(Duration::from_millis(100), coordinator.tree(&mount, "")) + .await + .unwrap() + .unwrap(); + assert_eq!(view.browse.state, CatalogState::Unavailable); + assert!(view.browse.partial); + assert!(!view.browse.refreshing); + assert_eq!(view.browse.observed_at, 0); + assert!( + view.browse + .last_error + .as_deref() + .unwrap() + .contains("browse path unavailable") + ); + } + assert!(!lock_recover(&coordinator.pending).contains_key(&key)); + assert_eq!(lock_recover(&coordinator.refresh)[&key].failures, 1); + std::fs::create_dir_all(source.join("recovered")).unwrap(); + lock_recover(&coordinator.refresh).remove(&key); + let (reply, wait) = oneshot::channel(); + coordinator.enqueue(&key, Some(reply)).unwrap(); + wait.await.unwrap().unwrap(); + let view = coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(view.tree.children[0].name, "recovered"); + assert!(view.browse.last_error.is_none()); + assert!(view.browse.observed_at > 0); + } + + #[tokio::test] + async fn blocked_s3_endpoint_does_not_block_another_mount_with_the_same_bucket() { + use axum::{ + Router, + http::{Method, StatusCode}, + }; + use persisting_pchronicle::storage::StoreConfig; + let temp = tempfile::tempdir().unwrap(); + let blocked = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let healthy = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let blocked_endpoint = format!("http://{}", blocked.local_addr().unwrap()); + let healthy_endpoint = format!("http://{}", healthy.local_addr().unwrap()); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = started.clone(); + let mut servers = tokio::task::JoinSet::new(); + servers.spawn(async move { + axum::serve( + blocked, + Router::new().fallback(move || { + let notify = notify.clone(); + async move { + notify.notify_one(); + std::future::pending::().await + } + }), + ) + .await + .unwrap(); + }); + servers.spawn(async move { + axum::serve(healthy, Router::new().fallback(|method: Method| async move { + if method == Method::HEAD { + (StatusCode::NOT_FOUND, "") + } else { + (StatusCode::OK, r#"same-bucketfalsenested/"#) + } + })).await.unwrap(); + }); + let make_mount = |name: &str, endpoint: String| { + DatasetMount::new(name, "s3://same-bucket") + .unwrap() + .with_backend(StoreConfig { + endpoint: Some(endpoint), + region: Some("us-east-1".into()), + access_key: Some("test-key".into()), + secret_key: Some("test-secret".into()), + }) + }; + let blocked_mount = make_mount("prod", blocked_endpoint); + let healthy_mount = make_mount("prod2", healthy_endpoint); + let coordinator = BrowseCoordinator::start_at( + vec![blocked_mount.clone(), healthy_mount.clone()], + temp.path().join("cache"), + ) + .await; + tokio::time::timeout(Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let view = tokio::time::timeout(Duration::from_secs(3), async { + loop { + let view = coordinator.tree(&healthy_mount, "").await.unwrap(); + if view.browse.observed_at > 0 { + break view; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("healthy endpoint must not wait for blocked endpoint's retry budget"); + assert_eq!(view.tree.children[0].name, "nested"); + // A cold request returns a truthful loading view promptly, even while + // its HTTP request is hung. It must not create a second remote refresh. + let view = + tokio::time::timeout(Duration::from_secs(1), coordinator.tree(&blocked_mount, "")) + .await + .unwrap() + .unwrap(); + assert_eq!(view.browse.state, CatalogState::Refreshing); + assert!(view.browse.partial); + assert!( + lock_recover(&coordinator.pending) + .contains_key(&TreeKey::new(&blocked_mount, "").unwrap()) + ); + } + + #[tokio::test] + async fn foreground_lists_while_same_prefix_manifest_is_blocked_then_uses_cached_metadata() { + use axum::{ + Router, + http::{Method, StatusCode, Uri}, + }; + use persisting_pchronicle::storage::StoreConfig; + let temp = tempfile::tempdir().unwrap(); + let manifest_dir = temp.path().join("manifest"); + std::fs::create_dir_all(&manifest_dir).unwrap(); + persisting_pchronicle::storage::write_compact_jsonl_manifest(&manifest_dir, 1, 42).unwrap(); + let manifest = std::fs::read_to_string(manifest_dir.join("chronicle.manifest")).unwrap(); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let started = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let requests = Arc::new(Mutex::new(Vec::new())); + let (notify, resume, recorded) = (started.clone(), release.clone(), requests.clone()); + let mut servers = tokio::task::JoinSet::new(); + servers.spawn(async move { + axum::serve(listener, Router::new().fallback(move |method: Method, uri: Uri| { + let (notify, resume, recorded, manifest) = (notify.clone(), resume.clone(), recorded.clone(), manifest.clone()); + async move { + lock_recover(&recorded).push((method.clone(), uri.to_string())); + if uri.path() == "/bucket/chronicle.manifest" { + notify.notify_one(); + resume.notified().await; + } + if uri.path() == "/bucket/leaf/chronicle.manifest" { + return (StatusCode::OK, manifest); + } + if method == Method::HEAD { return (StatusCode::NOT_FOUND, String::new()); } + (StatusCode::OK, r#"bucketfalseleaf/"#.to_owned()) + } + })).await.unwrap(); + }); + let mount = DatasetMount::new("s3", "s3://bucket") + .unwrap() + .with_backend(StoreConfig { + endpoint: Some(endpoint), + region: Some("us-east-1".into()), + access_key: Some("key".into()), + secret_key: Some("secret".into()), + }); + let coordinator = + BrowseCoordinator::start_at(vec![mount.clone()], temp.path().join("cache")).await; + tokio::time::timeout(Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let view = tokio::time::timeout(Duration::from_secs(2), async { + loop { + let view = coordinator.tree(&mount, "").await.unwrap(); + if view.browse.observed_at > 0 { + break view; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("LIST must not join or wait for the blocked manifest job"); + assert_eq!(view.tree.children[0].kind, "dir"); + assert_eq!(view.tree.dataset_count, Some(0)); + assert!(view.browse.partial); + assert_eq!( + lock_recover(&requests) + .iter() + .filter(|(method, _)| method == Method::HEAD) + .count(), + 1, + "foreground must issue no marker probes" + ); + assert_eq!( + lock_recover(&requests).len(), + 2, + "only background HEAD and foreground LIST" + ); + // Even after the short request cooldown expires, UI polling should + // read the fresh local directory projection, not perform another LIST. + tokio::time::sleep(Duration::from_millis(2100)).await; + coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(lock_recover(&requests).len(), 2); + release.notify_one(); + let root_key = TreeKey::new(&mount, "").unwrap(); + tokio::time::timeout(Duration::from_secs(3), async { + while coordinator + .manifests + .get(&manifest_key(&root_key, Some("leaf"))) + .await + .is_none() + { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let view = coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(view.tree.children[0].kind, "file"); + assert_eq!(view.tree.children[0].data_type, "compact-jsonl"); + assert_eq!(view.tree.children[0].trajectory_count, Some(42)); + assert_eq!(view.tree.dataset_count, Some(1)); + assert_eq!(view.tree.trajectory_count, Some(42)); + assert!(!view.browse.partial); + // Local metadata remains useful after the remote endpoint disappears. + servers.abort_all(); + let view = coordinator.tree(&mount, "").await.unwrap(); + assert_eq!(view.tree.trajectory_count, Some(42)); + } + #[tokio::test] async fn cold_tree_returns_loading_and_incomplete_descendants_are_partial() { let temp = tempfile::tempdir().unwrap(); @@ -917,7 +1133,7 @@ mod tests { receiver, )); tokio::time::timeout(Duration::from_secs(5), async { - while !lock_recover(&pending).contains_key(&root_key) { + while BROWSE_IO.available_permits() != 1 { tokio::task::yield_now().await; } }) @@ -933,7 +1149,7 @@ mod tests { .unwrap() .unwrap(); assert!( - lock_recover(&pending).contains_key(&root_key), + BROWSE_IO.available_permits() == 1, "background must still be blocked" ); task.abort(); @@ -969,29 +1185,14 @@ mod tests { .manifests .refresh_mount( &manifest_key(&root_key, None), - &DatasetLocation::parse(&mount.uri).unwrap(), + &DatasetLocation::parse_with_backend(&mount.uri, mount.backend().cloned()).unwrap(), ) .await .unwrap(); std::fs::remove_dir_all(&source).unwrap(); assert!(browse.index.values.read().await.is_empty()); for (prefix, datasets) in [("", 4), ("nested", 2)] { - let key = TreeKey::new(&mount, prefix).unwrap(); - let listing = browse - .manifests - .get(&manifest_key(&key, None)) - .await - .unwrap(); - let view = browse - .snapshot_with_summary( - &key, - IndexEntry { - tree: catalog_tree_from_path_list(&mount.name, prefix, &listing.entries), - generation: String::new(), - observed_at: now(), - }, - ) - .await; + let view = browse.tree(&mount, prefix).await.unwrap(); let json = serde_json::to_value(view).unwrap(); assert_eq!(json["dataset_count"], datasets); assert_eq!(json["trajectory_count"], 708); diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 832a3b55f..78c34cdbd 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -1,6 +1,8 @@ //! pChronicle 的持久化存储入口。 pub type Result = anyhow::Result; +#[cfg(feature = "lance-store")] +pub use crate::store::opendal_store::StoreConfig; /// Parse an integer byte size with binary IEC suffixes. pub fn parse_byte_size(value: &str) -> std::result::Result { @@ -64,7 +66,7 @@ pub use crate::store::object_store_io_gate::{ IoKind as ObjectStoreIoKind, ObjectStoreGateSnapshot, ObjectStoreThrottleEvent, ObjectStoreThrottleHookGuard, format_aimd_flow_label as format_object_store_aimd_flow_label, install_throttle_hook as install_object_store_throttle_hook, - snapshot as object_store_gate_snapshot, + snapshot as object_store_gate_snapshot, with_background_object_store_io, }; #[cfg(feature = "lance-store")] diff --git a/crates/persisting-pchronicle/src/store/catalog/identity.rs b/crates/persisting-pchronicle/src/store/catalog/identity.rs index 1a5117342..a4f2466c9 100644 --- a/crates/persisting-pchronicle/src/store/catalog/identity.rs +++ b/crates/persisting-pchronicle/src/store/catalog/identity.rs @@ -2,6 +2,7 @@ use anyhow::Result; use serde::Serialize; use crate::DocumentFormat; +use crate::store::opendal_store::StoreConfig; use super::DEFAULT_DATASET_NAME; @@ -48,6 +49,8 @@ pub struct DatasetMount { pub namespace: NamespacePath, #[serde(skip)] pub(super) format_hint: Option, + #[serde(skip)] + pub(crate) backend: Option, } impl DatasetMount { @@ -73,9 +76,19 @@ impl DatasetMount { uri, namespace, format_hint: None, + backend: None, }) } + pub fn with_backend(mut self, backend: StoreConfig) -> Self { + self.backend = Some(backend); + self + } + + pub fn backend(&self) -> Option<&StoreConfig> { + self.backend.as_ref() + } + pub fn default(uri: impl Into) -> Result { Self::new(DEFAULT_DATASET_NAME, uri) } diff --git a/crates/persisting-pchronicle/src/store/catalog/location.rs b/crates/persisting-pchronicle/src/store/catalog/location.rs index 8fdfe9065..d93c6a952 100644 --- a/crates/persisting-pchronicle/src/store/catalog/location.rs +++ b/crates/persisting-pchronicle/src/store/catalog/location.rs @@ -9,7 +9,7 @@ use anyhow::{Context, Result, anyhow}; use futures::{StreamExt, TryStreamExt}; use url::Url; -use crate::store::opendal_store::Store as OpendalStore; +use crate::store::opendal_store::{Store as OpendalStore, StoreConfig}; /// One discovery event while walking importable JSON objects. #[derive(Debug, Clone)] @@ -74,10 +74,15 @@ pub struct DatasetLocation { uri: String, kind: DatasetLocationKind, local_path: Option, + backend: Option, } impl DatasetLocation { pub fn parse(input: &str) -> Result { + Self::parse_with_backend(input, None) + } + + pub fn parse_with_backend(input: &str, backend: Option) -> Result { let input = input.trim(); anyhow::ensure!(!input.is_empty(), "Dataset URI must not be empty"); if !input.contains("://") { @@ -121,6 +126,7 @@ impl DatasetLocation { uri: trim_trailing_slashes(input), kind: DatasetLocationKind::ObjectStore, local_path: None, + backend, }) } "file" => { @@ -135,6 +141,7 @@ impl DatasetLocation { uri: trim_trailing_slashes(input), kind: DatasetLocationKind::Local, local_path: Some(path), + backend: None, }) } "local" => { @@ -146,6 +153,7 @@ impl DatasetLocation { uri: trim_trailing_slashes(input), kind: DatasetLocationKind::Local, local_path: Some(PathBuf::from(url.path())), + backend: None, }) } other => Err(anyhow!("unsupported Dataset URI scheme '{other}'")), @@ -157,6 +165,7 @@ impl DatasetLocation { uri, kind: DatasetLocationKind::Local, local_path: Some(path), + backend: None, } } @@ -217,7 +226,9 @@ impl DatasetLocation { if let Some(path) = &self.local_path { return Ok(path.exists()); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; store.exists().await } @@ -240,7 +251,9 @@ impl DatasetLocation { } return put_local_bytes(&path, bytes, true); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; store .write_overwrite(relative, bytes.to_vec()) .await @@ -258,7 +271,9 @@ impl DatasetLocation { let path = root.join(relative); return std::fs::read(&path).with_context(|| format!("read {}", path.display())); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let Some((bytes, _)) = store.read(relative).await? else { return Err(anyhow!("object not found: {relative} under {}", self.uri)); }; @@ -305,7 +320,9 @@ impl DatasetLocation { return Ok(None); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let join = |name: &str| { if relative.is_empty() { name.to_string() @@ -353,22 +370,35 @@ impl DatasetLocation { /// directories, leaf Datasets (with sidecar preview when present), and /// JSON / JSONL / NDJSON files. pub async fn list(&self, relative: &str) -> Result> { - self.list_impl(relative, true).await + self.list_impl(relative, true, true).await } /// Browse one remote level without probing every child. A child is shown /// as a directory until its own observation identifies it as a Dataset. pub(crate) async fn list_for_browse(&self, relative: &str) -> Result> { - self.list_impl(relative, !self.is_object_store()).await + self.list_impl(relative, true, !self.is_object_store()) + .await + } + + /// List names for interactive browsing. Remote dataset identity and counts + /// come from the local manifest cache, never HEAD/GET probes on this path. + pub async fn list_directory(&self, relative: &str) -> Result> { + let local = !self.is_object_store(); + self.list_impl(relative, local, local).await } - async fn list_impl(&self, relative: &str, probe_children: bool) -> Result> { + async fn list_impl( + &self, + relative: &str, + probe_self: bool, + probe_children: bool, + ) -> Result> { let relative = relative.trim().trim_matches('/'); anyhow::ensure!( !relative.split('/').any(|part| part == ".."), "relative object path must not contain '..'" ); - if let Some(kind) = self.probe_nav_dataset_kind(relative).await? { + if probe_self && let Some(kind) = self.probe_nav_dataset_kind(relative).await? { let preview = self.list_dataset_preview(relative, kind).await?; let name = if relative.is_empty() { ".".to_string() @@ -459,7 +489,9 @@ impl DatasetLocation { } return Ok(preview); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let key = if relative.is_empty() { crate::store::CHRONICLE_MANIFEST_FILE.to_string() } else { @@ -569,7 +601,9 @@ impl DatasetLocation { return Ok(out); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let prefix = if relative.is_empty() { String::new() } else { @@ -721,7 +755,9 @@ impl DatasetLocation { return Ok(()); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let mut pending = vec![String::new()]; let mut found = 0usize; while let Some(prefix) = pending.pop() { @@ -837,7 +873,9 @@ impl DatasetLocation { if let Some(path) = &self.local_path { return put_local_bytes(path, bytes, overwrite); } - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; // DatasetLocation represents a prefix; use a stable marker inside it. let path = ".dataset-marker"; if overwrite { @@ -879,7 +917,9 @@ impl DatasetLocation { !url.path().trim_matches('/').is_empty(), "refusing to drop an entire object-store bucket; name a Dataset prefix" ); - let store = OpendalStore::from_uri(&self.uri).await?; + let store = + OpendalStore::from_uri_with_config(&self.uri, self.backend.clone().unwrap_or_default()) + .await?; let entries = store .list("") .await diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index 7e8241325..338e92076 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -133,13 +133,43 @@ impl ManifestCache { }, observed_at: chrono::Utc::now().timestamp(), }; - self.values - .write() - .await - .insert(key.clone(), listing.clone()); + let removed = { + let mut values = self.values.write().await; + // Only a successful observation can retire vanished descendants. + // Otherwise their old manifests would keep inflating UI counts. + let removed = values + .keys() + .filter(|other| { + other + .strip_prefix(&key) + .is_some_and(|suffix| suffix.starts_with('\0') || suffix.starts_with('/')) + && !listing.entries.iter().any(|entry| { + let relative = entry + .path + .strip_prefix(&format!("{prefix}/")) + .unwrap_or(&entry.path); + let child_key = format!( + "{key}{}{relative}", + if prefix.is_empty() { "\0" } else { "/" } + ); + (entry.kind == crate::store::PathListKind::Directory + && (other.as_str() == child_key + || other.starts_with(&format!("{child_key}/")))) + || (entry.kind == crate::store::PathListKind::Dataset + && other.as_str() == child_key) + }) + }) + .cloned() + .collect::>(); + for key in &removed { + values.remove(key); + } + values.insert(key.clone(), listing.clone()); + removed + }; if let Err(error) = self .disk - .upsert(&key, &serde_json::to_value(&listing)?, &[]) + .upsert(&key, &serde_json::to_value(&listing)?, &removed) .await { tracing::warn!(target: "pchronicle.serve", prefix, %error, @@ -222,6 +252,43 @@ impl ManifestCache { }) } + /// Whether every directory below a mount-relative prefix has an observed + /// manifest listing. Uses the same keys as refresh_mount; no remote I/O. + pub async fn is_complete_under(&self, mount_key: &str, prefix: &str) -> bool { + let values = self.values.read().await; + let key = |path: &str| { + if path.is_empty() { + mount_key.to_owned() + } else { + format!("{mount_key}\0{path}") + } + }; + if !values.contains_key(&key(prefix)) { + return false; + } + let descendant_key = format!("{mount_key}\0"); + values + .iter() + .filter(|(key, _)| { + let path = if key.as_str() == mount_key { + Some("") + } else { + key.strip_prefix(&descendant_key) + }; + path.is_some_and(|path| { + prefix.is_empty() || path == prefix || path.starts_with(&format!("{prefix}/")) + }) + }) + .all(|(_, listing)| { + !listing.partial + && listing + .entries + .iter() + .filter(|entry| entry.kind == crate::store::PathListKind::Directory) + .all(|entry| values.contains_key(&key(&entry.path))) + }) + } + /// Breadth-first refresh of a mount. Refreshes serialize per observation /// key, allowing unrelated foreground directories to load concurrently. pub async fn refresh_mount( @@ -341,6 +408,37 @@ mod tests { use super::*; use crate::store::{PathListEntry, PathListKind}; + #[tokio::test] + async fn successful_parent_observation_retires_deleted_manifest_counts() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + let leaf = source.join("nested/leaf"); + std::fs::create_dir_all(&leaf).unwrap(); + crate::storage::write_compact_jsonl_manifest(&leaf, 1, 42).unwrap(); + let path = temp.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + cache.refresh_mount("mount", &location).await.unwrap(); + assert_eq!(cache.summary("mount").await.trajectories, 42); + assert!(cache.is_complete_under("mount", "").await); + std::fs::remove_dir_all(source.join("nested")).unwrap(); + cache + .refresh_for_browse("mount", &location, "") + .await + .unwrap(); + assert_eq!(cache.summary("mount").await.datasets, 0); + assert!(cache.get("mount\0nested").await.is_none()); + drop(cache); + assert_eq!( + ManifestCache::open(path) + .await + .summary("mount") + .await + .trajectories, + 0 + ); + } + #[tokio::test] async fn bounded_walk_reports_and_persists_partial_then_recovers() { let dir = tempfile::tempdir().unwrap(); diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index c0c5dc489..39611dd50 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -196,9 +196,25 @@ fn state_for<'a>( state } +tokio::task_local! { + static BACKGROUND_IO: (); +} + +/// Run manifest maintenance with independent admission and AIMD state. Call +/// inside the spawned task: Tokio task-local state is not inherited by spawn. +pub async fn with_background_object_store_io(work: F) -> F::Output { + BACKGROUND_IO.scope((), work).await +} + fn gate() -> &'static Gate { - static GATE: OnceLock = OnceLock::new(); - GATE.get_or_init(|| { + static FOREGROUND_GATE: OnceLock = OnceLock::new(); + static BACKGROUND_GATE: OnceLock = OnceLock::new(); + let slot = if BACKGROUND_IO.try_with(|_| ()).is_ok() { + &BACKGROUND_GATE + } else { + &FOREGROUND_GATE + }; + slot.get_or_init(|| { let concurrency = std::env::var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") .ok() .and_then(|value| value.parse::().ok()) @@ -235,7 +251,7 @@ pub(crate) fn scope_key(uri: &str) -> String { scope_for_endpoint(uri, &endpoint) } -fn scope_for_endpoint(uri: &str, endpoint: &str) -> String { +pub(crate) fn scope_for_endpoint(uri: &str, endpoint: &str) -> String { let uri = uri.split('#').next().unwrap_or(uri); let Some((scheme, rest)) = uri.split_once("://") else { return uri.to_owned(); @@ -644,6 +660,34 @@ pub(crate) fn note_failure(uri: &str, kind: IoKind) { mod tests { use super::*; + #[tokio::test] + async fn foreground_and_background_have_independent_admission_and_feedback() { + let uri = scope_for_endpoint("s3://workload-isolation/path", "http://localhost:18060"); + let mut foreground_permits = Vec::new(); + for _ in 0..gate().concurrency { + foreground_permits.push(acquire(&uri, IoKind::Read).await); + } + with_background_object_store_io(async { + let _permit = + tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) + .await + .expect("background must not wait on foreground admission"); + note_failure(&uri, IoKind::Read); + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 0); + drop(foreground_permits); + let _permit = tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) + .await + .expect("background cooldown must not delay foreground requests"); + note_success(&uri); + with_background_object_store_io(async { + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + } + #[test] fn registry_reclaims_idle_scopes_but_preserves_waits_and_cooldowns() { let mut states = HashMap::new(); diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 2462c5cfe..0bee48479 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -15,6 +15,39 @@ use std::sync::{Mutex, OnceLock}; use std::time::{Duration, Instant}; use url::Url; +#[derive(Clone, Default, PartialEq, Eq)] +pub struct StoreConfig { + pub endpoint: Option, + pub region: Option, + pub access_key: Option, + pub secret_key: Option, +} + +impl std::fmt::Debug for StoreConfig { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("StoreConfig") + .field("endpoint", &self.endpoint) + .field("region", &self.region) + .finish_non_exhaustive() + } +} + +impl StoreConfig { + pub(crate) fn fingerprint(&self) -> String { + blake3::hash( + &serde_json::to_vec(&( + &self.endpoint, + &self.region, + &self.access_key, + &self.secret_key, + )) + .expect("serialize S3 configuration"), + ) + .to_hex() + .to_string() + } +} + /// Retries for transient object-store failures (DNS blips, connect resets, /// 5xx, rate limits). Tuned for long imports over flaky endpoints: up to 8 /// retries with exponential backoff + jitter, capped at 30s. @@ -111,6 +144,56 @@ impl OperatorRegistry { static OPERATORS: OnceLock> = OnceLock::new(); impl Store { + pub(crate) async fn from_uri_with_config(uri: &str, config: StoreConfig) -> Result { + if !uri.starts_with("s3://") || config == StoreConfig::default() { + return Self::from_uri(uri).await; + } + let parsed = Url::parse(uri).context("parse object-store URI")?; + let bucket = parsed + .host_str() + .ok_or_else(|| anyhow!("S3 URI must name a bucket"))?; + let root = parsed.path().trim_matches('/'); + let io_scope = io_gate::scope_for_endpoint(uri, config.endpoint.as_deref().unwrap_or("")); + let cache_key = format!( + "{}\0{}", + operator_cache_key(uri, parsed.as_str()), + config.fingerprint() + ); + let registry = OPERATORS.get_or_init(|| Mutex::new(OperatorRegistry::default())); + let cached = registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&cache_key, Instant::now()); + let operator = if let Some(operator) = cached { + operator + } else { + let mut builder = opendal::services::S3::default() + .bucket(bucket) + .root(root) + .region(config.region.as_deref().unwrap_or("us-east-1")); + if let Some(v) = config.endpoint.as_deref() { + builder = builder.endpoint(v); + } + if let Some(v) = config.access_key.as_deref() { + builder = builder.access_key_id(v); + } + if let Some(v) = config.secret_key.as_deref() { + builder = builder.secret_access_key(v); + } + let operator = with_object_store_retries(Operator::new(builder)?.finish()); + registry + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(cache_key, operator.clone(), Instant::now()); + operator + }; + Ok(Self { + operator, + fallback_lock: None, + io_scope, + }) + } + pub(crate) async fn from_uri(uri: &str) -> Result { let uri = uri.trim(); let normalized = normalize_uri(uri)?; @@ -476,6 +559,43 @@ fn normalize_uri(uri: &str) -> Result { mod tests { use super::*; + #[tokio::test] + async fn explicit_backends_scope_gate_by_endpoint_and_bucket_not_credentials() { + let config = StoreConfig { + endpoint: Some("http://127.0.0.1:18060".into()), + region: Some("us-east-1".into()), + access_key: Some("scope-test-key".into()), + secret_key: Some("scope-test-secret".into()), + }; + let a = Store::from_uri_with_config("s3://scope-test/a", config.clone()) + .await + .unwrap(); + let mut other = config.clone(); + other.endpoint = Some("http://127.0.0.1:18061".into()); + let b = Store::from_uri_with_config("s3://scope-test/a", other.clone()) + .await + .unwrap(); + assert_ne!(a.io_scope, b.io_scope); + assert_ne!(config.fingerprint(), other.fingerprint()); + other = config.clone(); + other.secret_key = Some("rotated-secret".into()); + assert_ne!(config.fingerprint(), other.fingerprint()); + let c = Store::from_uri_with_config("s3://scope-test/b", other) + .await + .unwrap(); + assert_eq!(a.io_scope, c.io_scope); + assert!(!format!("{config:?}").contains("scope-test-secret")); + assert!(!format!("{config:?}").contains("scope-test-key")); + // Saturate A without making a network request: B must retain admission. + let _a = io_gate::acquire(&a.io_scope, IoKind::Read).await; + let _b = tokio::time::timeout( + Duration::from_millis(100), + io_gate::acquire(&b.io_scope, IoKind::Read), + ) + .await + .unwrap(); + } + #[test] fn temporary_and_exhausted_transport_errors_still_trigger_backoff() { for kind in [ErrorKind::Unexpected, ErrorKind::RateLimited] { diff --git a/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md b/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md index 43f6797ea..de04ab1bb 100644 --- a/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md +++ b/docs/src/en/rfcs/0016-pchronicle-catalog-resolution-and-cache.md @@ -71,20 +71,30 @@ unreadable cache MUST rebuild it and continue with an empty cache. Cache errors MUST NOT make `serve` fail when the authoritative location is still available. -The cache exposes three refresh paths: - -1. A foreground tree request refreshes only the requested level when its - cached view is missing or explicitly stale. -2. A periodic catalog refresh runs every 30 seconds and performs a breadth - first walk of each mount. This publishes shallow levels before deeper - levels and bounds the frontier. -3. A successful refresh updates memory and persistent storage atomically from - the caller's point of view. A failed refresh leaves the previous view - available and records the error for the UI. - -At most one manifest refresh is active per cache instance. This prevents -duplicate timers and concurrent scans from multiplying local or object-store -I/O. The UI browse worker also has a bounded process-level I/O gate. +The Warehouse separates interactive directory browsing from manifest maintenance: + +| Work | Remote I/O | Local result | Admission and backoff | +|---|---|---|---| +| `GET /api/explorer/tree` | One-level LIST on a directory cache miss or refresh; no remote manifest probes | Names enriched from `ManifestCache` with Dataset types and trajectory counts | Foreground AIMD, scoped by endpoint + bucket | +| Background manifest worker | Probe the current prefix and list immediate children, then walk breadth first | Update persistent manifest observations and the browse projection | Independent background AIMD, also scoped by endpoint + bucket | + +Foreground requests use an existing directory projection or manifest observation +immediately. A cold request starts a shallow LIST and waits up to 250 ms before +returning an explicit loading view. Metadata that has not been observed is +partial, not a certified zero count. Background work is bounded to 32 prefixes +per 30-second round and a 10,000-directory frontier; unfinished observations +remain partial. Each refresh job has a 10-second deadline. + +The two workloads do not share in-flight deduplication or failure cooldowns. +A foreground LIST of a prefix can run while that same prefix's background +manifest request is stalled. They may reuse an OpenDAL client, but their AIMD +semaphores, failure counters and cooldowns are independent. Each workload +continues to share admission across paths on the same endpoint and bucket. + +A successful observation updates memory and attempts persistent storage. A +failed observation retains the previous cache. The directory projection stores +navigation data; Dataset identity and aggregate counts come only from the +local manifest observations, scoped to the selected mount and prefix. ## Aggregation semantics diff --git a/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md b/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md index 43f6797ea..de04ab1bb 100644 --- a/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md +++ b/docs/src/zh/rfcs/0016-pchronicle-catalog-resolution-and-cache.md @@ -71,20 +71,30 @@ unreadable cache MUST rebuild it and continue with an empty cache. Cache errors MUST NOT make `serve` fail when the authoritative location is still available. -The cache exposes three refresh paths: - -1. A foreground tree request refreshes only the requested level when its - cached view is missing or explicitly stale. -2. A periodic catalog refresh runs every 30 seconds and performs a breadth - first walk of each mount. This publishes shallow levels before deeper - levels and bounds the frontier. -3. A successful refresh updates memory and persistent storage atomically from - the caller's point of view. A failed refresh leaves the previous view - available and records the error for the UI. - -At most one manifest refresh is active per cache instance. This prevents -duplicate timers and concurrent scans from multiplying local or object-store -I/O. The UI browse worker also has a bounded process-level I/O gate. +The Warehouse separates interactive directory browsing from manifest maintenance: + +| Work | Remote I/O | Local result | Admission and backoff | +|---|---|---|---| +| `GET /api/explorer/tree` | One-level LIST on a directory cache miss or refresh; no remote manifest probes | Names enriched from `ManifestCache` with Dataset types and trajectory counts | Foreground AIMD, scoped by endpoint + bucket | +| Background manifest worker | Probe the current prefix and list immediate children, then walk breadth first | Update persistent manifest observations and the browse projection | Independent background AIMD, also scoped by endpoint + bucket | + +Foreground requests use an existing directory projection or manifest observation +immediately. A cold request starts a shallow LIST and waits up to 250 ms before +returning an explicit loading view. Metadata that has not been observed is +partial, not a certified zero count. Background work is bounded to 32 prefixes +per 30-second round and a 10,000-directory frontier; unfinished observations +remain partial. Each refresh job has a 10-second deadline. + +The two workloads do not share in-flight deduplication or failure cooldowns. +A foreground LIST of a prefix can run while that same prefix's background +manifest request is stalled. They may reuse an OpenDAL client, but their AIMD +semaphores, failure counters and cooldowns are independent. Each workload +continues to share admission across paths on the same endpoint and bucket. + +A successful observation updates memory and attempts persistent storage. A +failed observation retains the previous cache. The directory projection stores +navigation data; Dataset identity and aggregate counts come only from the +local manifest observations, scoped to the selected mount and prefix. ## Aggregation semantics diff --git a/pchronicle-web/src/catalog.rs b/pchronicle-web/src/catalog.rs index ba69d065d..16963d1ec 100644 --- a/pchronicle-web/src/catalog.rs +++ b/pchronicle-web/src/catalog.rs @@ -34,12 +34,14 @@ pub fn CatalogExplorer( p { "{catalog_subtitle(tree.as_ref())}" } if let Some(status) = tree.as_ref().and_then(|tree| tree.browse.as_ref()) { p { role: "status", - if status.partial { - "Partial view · Some directories have not been loaded" - } else if status.observed_at == 0 && status.refreshing { + if status.observed_at == 0 && status.refreshing { "Loading directory contents…" + } else if status.observed_at == 0 && status.state == "unavailable" { + "Directory unavailable · Retrying automatically" } else if status.last_error.is_some() { "Showing cached view · Refresh failed; retrying automatically" + } else if status.partial { + "Partial view · Some directories have not been loaded" } else if status.refreshing { "Showing cached view · Refreshing in background" } else if status.stale { @@ -69,7 +71,7 @@ pub fn CatalogExplorer( button { class: "button primary", onclick: on_settings, "Open Keys" } } } else if tree.as_ref().and_then(|tree| tree.browse.as_ref()) - .is_some_and(|status| status.observed_at == 0 && status.last_error.is_some()) { + .is_some_and(|status| status.observed_at == 0 && (status.state == "unavailable" || status.last_error.is_some())) { div { class: "pc-catalog-empty", strong { "Directory unavailable" } span { "Retrying automatically. Check the storage connection if this persists." } } } else if tree.as_ref().is_none_or(|tree| tree.children.is_empty() && tree.run_count == 0) { div { class: "pc-catalog-empty", strong { "No datasets" } span { "Add a dataset, then refresh this page." } } diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index f0e54a68e..80a669189 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -284,6 +284,7 @@ pub fn App() -> Element { let mut catalog_tree = use_signal(|| None::); let mut catalog_loading = use_signal(|| false); let mut catalog_generation = use_signal(|| 0u64); + let mut catalog_task = use_signal(|| None::); let mut offset = use_signal(|| 0usize); let mut error = use_signal(|| None::); let mut catalog_auth_configured = use_signal(|| catalog_auth::load().is_configured()); @@ -346,27 +347,20 @@ pub fn App() -> Element { }); use_effect(move || { + // One task per navigation. Cancel even on A -> B -> A, so old pollers + // cannot resume or overwrite the current page. + if let Some(task) = *catalog_task.peek() { + task.cancel(); + } + let requested_generation = (*catalog_generation.peek()).saturating_add(1); + catalog_generation.set(requested_generation); if page() != "catalog" { return; } let dataset = catalog_dataset(); let prefix = catalog_prefix(); - load_catalog_tree( - dataset.clone(), - prefix.clone(), - catalog_tree, - catalog_loading, - error, - catalog_generation, - catalog_generation(), - ); - spawn(async move { + catalog_task.set(Some(spawn(async move { loop { - TimeoutFuture::new(CATALOG_REFRESH_MS).await; - if page() != "catalog" || catalog_dataset() != dataset || catalog_prefix() != prefix - { - break; - } load_catalog_tree( dataset.clone(), prefix.clone(), @@ -374,10 +368,14 @@ pub fn App() -> Element { catalog_loading, error, catalog_generation, - catalog_generation(), - ); + requested_generation, + ) + .await; + // Wait after completion. Slow requests must never overlap the + // next poll, and failures must not cause a tight retry loop. + TimeoutFuture::new(CATALOG_REFRESH_MS).await; } - }); + }))); }); use_effect(move || { @@ -1148,7 +1146,7 @@ fn merge_run_pages(pages: &[RunPage], filters: &RunFilters) -> RunPage { } } -fn load_catalog_tree( +async fn load_catalog_tree( dataset: String, prefix: String, mut tree: Signal>, @@ -1158,31 +1156,27 @@ fn load_catalog_tree( requested_generation: u64, ) { loading.set(true); - spawn(async move { - match api::explorer_tree(&dataset, &prefix).await { - Ok(value) if generation() == requested_generation => tree.set(Some(value)), - Err(failure) - if matches!(failure.status, 400 | 401) - && dataset.is_empty() - && prefix.is_empty() => - { - match api::explorer_tree_anonymous(&dataset, &prefix).await { - Ok(value) if generation() == requested_generation => tree.set(Some(value)), - Err(failure) if generation() == requested_generation => { - error.set(Some(workspace_notice(&failure))) - } - _ => {} + match api::explorer_tree(&dataset, &prefix).await { + Ok(value) if *generation.peek() == requested_generation => tree.set(Some(value)), + Err(failure) + if matches!(failure.status, 400 | 401) && dataset.is_empty() && prefix.is_empty() => + { + match api::explorer_tree_anonymous(&dataset, &prefix).await { + Ok(value) if *generation.peek() == requested_generation => tree.set(Some(value)), + Err(failure) if *generation.peek() == requested_generation => { + error.set(Some(workspace_notice(&failure))) } + _ => {} } - Err(failure) if generation() == requested_generation => { - error.set(Some(workspace_notice(&failure))) - } - _ => {} } - if generation() == requested_generation { - loading.set(false); + Err(failure) if *generation.peek() == requested_generation => { + error.set(Some(workspace_notice(&failure))) } - }); + _ => {} + } + if *generation.peek() == requested_generation { + loading.set(false); + } } #[allow(clippy::too_many_arguments)] From 04c4ca8b2458320d7e1a20f21eae4ff0beef62c8 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 17:46:15 +0800 Subject: [PATCH 13/22] feat(request_progress): implement request diagnostics and progress tracking - Introduced a new `request_progress` module to provide bounded, opt-in diagnostics for API requests, allowing users to track execution phases and outcomes. - Added functionality to capture and report the state of requests, including phases such as authentication, execution, and response. - Enhanced middleware to integrate request progress tracking, ensuring that diagnostic information is available for ongoing requests. - Updated the UI to display request diagnostics, improving user visibility into API interactions and their statuses. This commit aims to enhance the observability and user experience of API requests in the pChronicle application. --- .../src/server/catalog.rs | 5 +- .../src/server/catalog_worker.rs | 49 +- .../src/server/mod.rs | 54 ++- .../src/server/query_admission.rs | 2 + .../src/server/request_log.rs | 41 +- .../src/server/request_progress.rs | 434 ++++++++++++++++++ .../src/server/tests.rs | 51 +- .../src/server/ui_cache.rs | 6 + .../src/store/catalog/discovery.rs | 63 ++- .../src/store/catalog/tests.rs | 35 ++ docs/src/en/pchronicle/reference/cli.md | 29 ++ docs/src/zh/pchronicle/reference/cli.md | 22 + pchronicle-web/Cargo.toml | 2 +- pchronicle-web/assets/workbench.css | 5 + pchronicle-web/src/api.rs | 33 +- pchronicle-web/src/main.rs | 1 + pchronicle-web/src/requests.rs | 286 ++++++++++++ pchronicle-web/src/workspace.rs | 85 +++- 18 files changed, 1137 insertions(+), 66 deletions(-) create mode 100644 crates/persisting-pchronicle-cli/src/server/request_progress.rs create mode 100644 pchronicle-web/src/requests.rs diff --git a/crates/persisting-pchronicle-cli/src/server/catalog.rs b/crates/persisting-pchronicle-cli/src/server/catalog.rs index 3b9db94f3..4d60e28ae 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog.rs @@ -831,7 +831,8 @@ fn parent_handles_path(path: &str) -> bool { .strip_prefix("/api/v1") .or_else(|| path.strip_prefix("/api")) .unwrap_or(path); - rest == "/health" + rest.starts_with("/requests/") + || rest == "/health" || rest == "/ui" || rest == "/catalog/datasets" || rest.starts_with("/catalog/datasets/") @@ -1359,6 +1360,8 @@ dataset = "prod" #[test] fn parent_keeps_health_and_catalog_ticket_routes() { assert!(parent_handles_path("/api/health")); + assert!(parent_handles_path("/api/requests/example")); + assert!(parent_handles_path("/api/v1/requests/example")); assert!(parent_handles_path("/api/v1/catalog/datasets")); assert!(parent_handles_path("/api/v1/catalog/datasets/prod")); assert!(!parent_handles_path("/api/catalog")); diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index 12d244f2f..2cffed552 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -53,6 +53,13 @@ impl WorkerResponse { } } +#[derive(Serialize, Deserialize)] +#[serde(tag = "type", content = "value")] +enum WorkerEvent { + Progress(super::request_progress::Snapshot), + Response(WorkerResponse), +} + #[derive(Serialize, Deserialize)] struct Bootstrap { mounts: Vec, @@ -114,14 +121,17 @@ impl WorkerPool { request: WorkerRequest, ) -> Result { tokio::time::timeout(REQUEST_TIMEOUT, async { + super::request_progress::phase("worker_queue"); let slot = self.slot(&scope).await?; let mut guard = slot.lock().await; // Ownership stays in this future during IPC: cancellation, timeout // or a partial frame drops/kills it rather than reusing dirty pipes. + super::request_progress::phase("worker_start"); let mut worker = match guard.take() { Some(worker) => worker, None => Worker::start(&scope, mounts).await.map_err(worker_error)?, }; + super::request_progress::phase("worker_execution"); let response = worker.exchange(&request).await.map_err(worker_error)?; let response = response.into_response().map_err(worker_error)?; *guard = Some(worker); @@ -238,7 +248,16 @@ impl Worker { async fn exchange(&mut self, request: &WorkerRequest) -> Result { self.send(request).await?; - self.receive().await + loop { + match self.receive::().await? { + WorkerEvent::Progress(snapshot) => { + if let Some(p) = super::request_progress::current() { + p.worker(snapshot); + } + } + WorkerEvent::Response(response) => return Ok(response), + } + } } } @@ -318,7 +337,20 @@ pub(crate) fn run() -> Result<()> { job.body.len() <= 1024 * 1024, "worker request body too large" ); + let id = job + .headers + .iter() + .find(|(name, _)| name == "x-request-id") + .map(|(_, v)| v.clone()) + .unwrap_or_default(); + let progress = super::request_progress::Progress::new( + id, + job.method.clone(), + job.uri.split('?').next().unwrap_or_default().to_owned(), + true, + ); let result = runtime.block_on(async { + let operation = async { let mut builder = axum::http::Request::builder() .method(job.method.as_str()) .uri(job.uri); @@ -327,7 +359,8 @@ pub(crate) fn run() -> Result<()> { } let response = warehouse .router() - .oneshot(builder.body(axum::body::Body::from(job.body))?) + .oneshot({ let mut request=builder.body(axum::body::Body::from(job.body))?; + request.extensions_mut().insert(progress.clone()); request }) .await?; let status = response.status().as_u16(); let headers = response @@ -359,8 +392,18 @@ pub(crate) fn run() -> Result<()> { headers, body, }) + }; + tokio::pin!(operation); + let mut tick = tokio::time::interval(Duration::from_millis(250)); + loop { + tokio::select! { + result = &mut operation => break result, + _ = tick.tick() => { output.write_all(&encode(&WorkerEvent::Progress(progress.snapshot()))?)?; output.flush()?; } + } + } })?; - output.write_all(&encode(&result)?)?; + output.write_all(&encode(&WorkerEvent::Progress(progress.snapshot()))?)?; + output.write_all(&encode(&WorkerEvent::Response(result))?)?; output.flush()?; } Ok(()) diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 2ac7b19f1..1d00bbbe6 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -9,6 +9,7 @@ mod physical; pub(crate) mod problem; mod query_admission; pub(crate) mod request_log; +mod request_progress; mod ui_cache; use std::collections::{BTreeMap, BTreeSet}; @@ -68,6 +69,7 @@ struct AppState { catalog_query_worker: bool, catalog_workers: Arc, browse_mounts: Arc>, + request_progress: Arc, browse: Arc>, scoped_queries: Arc, } @@ -239,6 +241,7 @@ fn app_state_with_catalog_refresh_interval( catalog_query_worker: false, catalog_workers: Arc::new(catalog_worker::WorkerPool::default()), browse_mounts: Arc::new(browse_mounts), + request_progress: Arc::new(request_progress::Registry::default()), browse: Arc::new(tokio::sync::OnceCell::new()), scoped_queries: Arc::new(query_admission::ScopedQueries::default()), } @@ -379,6 +382,7 @@ async fn browse_coordinator(state: &AppState) -> &ui_cache::BrowseCoordinator { fn api_routes() -> Router { Router::new() .route("/health", get(warehouse_health)) + .route("/requests/{id}", get(request_progress::get)) .route("/ui", get(ui_config)) .route("/runs", get(runs)) .route("/explorer/runs", get(explorer_runs)) @@ -417,7 +421,8 @@ fn finish_routes(state: AppState) -> Router { state.clone(), catalog::catalog_data_plane_layer, )) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + state.clone(), request_log::warehouse_request_layer, )) .with_state(state) @@ -503,6 +508,7 @@ async fn ui_config(State(state): State) -> Json { async fn build_catalog_runtime( config: &ChronicleServerConfig, ) -> anyhow::Result> { + request_progress::phase("source_metadata"); let snapshot = Arc::new( DatasetCatalogSnapshot::discover( config.datasets.clone(), @@ -511,6 +517,7 @@ async fn build_catalog_runtime( ) .await?, ); + request_progress::phase("query"); let engine = Arc::new( snapshot .clone() @@ -530,6 +537,7 @@ async fn build_scoped_query_runtime( scope: persisting_pchronicle::storage::QueryScope, cached_files: Vec, ) -> anyhow::Result> { + request_progress::phase("source_metadata"); let target = match scope.source_file.clone() { Some(file) => persisting_pchronicle::storage::ResolveTarget::Dataset { mount: scope.dataset.clone(), @@ -557,6 +565,7 @@ async fn build_scoped_query_runtime( config.catalog_options, ); let snapshot = Arc::new(resolver.resolve(target, mode, &cached_datasets).await?); + request_progress::phase("query"); let engine = Arc::new( snapshot .clone() @@ -578,6 +587,7 @@ async fn current_catalog( if let Some(runtime) = state.catalog.read().await.as_ref() { return Ok(Arc::clone(runtime)); } + request_progress::phase("catalog_wait"); let _refresh = state.catalog_refresh.lock().await; if let Some(runtime) = state.catalog.read().await.as_ref() { return Ok(Arc::clone(runtime)); @@ -1025,6 +1035,7 @@ async fn open_storyline_source( file: &str, request_id: &RequestId, ) -> Result, ApiError> { + request_progress::phase("source_metadata"); let Some(mount) = state .config .datasets @@ -1093,6 +1104,7 @@ async fn try_on_demand_storyline_runs_page( let Some(store) = open_storyline_source(state, dataset_name, file, request_id).await? else { return Ok(None); }; + request_progress::phase("storage_read"); let Some((_generation, ids)) = store .document_ids_snapshot() .await @@ -1194,6 +1206,7 @@ async fn explorer_runs_inner( fts: FtsDiagnostics, query: Result, QueryRejection>, ) -> Result, ApiError> { + request_progress::phase("query"); let query = api_query(query)?; if let Some(page) = try_on_demand_storyline_runs_page(&state, &query, &request_id).await? { return Ok(Json(page)); @@ -1466,6 +1479,7 @@ async fn explorer_tree( ); let Some(name) = dataset else { let started = Instant::now(); + request_progress::phase("browse_cache"); let view = browse_coordinator(&state) .await .roots(&state.browse_mounts) @@ -1648,12 +1662,11 @@ async fn try_resolve_on_demand_storyline_run( } async fn load_on_demand_storyline_bundle( - state: &AppState, + runtime: &CatalogRuntime, run: &RunSummary, request_id: &RequestId, op: &'static str, ) -> Result, ApiError> { - let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; let Some(dataset) = runtime.snapshot.dataset(&run.dataset) else { return Ok(None); }; @@ -1706,11 +1719,12 @@ async fn load_on_demand_storyline_bundle( async fn catalog_or_on_demand_trajectory_bundle( state: &AppState, + runtime: &CatalogRuntime, run: &RunSummary, request_id: &RequestId, op: &'static str, ) -> Result { - let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; + request_progress::phase("storage_read"); let key = catalog_storyline_key(run); let catalog_result = if state.live_reads { runtime.snapshot.load_live_trajectory_bundle(&key).await @@ -1719,12 +1733,12 @@ async fn catalog_or_on_demand_trajectory_bundle( }; match catalog_result { Ok(Some(bundle)) => Ok(bundle), - Ok(None) => load_on_demand_storyline_bundle(state, run, request_id, op) + Ok(None) => load_on_demand_storyline_bundle(runtime, run, request_id, op) .await? .ok_or_else(|| ApiError::not_found("run was not found")), Err(error) => { if let Some(bundle) = - load_on_demand_storyline_bundle(state, run, request_id, op).await? + load_on_demand_storyline_bundle(runtime, run, request_id, op).await? { Ok(bundle) } else { @@ -1780,8 +1794,10 @@ async fn load_events( request_id: &RequestId, ) -> Result { let run = resolve_run_summary(state, query, request_id, None).await?; + let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; let bundle = - catalog_or_on_demand_trajectory_bundle(state, &run, request_id, "load_events").await?; + catalog_or_on_demand_trajectory_bundle(state, &runtime, &run, request_id, "load_events") + .await?; let document = bundle.event_view; let offset = query .offset @@ -1843,8 +1859,10 @@ async fn storyline( ) -> Result, ApiError> { let query = api_query(query)?; let run = resolve_run_summary(&state, &query, &request_id, None).await?; + let runtime = catalog_for_source(&state, Some(&run.dataset), &run.file, &request_id).await?; let bundle = - catalog_or_on_demand_trajectory_bundle(&state, &run, &request_id, "storyline").await?; + catalog_or_on_demand_trajectory_bundle(&state, &runtime, &run, &request_id, "storyline") + .await?; Ok(Json( serde_json::to_value(bundle.storyline) .map_err(anyhow::Error::from) @@ -1971,6 +1989,7 @@ fn event_seqs_for_turn(turn: &StorylineTurn, by_call: &BTreeMap #[derive(Clone)] struct LoadedTrajectory { + runtime: Arc, run: RunSummary, event_provenance: CatalogEventProvenance, records: Vec, @@ -1987,7 +2006,7 @@ async fn load_trajectory( let run = resolve_run_summary(state, query, request_id, Some(metrics)).await?; metrics.record("resolve", phase); let phase = Instant::now(); - let runtime = current_catalog(state, request_id).await?; + let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; metrics.record("catalog", phase); let cache_key = format!( "{}\u{1f}{}\u{1f}{}\u{1f}{}", @@ -2009,6 +2028,7 @@ async fn load_trajectory( } if run.format.as_deref() == Some("compact-jsonl/v1") { return Ok(LoadedTrajectory { + runtime, run, event_provenance: CatalogEventProvenance::SyntheticFromStoryline, records: Vec::new(), @@ -2016,8 +2036,14 @@ async fn load_trajectory( }); } let phase = Instant::now(); - let bundle = - catalog_or_on_demand_trajectory_bundle(state, &run, request_id, "load_trajectory").await?; + let bundle = catalog_or_on_demand_trajectory_bundle( + state, + &runtime, + &run, + request_id, + "load_trajectory", + ) + .await?; metrics.record("trajectory_read", phase); let event_provenance = bundle.event_view.provenance; let records = bundle.event_view.document.events; @@ -2083,6 +2109,7 @@ async fn load_trajectory( }) .collect(); let loaded = LoadedTrajectory { + runtime, run, event_provenance, records, @@ -2154,7 +2181,7 @@ async fn explorer_record( return Err(ApiError::not_found("run is not a compact JSONL record")); } let key = catalog_storyline_key(&run); - let record = current_catalog(&state, &request_id) + let record = catalog_for_source(&state, Some(&run.dataset), &run.file, &request_id) .await? .snapshot .compact_record(&key) @@ -2204,7 +2231,7 @@ async fn explorer_turns( let session = query.session(); let loaded = load_trajectory(&state, &session, &request_id, &metrics).await?; let phase = Instant::now(); - let runtime = current_catalog(&state, &request_id).await?; + let runtime = &loaded.runtime; metrics.record("turn_catalog", phase); // Nested Directory Storylines are opened on-demand and are absent from the // prepared catalog; skip FTS path probing and keep in-memory turn pages. @@ -2250,7 +2277,6 @@ async fn explorer_turns( let expression = crate::combine_match_expressions(&[needle.to_owned()]) .map_err(|error| ApiError::invalid_request(error.to_string()))? .ok_or_else(|| ApiError::invalid_request("search query must not be empty"))?; - let runtime = current_catalog(&state, &request_id).await?; let phase = Instant::now(); let (predicate, available, fts_errors) = crate::find_expression_predicate_for_dataset( &runtime.snapshot, diff --git a/crates/persisting-pchronicle-cli/src/server/query_admission.rs b/crates/persisting-pchronicle-cli/src/server/query_admission.rs index 17938fc03..13ea2ee9f 100644 --- a/crates/persisting-pchronicle-cli/src/server/query_admission.rs +++ b/crates/persisting-pchronicle-cli/src/server/query_admission.rs @@ -212,12 +212,14 @@ impl ScopedQueries { flight } }; + super::request_progress::phase("query_queue"); // OnceCell transfers initialization to a waiter if the initializing // request is cancelled. No detached tasks or permanently owned pins. flight .result .get_or_init(|| async { let _slot = self.slots.acquire().await.expect("admission never closes"); + super::request_progress::phase("query"); execute().await.map_err(SharedAccelerationFailure::new) }) .await diff --git a/crates/persisting-pchronicle-cli/src/server/request_log.rs b/crates/persisting-pchronicle-cli/src/server/request_log.rs index d077696a9..f2512a39c 100644 --- a/crates/persisting-pchronicle-cli/src/server/request_log.rs +++ b/crates/persisting-pchronicle-cli/src/server/request_log.rs @@ -113,10 +113,17 @@ where } } -pub(crate) async fn warehouse_request_layer( +pub(super) async fn warehouse_request_layer( + axum::extract::State(state): axum::extract::State, mut request: Request, next: Next, ) -> Response { + // Progress polling must stay cheap and must not generate more request records. + if request.uri().path().starts_with("/api/requests/") + || request.uri().path().starts_with("/api/v1/requests/") + { + return next.run(request).await; + } let incoming = request .headers() .get("x-request-id") @@ -143,8 +150,36 @@ pub(crate) async fn warehouse_request_layer( query = %truncate_utf8(&query, QUERY_LOG_LIMIT), "warehouse request start" ); - let response = next.run(request).await; + // Worker requests carry their trace over private IPC; public observers + // never acquire a worker slot just to inspect a stalled request. + let progress = request + .extensions() + .get::() + .cloned() + .unwrap_or_else(|| { + super::request_progress::Progress::new( + request_id.clone(), + method.clone(), + path.clone(), + false, + ) + }); + if !path.contains("/requests/") && path.starts_with("/api/") { + state + .request_progress + .insert(request.headers(), progress.clone()); + } + let _cancel = super::request_progress::CancelOnDrop(progress.clone()); + progress.phase(if state.catalog_query_worker { + "execution" + } else { + "authentication" + }); + let response = super::request_progress::scope(progress.clone(), next.run(request)).await; let status = response.status(); + if status.is_success() { + progress.phase("response"); + } let root_cause = response .extensions() .get::() @@ -165,6 +200,8 @@ pub(crate) async fn warehouse_request_layer( query = %truncate_utf8(&query, QUERY_LOG_LIMIT), "warehouse request" ); + let diagnostic_error = error_fields.as_ref().map(|(_, message)| message.clone()); + progress.finish(Some(status.as_u16()), diagnostic_error); if (400..500).contains(&status.as_u16()) { let (code, message) = error_fields.unwrap_or_default(); let fts_errors = fts.joined(); diff --git a/crates/persisting-pchronicle-cli/src/server/request_progress.rs b/crates/persisting-pchronicle-cli/src/server/request_progress.rs new file mode 100644 index 000000000..d66a4d550 --- /dev/null +++ b/crates/persisting-pchronicle-cli/src/server/request_progress.rs @@ -0,0 +1,434 @@ +//! Bounded, opt-in request diagnostics. The observer token is a capability, +//! separate from catalog credentials, so even authentication failures are visible. +use super::problem::ApiError; +use axum::{ + Json, + extract::{Path, State}, + http::HeaderMap, +}; +use serde::{Deserialize, Serialize}; +use std::{ + collections::HashMap, + future::Future, + sync::{Arc, Mutex}, + time::{Duration, Instant}, +}; + +pub const OBSERVER_HEADER: &str = "x-pchronicle-observer"; +const LIMIT: usize = 512; +const TTL: Duration = Duration::from_secs(600); + +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct Phase { + pub name: String, + pub state: String, + pub elapsed_ms: u64, +} +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct Snapshot { + pub request_id: String, + pub method: String, + pub path: String, + pub state: String, + pub elapsed_ms: u64, + pub status: Option, + pub error: Option, + pub note: Option, + pub phases: Vec, + pub worker: Option>, +} +struct Running { + snapshot: Snapshot, + started: Instant, + phase_started: Instant, + worker_updated: Instant, +} +#[derive(Clone)] +pub struct Progress(Arc>); +#[derive(Default)] +pub struct Registry(Mutex>); + +tokio::task_local! { static CURRENT: Progress; } +pub async fn scope(progress: Progress, work: F) -> F::Output { + CURRENT.scope(progress, work).await +} +pub fn current() -> Option { + CURRENT.try_with(Clone::clone).ok() +} +pub fn phase(name: &str) { + if let Some(p) = current() { + p.phase(name); + } +} +fn ms(d: Duration) -> u64 { + d.as_millis().min(u64::MAX as u128) as u64 +} + +impl Progress { + pub fn new(id: String, method: String, path: String, worker: bool) -> Self { + let mut names = if worker { + vec!["execution"] + } else { + vec![ + "authentication", + "worker_queue", + "worker_start", + "worker_execution", + ] + }; + if path.ends_with("/explorer/tree") { + names.extend(["browse_cache", "directory_wait", "manifest_summary"]); + } else { + names.extend(["source_metadata", "storage_read", "query"]); + } + names.push("response"); + let now = Instant::now(); + Self(Arc::new(Mutex::new(Running { + snapshot: Snapshot { + request_id: id, + method, + path, + state: "running".into(), + elapsed_ms: 0, + status: None, + error: None, + note: None, + phases: names + .into_iter() + .map(|name| Phase { + name: name.into(), + state: "pending".into(), + elapsed_ms: 0, + }) + .collect(), + worker: None, + }, + started: now, + phase_started: now, + worker_updated: now, + }))) + } + pub fn note(&self, note: &str) { + self.0 + .lock() + .unwrap_or_else(|e| e.into_inner()) + .snapshot + .note = Some(note.into()); + } + pub fn phase(&self, name: &str) { + let mut v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + if v.snapshot.state != "running" { + return; + } + if v.snapshot + .phases + .iter() + .any(|p| p.name == name && p.state == "running") + { + return; + } + let elapsed = ms(v.phase_started.elapsed()); + for p in &mut v.snapshot.phases { + if p.state == "running" { + p.state = "completed".into(); + p.elapsed_ms += elapsed; + } + } + if let Some(p) = v.snapshot.phases.iter_mut().find(|p| p.name == name) { + p.state = "running".into(); + } else { + v.snapshot.phases.push(Phase { + name: name.into(), + state: "running".into(), + elapsed_ms: 0, + }); + } + v.phase_started = Instant::now(); + } + pub fn snapshot(&self) -> Snapshot { + let v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + let mut s = v.snapshot.clone(); + if s.state == "running" { + s.elapsed_ms = ms(v.started.elapsed()); + for p in &mut s.phases { + if p.state == "running" { + p.elapsed_ms += ms(v.phase_started.elapsed()); + } + } + if let Some(w) = s.worker.as_mut() { + if w.state == "running" { + let elapsed = ms(v.worker_updated.elapsed()); + w.elapsed_ms += elapsed; + for p in &mut w.phases { + if p.state == "running" { + p.elapsed_ms += elapsed; + } + } + } + } + } + s + } + pub fn worker(&self, snapshot: Snapshot) { + let mut v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + v.snapshot.worker = Some(Box::new(snapshot)); + v.worker_updated = Instant::now(); + } + pub fn finish(&self, status: Option, error: Option) { + let mut v = self.0.lock().unwrap_or_else(|e| e.into_inner()); + if v.snapshot.state != "running" { + return; + } + let state = if status.is_none() { + "cancelled" + } else if status.unwrap() >= 400 || error.is_some() { + "failed" + } else { + "completed" + }; + let elapsed = ms(v.phase_started.elapsed()); + for p in &mut v.snapshot.phases { + if p.state == "running" { + p.state = state.into(); + p.elapsed_ms += elapsed; + } else if p.state == "pending" { + p.state = "skipped".into(); + } + } + if let Some(w) = v.snapshot.worker.as_mut() { + if w.state == "running" && state != "completed" { + w.state = state.into(); + for p in &mut w.phases { + if p.state == "running" { + p.state = state.into(); + } + } + } + } + v.snapshot.state = state.into(); + v.snapshot.elapsed_ms = ms(v.started.elapsed()); + v.snapshot.status = status; + v.snapshot.error = error; + } +} +pub struct CancelOnDrop(pub Progress); +impl Drop for CancelOnDrop { + fn drop(&mut self) { + self.0 + .finish(None, Some("Request execution was cancelled".into())); + } +} + +fn observer(headers: &HeaderMap) -> Option { + let token = headers.get(OBSERVER_HEADER)?.to_str().ok()?; + if token.len() != 32 || !token.bytes().all(|b| b.is_ascii_hexdigit()) { + return None; + } + Some(blake3::hash(token.as_bytes()).to_hex().to_string()) +} +impl Registry { + pub fn insert(&self, headers: &HeaderMap, progress: Progress) { + let Some(owner) = observer(headers) else { + return; + }; + let mut entries = self.0.lock().unwrap_or_else(|e| e.into_inner()); + entries.retain(|_, p| { + p.0.lock() + .unwrap_or_else(|e| e.into_inner()) + .started + .elapsed() + < TTL + }); + if entries.len() >= LIMIT { + if let Some(oldest) = entries + .iter() + .max_by_key(|(_, p)| { + p.0.lock() + .unwrap_or_else(|e| e.into_inner()) + .started + .elapsed() + }) + .map(|(k, _)| k.clone()) + { + entries.remove(&oldest); + } + } + let id = progress.snapshot().request_id; + entries.insert((owner, id), progress); + } + fn get(&self, headers: &HeaderMap, id: &str) -> Option { + let owner = observer(headers)?; + let entries = self.0.lock().unwrap_or_else(|e| e.into_inner()); + let p = entries.get(&(owner, id.to_owned()))?; + if p.0 + .lock() + .unwrap_or_else(|e| e.into_inner()) + .started + .elapsed() + >= TTL + { + return None; + } + Some(p.snapshot()) + } +} +pub(super) async fn get( + State(state): State, + Path(id): Path, + headers: HeaderMap, +) -> Result, ApiError> { + state + .request_progress + .get(&headers, &id) + .map(Json) + .ok_or_else(|| { + ApiError::not_found("Request diagnostics expired or are not available to this browser") + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn diagnostics_respond_while_request_is_blocked_and_preserve_failure() { + use axum::{Router, body::Body, http::Request, middleware, routing::get}; + use tower::ServiceExt; + let state = super::super::app_state(super::super::ChronicleServerConfig::front_only()); + let entered = Arc::new(tokio::sync::Notify::new()); + let release = Arc::new(tokio::sync::Notify::new()); + let app = Router::new() + .route( + "/api/slow", + get({ + let entered = entered.clone(); + let release = release.clone(); + move || { + let entered = entered.clone(); + let release = release.clone(); + async move { + phase("storage_read"); + entered.notify_one(); + release.notified().await; + Err::<(), _>(ApiError::invalid_request("source format is invalid")) + } + } + }), + ) + .route("/api/requests/{id}", get(super::get)) + .layer(middleware::from_fn_with_state( + state.clone(), + super::super::request_log::warehouse_request_layer, + )) + .with_state(state.clone()); + let token = "0123456789abcdef0123456789abcdef"; + let request = |uri: &str, owner: &str| { + Request::builder() + .uri(uri) + .header(OBSERVER_HEADER, owner) + .header("x-request-id", "live-test") + .body(Body::empty()) + .unwrap() + }; + let task = tokio::spawn( + app.clone() + .oneshot(request("/api/slow?secret=not-recorded", token)), + ); + entered.notified().await; + let response = tokio::time::timeout( + Duration::from_secs(1), + app.clone() + .oneshot(request("/api/requests/live-test", token)), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(response.status(), 200); + let bytes = axum::body::to_bytes(response.into_body(), 65536) + .await + .unwrap(); + assert!(!String::from_utf8_lossy(&bytes).contains("not-recorded")); + let snapshot: Snapshot = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(snapshot.state, "running"); + assert!( + snapshot + .phases + .iter() + .any(|p| p.name == "storage_read" && p.state == "running") + ); + let denied = app + .clone() + .oneshot(request( + "/api/requests/live-test", + "ffffffffffffffffffffffffffffffff", + )) + .await + .unwrap(); + assert_eq!(denied.status(), 404); + release.notify_one(); + assert_eq!(task.await.unwrap().unwrap().status(), 400); + let response = app + .oneshot(request("/api/requests/live-test", token)) + .await + .unwrap(); + let bytes = axum::body::to_bytes(response.into_body(), 65536) + .await + .unwrap(); + let snapshot: Snapshot = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(snapshot.state, "failed"); + assert_eq!(snapshot.error.as_deref(), Some("source format is invalid")); + assert!( + snapshot + .phases + .iter() + .any(|p| p.name == "storage_read" && p.state == "failed") + ); + assert_eq!( + state.request_progress.0.lock().unwrap().len(), + 1, + "polls are never registered" + ); + } + + #[test] + fn progress_is_live_isolated_bounded_and_finishes_on_cancel() { + let registry = Registry::default(); + let mut headers = HeaderMap::new(); + headers.insert( + OBSERVER_HEADER, + "0123456789abcdef0123456789abcdef".parse().unwrap(), + ); + let p = Progress::new( + "id".into(), + "GET".into(), + "/api/explorer/runs".into(), + false, + ); + registry.insert(&headers, p.clone()); + p.phase("authentication"); + p.phase("worker_queue"); + assert!(registry.get(&HeaderMap::new(), "id").is_none()); + let snapshot = registry.get(&headers, "id").unwrap(); + assert_eq!(snapshot.phases[0].state, "completed"); + assert_eq!(snapshot.phases[1].state, "running"); + let mut other = headers.clone(); + other.insert( + OBSERVER_HEADER, + "ffffffffffffffffffffffffffffffff".parse().unwrap(), + ); + assert!(registry.get(&other, "id").is_none()); + drop(CancelOnDrop(p)); + assert_eq!(registry.get(&headers, "id").unwrap().state, "cancelled"); + for i in 0..LIMIT + 5 { + registry.insert( + &headers, + Progress::new(i.to_string(), "GET".into(), "/api/runs".into(), false), + ); + } + assert_eq!(registry.0.lock().unwrap().len(), LIMIT); + let expired = Progress::new("expired".into(), "GET".into(), "/api/runs".into(), false); + registry.insert(&headers, expired.clone()); + expired.0.lock().unwrap().started = Instant::now() - TTL; + assert!(registry.get(&headers, "expired").is_none()); + } +} diff --git a/crates/persisting-pchronicle-cli/src/server/tests.rs b/crates/persisting-pchronicle-cli/src/server/tests.rs index 1b0c9641e..68bdbb695 100644 --- a/crates/persisting-pchronicle-cli/src/server/tests.rs +++ b/crates/persisting-pchronicle-cli/src/server/tests.rs @@ -343,7 +343,8 @@ async fn middleware_echoes_request_id_on_json_errors() { } let app = axum::Router::new() .route("/api/boom", axum::routing::get(boom)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let response = app @@ -387,7 +388,8 @@ async fn four_xx_warn_includes_root_cause_when_chain_is_deeper() { } let app = axum::Router::new() .route("/api/boom", axum::routing::get(boom)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let response = app @@ -451,7 +453,8 @@ async fn middleware_rejects_illegal_incoming_id() { } let app = axum::Router::new() .route("/api/boom", axum::routing::get(boom)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let response = app @@ -489,7 +492,8 @@ async fn middleware_info_logs_static_assets() { } let app = axum::Router::new() .route("/assets/app.css", axum::routing::get(missing)) - .layer(axum::middleware::from_fn( + .layer(axum::middleware::from_fn_with_state( + app_state(ChronicleServerConfig::front_only()), crate::server::request_log::warehouse_request_layer, )); let _ = app @@ -2670,6 +2674,45 @@ async fn exact_runs_request_does_not_wait_for_global_catalog() -> anyhow::Result Ok(()) } +#[tokio::test] +async fn exact_trajectory_endpoints_do_not_wait_for_global_catalog() -> anyhow::Result<()> { + let root = tempfile::tempdir()?; + let store = + persisting_pchronicle::storage::StorylineLanceStore::open(root.path().join("nested/story")) + .await?; + store + .replace_storyline(&storyline_document("session-a", "run-a")) + .await?; + let state = app_state(ChronicleServerConfig::mounted(vec![DatasetMount::new( + "prod2", + root.path().to_string_lossy(), + )?])?); + let _refresh = state.catalog_refresh.lock().await; + let app = finish_routes(state.clone()); + let coords = "dataset=prod2&file=nested/story&agent_id=storyline&session_id=session-a"; + for endpoint in [ + "explorer/run", + "explorer/turns", + "explorer/turn", + "trajectory-view", + "events", + "storyline", + ] { + let (status, body) = tokio::time::timeout( + Duration::from_secs(5), + get_json(&app, &format!("/api/{endpoint}?{coords}&turn_id=1")), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{endpoint}: {body}"); + if endpoint == "explorer/run" { + assert_eq!(body["event_provenance"], "synthetic_from_storyline"); + assert!(body["turn_count"].as_u64().unwrap() > 0); + } + } + assert!(state.catalog.read().await.is_none()); + Ok(()) +} + #[tokio::test] async fn browse_tree_does_not_build_query_runtime() -> anyhow::Result<()> { use http_body_util::BodyExt; diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index fa5e1991b..5ed8ef274 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -309,6 +309,7 @@ impl BrowseCoordinator { } pub(crate) async fn tree(&self, mount: &DatasetMount, prefix: &str) -> Result { + super::request_progress::phase("browse_cache"); let key = TreeKey::new(mount, prefix)?; tracing::info!( target: "pchronicle.serve", @@ -343,6 +344,7 @@ impl BrowseCoordinator { } // A cold page is user-visible work: attach to the same single-flight // refresh as the background walker and return once this prefix exists. + super::request_progress::phase("directory_wait"); let (reply, wait) = oneshot::channel(); self.enqueue(&key, Some(reply))?; // enqueue may decline during cooldown; it drops the sender in that @@ -353,6 +355,9 @@ impl BrowseCoordinator { return Ok(self.snapshot_with_summary(&key, entry).await); } let refreshing = lock_recover(&self.pending).contains_key(&key); + if let Some(progress) = super::request_progress::current() { + progress.note(if refreshing { "Directory listing is still refreshing in the background; this response contains a partial view." } else { "Directory listing is unavailable. Check storage connectivity and the server log for this request." }); + } let last_error = lock_recover(&self.refresh) .get(&key) .and_then(|s| s.error.clone()); @@ -428,6 +433,7 @@ impl BrowseCoordinator { } async fn snapshot_with_summary(&self, key: &TreeKey, mut entry: IndexEntry) -> BrowseSnapshot { + super::request_progress::phase("manifest_summary"); let current_manifest_key = manifest_key(key, None); let observation = self.manifests.get(¤t_manifest_key).await; // The directory projection supplies names; only local manifest diff --git a/crates/persisting-pchronicle/src/store/catalog/discovery.rs b/crates/persisting-pchronicle/src/store/catalog/discovery.rs index 757db7aec..50f82b10d 100644 --- a/crates/persisting-pchronicle/src/store/catalog/discovery.rs +++ b/crates/persisting-pchronicle/src/store/catalog/discovery.rs @@ -603,10 +603,7 @@ pub(super) async fn discover_candidate_at( "invalid source scope" ); let Some(root) = local_mount_path(&mount.uri) else { - // A missing remote manifest is not proof that a prefix or JSON object - // is absent. Preserve full membership and projection binding here until - // object-store discovery supports a bounded ancestor-aware traversal. - return discover_candidates(mount, options).await; + return discover_object_candidate_at(&mount.uri, file, options).await; }; let root_metadata = fs::metadata(&root).context("inspect scoped Dataset root")?; if file == "." || root_metadata.is_file() { @@ -961,6 +958,64 @@ async fn discover_object_candidates( } } +// Follow only the requested ancestry. Opaque leaves and canonical event bundles +// keep their namespace semantics; unrelated siblings must never gate an exact read. +async fn discover_object_candidate_at( + uri: &str, + file: &str, + mut options: LocalQueryManifestOptions, +) -> Result> { + if file == "." { + return discover_object_candidates(uri, options).await; + } + let store = OpendalStore::from_uri(uri).await?; + let mut budget = DiscoveryBudget::new(options); + let mut current = String::new(); + let mut parts = file.split('/'); + loop { + budget.observe_entry()?; + if current == file && is_json_candidate(Path::new(file)) { + if let Some(entry) = store.stat_file(file).await? { + budget.observe_source()?; + return Ok(vec![Candidate::RemoteFile { + file: file.into(), + store: store.clone(), + meta: RemoteObjectMeta::from(entry), + }]); + } + } + match probe_object_prefix(&store, uri, ¤t, root_source_path(¤t)).await? { + Some(ObjectProbe::Source(candidate)) => { + let sidecar = object_storyline_sidecar(&store, uri, &candidate).await?; + let mut candidates = vec![candidate]; + candidates.extend(sidecar); + // Both sides are needed for canonical projection binding before + // discover_impl filters to the requested source namespace. + for _ in &candidates { + budget.observe_source()?; + } + if !candidates.iter().any(|candidate| { + let source = candidate.source_stub().file; + source == file || source.starts_with(&format!("{file}/")) + }) { + return Ok(Vec::new()); + } + return Ok(candidates); + } + None if current.ends_with(".lance") => return Ok(Vec::new()), + Some(ObjectProbe::Branch) | None => {} + } + let Some(part) = parts.next() else { + options.max_entries = options.max_entries.saturating_sub(budget.entries); + return collect_object_virtual(&store, uri, file, options).await; + }; + if !current.is_empty() { + current.push('/'); + } + current.push_str(part); + } +} + enum ObjectProbe { Source(Candidate), Branch, diff --git a/crates/persisting-pchronicle/src/store/catalog/tests.rs b/crates/persisting-pchronicle/src/store/catalog/tests.rs index 340d0089c..754a36f10 100644 --- a/crates/persisting-pchronicle/src/store/catalog/tests.rs +++ b/crates/persisting-pchronicle/src/store/catalog/tests.rs @@ -1402,6 +1402,10 @@ async fn scoped_remote_discovery_includes_json_and_new_prefix_members() -> Resul store .write_overwrite("nested2/other.json", b"[]".to_vec()) .await?; + // A sibling that cannot be parsed must never be visited by scoped reads. + store + .write_overwrite("nested2/chronicle.manifest", b"invalid = [".to_vec()) + .await?; let discover = |file: &str| { DatasetCatalogSnapshot::discover_scoped( vec![DatasetMount::default(uri.clone()).unwrap()], @@ -1429,6 +1433,37 @@ async fn scoped_remote_discovery_includes_json_and_new_prefix_members() -> Resul Ok(()) } +#[tokio::test] +async fn scoped_remote_candidates_preserve_event_binding_and_opaque_ancestors() -> Result<()> { + let uri = format!( + "shared-memory://scoped-bundle-{}/root", + uuid::Uuid::new_v4().simple() + ); + let store = OpendalStore::from_uri(&uri).await?; + for path in [ + "agent/run/events.lance/_manifest.json", + "agent/run/storyline/CURRENT", + "opaque/CURRENT", + "opaque/interior.json", + ] { + store.write_overwrite(path, b"{}".to_vec()).await?; + } + let mount = DatasetMount::default(uri)?; + let options = CatalogSnapshotOptions::default().manifest; + for file in ["agent/run/events.lance", "agent/run/storyline"] { + let candidates = super::discovery::discover_candidate_at(&mount, file, options).await?; + let files: Vec<_> = candidates.iter().map(|c| c.source_stub().file).collect(); + assert_eq!(files, ["agent/run/events.lance", "agent/run/storyline"]); + } + let candidates = + super::discovery::discover_candidate_at(&mount, "opaque/interior.json", options).await?; + assert!( + candidates.is_empty(), + "opaque dataset contents are not separate sources" + ); + Ok(()) +} + #[tokio::test] async fn scoped_discovery_preserves_single_file_mount_identity() -> Result<()> { let temp = tempfile::tempdir()?; diff --git a/docs/src/en/pchronicle/reference/cli.md b/docs/src/en/pchronicle/reference/cli.md index 693249542..7da3274aa 100644 --- a/docs/src/en/pchronicle/reference/cli.md +++ b/docs/src/en/pchronicle/reference/cli.md @@ -400,3 +400,32 @@ pchronicle serve catalog dataset list --catalog-config FILE registers the URI and optional backend storage credentials without creating or deleting object-store data. `grant` / `revoke` add or remove library names on that user (v1 grants are library membership, not fine-grained permission flags). + + +### Request execution diagnostics + +The **Requests** tab shows the current browser tab session's recent API requests, their request IDs, +HTTP outcomes, execution phases and elapsed time. The indicator above Local / profile +opens the same tab. Select a request or enter its ID to inspect it while it is running. +Worker admission, worker startup and worker execution are separate stages; directory +browsing and scoped Runs queries report their own stages. `pending` stages have not +started; `skipped` stages were not used. Stages describe application operations, not +individual DNS, TCP or S3 requests. A successful partial directory response can still +report a background refresh in its note. + +`GET /api/requests/{request_id}` (also under `/api/v1`) returns the live snapshot: +`request_id`, `method`, `path`, `state`, `elapsed_ms`, `status`, `error`, `note`, +`phases` (`name`, `state`, `elapsed_ms`) and an optional nested `worker` snapshot. +This endpoint runs in the front server, independently of busy query workers. + +Tracking is opt-in: send a cryptographically random 32-character hexadecimal +`x-pchronicle-observer` token with the original request and the diagnostic lookup. +Set `x-request-id` on the original request to look it up before it completes. +The UI supplies both headers automatically. The observer token is a private capability, +not a replacement for catalog credentials; a different token cannot read the record. +The server retains at most 512 records for up to 10 minutes, in memory only; restarting +the server clears them. Unknown, expired or inaccessible IDs return 404. The UI keeps +40 recent entries and polls unfinished requests once per second. Diagnostic connection +failures are shown separately and can be retried. Query strings, bodies and credentials +are not included in the diagnostic record. Internal error details remain in server logs, +correlated by request ID. diff --git a/docs/src/zh/pchronicle/reference/cli.md b/docs/src/zh/pchronicle/reference/cli.md index f9cc6e8f0..662354d55 100644 --- a/docs/src/zh/pchronicle/reference/cli.md +++ b/docs/src/zh/pchronicle/reference/cli.md @@ -576,3 +576,25 @@ pchronicle \ 定位后再写 SQL 见 [发现并查询](../guides/discover-and-query.md),交换见 [导入与导出](../guides/exchange.md),只读服务见 [本地服务 Dataset](../guides/serve.md)。 Snapshot 构造见 [Snapshot 设计](../design/catalog.md)。 + + +### 请求执行诊断 + +**Requests** 标签页展示当前浏览器标签页会话中的近期 API 请求的 request ID、HTTP 结果、执行阶段和耗时。 +左下角 Local / profile 上方的状态提示可打开该页。选择请求或输入 request ID,即可在请求 +尚未结束时查看进度。Worker 排队、启动和执行分别计时;目录浏览与 Runs 查询报告各自的 +执行阶段。`pending` 表示尚未开始,`skipped` 表示未使用该阶段。阶段描述应用操作,并非 +逐条 DNS、TCP 或 S3 请求。目录返回 200 但仍在后台刷新时,会显示相应提示。 + +`GET /api/requests/{request_id}`(也支持 `/api/v1`)返回实时快照: +`request_id`、`method`、`path`、`state`、`elapsed_ms`、`status`、`error`、`note`、 +`phases`(`name`、`state`、`elapsed_ms`),以及可选的嵌套 `worker` 快照。 +该接口由前置服务直接处理,不等待业务 worker。 + +追踪需要显式开启:原请求和诊断查询必须携带同一个随机生成的 32 位十六进制 +`x-pchronicle-observer` 请求头;原请求可通过 `x-request-id` 指定 ID,以便完成前查询。 +UI 自动生成这两个请求头。Observer token 是私有查询凭证,不替代 catalog 身份认证, +其他 token 无法访问该记录。服务端最多保留 512 条记录、最长 10 分钟,仅保存在内存中, +重启即清空。记录不存在、过期或无权访问时均返回 404。UI 保留最近 40 条,每秒查询未完成 +请求;诊断连接失败会单独提示,可手动重试。记录不包含查询参数、请求体或凭据;内部错误 +详情仍需通过 request ID 在服务端日志中定位。 diff --git a/pchronicle-web/Cargo.toml b/pchronicle-web/Cargo.toml index 8b43b8025..e092db4fd 100644 --- a/pchronicle-web/Cargo.toml +++ b/pchronicle-web/Cargo.toml @@ -18,7 +18,7 @@ time = { version = "=0.3.55", features = ["formatting", "parsing"] } urlencoding = "2" wasm-bindgen = "0.2" wasm-bindgen-futures = "0.4" -web-sys = { version = "0.3", features = ["Clipboard", "Document", "DomRect", "Element", "EventTarget", "History", "HtmlElement", "HtmlTextAreaElement", "KeyboardEvent", "Location", "Navigator", "Storage", "UrlSearchParams", "Window"] } +web-sys = { version = "0.3", features = ["Crypto", "Clipboard", "Document", "DomRect", "Element", "EventTarget", "History", "HtmlElement", "HtmlTextAreaElement", "KeyboardEvent", "Location", "Navigator", "Storage", "UrlSearchParams", "Window"] } web-time = "=1.1.0" [features] diff --git a/pchronicle-web/assets/workbench.css b/pchronicle-web/assets/workbench.css index 45f22ce1e..e7d53e075 100644 --- a/pchronicle-web/assets/workbench.css +++ b/pchronicle-web/assets/workbench.css @@ -150,3 +150,8 @@ .pc2-shell .rail-secondary .rail-button{color:#b8cbe4} @media(max-width:850px){.pc2-shell .rail-secondary{padding-top:0;border-top:0}} .pc2-form select{width:100%;height:42px;padding:0 12px;border:1px solid #cbd5e1;border-radius:9px;background:#fff;color:#344054;font:inherit}.pc2-identity-list{display:flex;flex-direction:column;gap:8px}.pc2-identity-list .button{width:100%;text-align:left}.pc2-settings .button.danger{border-color:#fecaca;color:#b42318;background:#fff}.pc2-settings .button.danger:hover{background:#fff1f0} + +/* Request diagnostics uses the existing workspace rail and scroll surface. */ +.request-indicator{display:flex;align-items:center;gap:7px;width:100%;padding:10px 14px;border:0;border-radius:8px;background:#ffffff0b;color:#cbd5e1;text-align:left;font-size:12px;cursor:pointer}.request-indicator .spinner{margin:0;width:12px;height:12px;flex:none}.requests-panel{padding:24px;overflow:auto;height:100%;box-sizing:border-box}.requests-panel header p{color:var(--muted)}.requests-panel form{display:flex;gap:8px;margin:16px 0}.requests-panel input{min-width:260px;padding:8px;border:1px solid var(--border);border-radius:6px}.requests-grid{display:grid;grid-template-columns:minmax(240px,1fr) minmax(400px,2fr);gap:20px}.requests-grid aside{max-height:65vh;overflow:auto}.request-row{display:flex;flex-direction:column;gap:6px;width:100%;padding:12px;border:1px solid var(--border);background:white;text-align:left;cursor:pointer;overflow-wrap:anywhere}.request-row span{font-size:11px;color:var(--muted)}.request-row.active{background:#eff6ff;border-color:#2563eb}.requests-grid article{min-width:0;padding:20px;border:1px solid var(--border);border-radius:10px;background:white;overflow-wrap:anywhere}.request-phases{width:100%;border-collapse:collapse;font-size:13px}.request-phases caption{text-align:left;font-weight:600;padding:14px 0}.request-phases td,.request-phases th{text-align:left;padding:10px 6px;border-bottom:1px solid var(--border)}.request-error{color:#b42318}.request-indicator:focus-visible,.request-row:focus-visible{outline:2px solid #60a5fa;outline-offset:2px}@media(max-width:850px){.requests-grid{grid-template-columns:1fr}.requests-grid aside{max-height:220px}} + +.request-indicator{flex-wrap:wrap;box-sizing:border-box}.request-indicator small{width:100%;font-size:10px;overflow-wrap:anywhere}.request-row strong{font-size:13px}.requests-panel input:focus-visible{outline:2px solid #2563eb;outline-offset:2px}@media(max-width:850px){.request-indicator{font-size:9px;padding:8px 2px;overflow-wrap:anywhere}.request-indicator small{display:none}.requests-panel input{min-width:0;width:100%}} diff --git a/pchronicle-web/src/api.rs b/pchronicle-web/src/api.rs index dc50a679f..a443d4824 100644 --- a/pchronicle-web/src/api.rs +++ b/pchronicle-web/src/api.rs @@ -9,6 +9,7 @@ use crate::model::{ PhysicalSource, QueryCatalog, QueryEvidence, RunAnalysis, RunPage, RunSummary, TurnDetail, TurnPage, }; +use crate::requests::TrackedSend; use gloo_net::http::{Request, RequestBuilder, Response}; use serde::de::DeserializeOwned; use serde_json::json; @@ -164,7 +165,7 @@ pub async fn explorer_runs( urlencoding::encode(path), urlencoding::encode(file), ); - json_checked(with_catalog_headers(Request::get(&url)).send().await).await + json_checked(with_catalog_headers(Request::get(&url)).send_tracked().await).await } pub async fn explorer_tree(dataset: &str, prefix: &str) -> Result { @@ -173,7 +174,7 @@ pub async fn explorer_tree(dataset: &str, prefix: &str) -> Result Result { json_checked( with_catalog_headers(Request::get(&format!("/api/explorer/run?{}", run.query()))) - .send() + .send_tracked() .await, ) .await @@ -203,7 +204,7 @@ pub async fn compact_record(run: &RunSummary) -> Result Result Result { @@ -225,7 +226,7 @@ pub async fn turn_detail(run: &RunSummary, turn_id: i64) -> Result Result { json_checked( with_catalog_headers(Request::get("/api/query/tables?ui=true")) - .send() + .send_tracked() .await, ) .await } pub async fn ui_config() -> Result { - json_checked(Request::get("/api/ui").send().await).await + json_checked(Request::get("/api/ui").send_tracked().await).await } pub async fn refresh_catalog() -> Result<(), ApiFailure> { send_checked( with_catalog_headers(Request::post("/api/catalog")) - .send() + .send_tracked() .await, ) .await?; @@ -303,7 +304,7 @@ pub async fn refresh_catalog() -> Result<(), ApiFailure> { pub async fn physical_sources() -> Result, ApiFailure> { json_checked( with_catalog_headers(Request::get("/api/physical/sources")) - .send() + .send_tracked() .await, ) .await @@ -315,7 +316,7 @@ pub async fn physical_layout(dataset: &str, file: &str) -> Result, + pub error: Option, + pub note: Option, + pub phases: Vec, + pub worker: Option>, +} +#[derive(Clone, PartialEq)] +pub struct Entry { + pub id: String, + pub method: String, + pub path: String, + pub transport: String, + pub started: Instant, + pub snapshot: Option, + pub diagnostic_error: Option, +} +pub static REQUESTS: GlobalSignal> = Signal::global(Vec::new); + +fn random_id() -> Option { + let mut bytes = [0u8; 16]; + web_sys::window()? + .crypto() + .ok()? + .get_random_values_with_u8_array(&mut bytes) + .ok()?; + Some(bytes.iter().map(|b| format!("{b:02x}")).collect()) +} +fn observer() -> Option { + let storage = web_sys::window()?.session_storage().ok()??; + if let Ok(Some(token)) = storage.get_item("pchronicle.request_observer") { + return Some(token); + } + let token = random_id()?; + storage + .set_item("pchronicle.request_observer", &token) + .ok()?; + Some(token) +} +fn update(id: &str, apply: impl FnOnce(&mut Entry)) { + if let Some(entry) = REQUESTS.write().iter_mut().find(|e| e.id == id) { + apply(entry); + } +} +struct CancelOnDrop(String); +impl Drop for CancelOnDrop { + fn drop(&mut self) { + update(&self.0, |e| { + if e.transport == "running" { + e.transport = "cancelled".into(); + } + }); + } +} + +pub trait TrackedSend { + async fn send_tracked(self) -> Result; +} +impl TrackedSend for RequestBuilder { + async fn send_tracked(self) -> Result { + self.build()?.send_tracked().await + } +} +impl TrackedSend for Request { + async fn send_tracked(self) -> Result { + let (Some(token), Some(id)) = (observer(), random_id()) else { + return self.send().await; + }; + self.headers().set("x-request-id", &id); + self.headers().set(OBSERVER, &token); + let url = self.url(); + let path = format!( + "/api/{}", + url.split("/api/") + .nth(1) + .unwrap_or("") + .split('?') + .next() + .unwrap_or("") + ); + { + let mut entries = REQUESTS.write(); + // Keep failures available while routine catalog polling continues. + if entries.len() >= LIMIT { + let index = entries + .iter() + .position(|e| e.transport == "completed") + .unwrap_or(0); + entries.remove(index); + } + entries.push(Entry { + id: id.clone(), + method: self.method().to_string(), + path, + transport: "running".into(), + started: Instant::now(), + snapshot: None, + diagnostic_error: None, + }); + } + let _cancel = CancelOnDrop(id.clone()); + let result = self.send().await; + update(&id, |e| { + e.transport = match &result { + Ok(r) if r.ok() => "completed", + _ => "failed", + } + .into(); + if let Err(error) = &result { + e.diagnostic_error = Some(format!("Could not reach the server: {error}")); + } + }); + result + } +} + +pub fn use_request_polling() { + use_future(|| async { + loop { + let pending = REQUESTS + .peek() + .iter() + .filter(|e| { + e.diagnostic_error.is_none() + && e.snapshot.as_ref().is_none_or(|s| s.state == "running") + }) + .map(|e| (e.id.clone(), e.started)) + .collect::>(); + if let Some(token) = observer() { + let queries=pending.into_iter().map(|(id,started)| { let token=token.clone(); async move { + let fetch=async { + let response=Request::get(&format!("/api/requests/{id}")).header(OBSERVER,&token).send().await.map_err(|_|"Diagnostics connection failed".to_string())?; + if response.status()==404 && started.elapsed().as_secs()<10 { return Ok(None); } + if !response.ok() { return Err("Diagnostics expired or are unavailable; the original request may still be running".into()); } + response.json::().await.map(Some).map_err(|_|"Invalid diagnostics response".into()) + }; + let result= match futures_util::future::select(Box::pin(fetch),Box::pin(TimeoutFuture::new(3000))).await { + futures_util::future::Either::Left((r,_))=>r, + _=>Err("Diagnostics timed out; server status is unknown".into()), + }; + update(&id,|e|match result { Ok(Some(s))=>{ + if e.path == "Lookup" { e.method=s.method.clone(); e.path=s.path.clone(); } + e.snapshot=Some(s); + }, Ok(None)=>{}, Err(error)=>e.diagnostic_error=Some(error) }); + }}); + futures_util::future::join_all(queries).await; + } + TimeoutFuture::new(1000).await; + } + }); +} +fn label(name: &str) -> &str { + match name { + "authentication" => "Check catalog identity", + "worker_queue" => "Wait for worker", + "worker_start" => "Start worker", + "worker_execution" => "Execute in worker", + "execution" => "Accept request", + "browse_cache" => "Read directory cache", + "manifest_summary" => "Read local manifest summaries", + "directory_wait" => "Wait for directory listing", + "query_queue" => "Wait for query slot / shared result", + "source_metadata" => "Resolve source metadata", + "storage_read" => "Read storage", + "query" => "Build / execute query", + "catalog_wait" => "Wait for catalog refresh", + "response" => "Prepare response", + _ => name, + } +} +#[component] +pub fn RequestIndicator(on_open: EventHandler<()>) -> Element { + let entries = REQUESTS.read(); + let running = entries.iter().filter(|e| e.transport == "running" || e.snapshot.as_ref().is_some_and(|s|s.state=="running")).count(); + let failed = entries + .iter() + .filter(|e| { + e.transport == "failed" || e.snapshot.as_ref().is_some_and(|s| s.state == "failed") + }) + .count(); + let current = entries + .iter() + .rev() + .find(|e| e.transport == "running" || e.snapshot.as_ref().is_some_and(|s|s.state=="running")) + .and_then(|e| e.snapshot.as_ref()) + .and_then(|s| { + let s = s.worker.as_deref().unwrap_or(s); + s.phases + .iter() + .find(|p| p.state == "running") + .map(|p| format!("{} · {} ms", label(&p.name), p.elapsed_ms)) + }); + rsx! { button { class:"request-indicator", onclick:move |_|on_open.call(()), + if running>0 { span { class:"spinner" } "{running} running" if let Some(stage)=current { small { "{stage}" } } } + else if failed>0 { "{failed} failed · View requests" } + else { "Requests · Idle" } + } } +} +#[component] +fn Phases(snapshot: Snapshot) -> Element { + rsx! { + if let Some(note)=&snapshot.note { p { role:"status", "{note}" } } + table { class:"request-phases", caption { "Execution stages" } + thead { tr { th { "Stage" } th { "Status" } th { "Elapsed" } } } + tbody { for phase in &snapshot.phases { tr { key:"{phase.name}", + td { "{label(&phase.name)}" } td { "{phase.state}" } td { "{phase.elapsed_ms} ms" } + } } } + } + if let Some(worker)=snapshot.worker { h3 { "Worker execution" } Phases { snapshot:*worker } } + } +} +#[component] +pub fn RequestsPanel() -> Element { + let mut selected = use_signal(String::new); + let mut lookup = use_signal(String::new); + let entries = REQUESTS.read().clone(); + let active = if selected().is_empty() { + entries.last() + } else { + entries.iter().find(|e| e.id == selected()) + } + .cloned(); + rsx! { section { class:"requests-panel", + header { h1 { "Requests" } p { "Inspect this browser’s recent requests, execution stages and failures. Server history is retained for up to 10 minutes." } + form { onsubmit:move |event| { + event.prevent_default(); + let id=lookup().trim().to_owned(); + if id.is_empty() || id.len()>64 || !id.bytes().all(|b|b.is_ascii_alphanumeric() || b==b'-' || b==b'_') { return; } + selected.set(id.clone()); + let mut entries=REQUESTS.write(); + if let Some(entry)=entries.iter_mut().find(|e|e.id==id) { entry.diagnostic_error=None; } + else { + if entries.len()>=LIMIT { entries.remove(0); } + entries.push(Entry { id,method:"GET".into(),path:"Lookup".into(),transport:"unknown".into(),started:Instant::now()-std::time::Duration::from_secs(10),snapshot:None,diagnostic_error:None }); + } + }, + input { aria_label:"Request ID", placeholder:"Find by request ID", value:"{lookup}", oninput:move |e|lookup.set(e.value()) } + button { class:"button", r#type:"submit", "Find" } + } + } + if !selected().is_empty() && !entries.iter().any(|e|e.id==selected()) { p { role:"status", "This request is not in this browser’s recent history." } } + div { class:"requests-grid", + aside { aria_label:"Recent requests", + for entry in entries.iter().rev() { button { class:if active.as_ref().is_some_and(|e|e.id==entry.id) {"request-row active"} else {"request-row"}, + onclick:{let id=entry.id.clone();move |_|selected.set(id.clone())}, + strong { "{entry.method} {entry.path}" } span { "{entry.transport} · {entry.id}" } + } } + } + article { + if let Some(entry)=active { + h2 { "{entry.method} {entry.path}" } code { "{entry.id}" } + p { "Browser request: {entry.transport}" } + if let Some(error)=entry.diagnostic_error { p { class:"request-error", role:"alert", "{error}" } + button { class:"button", onclick:{let id=entry.id.clone();move |_|update(&id,|e|e.diagnostic_error=None)}, "Retry diagnostics" } + } + if let Some(snapshot)=entry.snapshot { + p { role:"status", "Server: {snapshot.state} · {snapshot.elapsed_ms} ms" } + if let Some(status)=snapshot.status { p { "HTTP {status}" } } + if let Some(ref error)=snapshot.error { p { class:"request-error", role:"alert", "{error}" } } + Phases { snapshot } + } else { p { role:"status", "Waiting for server diagnostics…" } } + } else { p { "Browse a dataset or open Runs to inspect a request." } } + } + } + } } +} diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 80a669189..8f4d86662 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -88,6 +88,7 @@ fn page_from_query(page: Option<&str>, has_run: bool) -> &'static str { Some("runs") => "runs", Some("physical") => "physical", Some("catalog") => "catalog", + Some("requests") => "requests", Some("detail") => "detail", _ => "home", } @@ -207,6 +208,7 @@ fn navigate_app_back(app_history_depth: i32) -> bool { } pub fn App() -> Element { + crate::requests::use_request_polling(); let initial_agent = url_param("agent_id"); let initial_session = url_param("session_id"); let initial_root = url_param("root_session_id"); @@ -304,6 +306,8 @@ pub fn App() -> Element { let mut expanded_turn_id = use_signal(|| url_param("turn").and_then(|value| value.parse::().ok())); let detail_loading = use_signal(|| false); + let detail_failed = use_signal(|| false); + let detail_generation = use_signal(|| 0u64); let turn_loading = use_signal(|| false); let mut detail_mode = use_signal(|| url_param("workspace").unwrap_or_else(|| "trace".into())); let mut trace_mode = use_signal(|| { @@ -391,6 +395,8 @@ pub fn App() -> Element { turn_search, compact_record, detail_loading, + detail_failed, + detail_generation, error, ); } @@ -557,11 +563,13 @@ pub fn App() -> Element { RailButton { active: page() == "tools", icon: "analysis", label: ANALYSIS, onclick: move |_| page.set("tools".into()) } RailButton { active: page() == "physical", icon: "storage", label: STORAGE, onclick: move |_| page.set("physical".into()) } } + RailButton { active: page() == "requests", icon: "analysis", label: "Requests", onclick: move |_| page.set("requests".into()) } div { class: "rail-spacer" } div { class: "rail-secondary", aria_label: "Assistant and settings", button { class: if copilot_open() { "rail-button active" } else { "rail-button" }, aria_label: "Toggle Assistant", aria_expanded: copilot_open(), onclick: move |_| copilot_open.set(!copilot_open()), WorkspaceIcon { name: "assistant" } span { {ASSISTANT} } } button { class: if settings_open() { "rail-button active" } else { "rail-button" }, aria_label: "Settings", onclick: move |_| settings_open.set(true), WorkspaceIcon { name: "keys" } span { "Keys" } } } + crate::requests::RequestIndicator { on_open: move |_| page.set("requests".into()) } { let identity = catalog_auth::load(); let configured = identity.is_configured(); @@ -587,6 +595,7 @@ pub fn App() -> Element { } } match page().as_str() { + "requests" => rsx! { crate::requests::RequestsPanel {} }, "catalog" => rsx! { CatalogExplorer { tree: catalog_tree(), @@ -777,6 +786,17 @@ pub fn App() -> Element { page.set("tools".into()); }, } + } else if detail_failed() && !detail_loading() { + div { class: "pc2-loading", role: "alert", + strong { "Could not load run details" } + p { "Open Requests to inspect the failed stage, or retry this run." } + button { class: "button", onclick: move |_| { + if let Some(run)=selected_run.peek().clone() { + load_workspace(run,turn_query(),source(),analysis,turns,turn_search,compact_record,detail_loading,detail_failed,detail_generation,error); + } + }, "Retry" } + button { class: "button", onclick: move |_| page.set("requests".into()), "View requests" } + } } else { LoadingWorkspace { label: "Loading run details…" } } } } } @@ -1189,33 +1209,56 @@ fn load_workspace( mut turn_search: Signal, mut compact_record: Signal>, mut loading: Signal, + mut failed: Signal, + mut generation: Signal, mut error: Signal>, ) { + // Untracked reads avoid subscribing the caller's effect to its own writes. + let requested = *generation.peek() + 1; + generation.set(requested); compact_record.set(None); + failed.set(false); loading.set(true); - spawn({ - let run = run.clone(); - async move { - let (next_analysis, next_turns) = - futures_util::join!(api::run_analysis(&run), api::turns(&run, &query, &source),); - match (next_analysis, next_turns) { - (Ok(next_analysis), Ok(next_turns)) => { - if next_analysis.run.is_compact_jsonl() { - match api::compact_record(&next_analysis.run).await { - Ok(value) => compact_record.set(Some(value)), - Err(failure) => error.set(Some(workspace_notice(&failure))), - } - } - analysis.set(Some(next_analysis)); - turns.set(next_turns.records); - turn_search.set(next_turns.search); - } - (Err(failure), _) | (_, Err(failure)) => { - error.set(Some(workspace_notice(&failure))); - } + spawn(async move { + let work = async { + let (next_analysis, next_turns) = futures_util::try_join!( + api::run_analysis(&run), + api::turns(&run, &query, &source) + )?; + let record = if next_analysis.run.is_compact_jsonl() { + Some(api::compact_record(&next_analysis.run).await?) + } else { + None + }; + Ok::<_, api::ApiFailure>((next_analysis, next_turns, record)) + }; + let result = match futures_util::future::select( + Box::pin(work), + Box::pin(TimeoutFuture::new(65_000)), + ) + .await + { + futures_util::future::Either::Left((result, _)) => result, + _ => Err(api::ApiFailure::network( + "Run details timed out. Open Requests to inspect server progress, then retry.", + )), + }; + if *generation.peek() != requested { + return; + } + match result { + Ok((next_analysis, next_turns, record)) => { + analysis.set(Some(next_analysis)); + turns.set(next_turns.records); + turn_search.set(next_turns.search); + compact_record.set(record); + } + Err(failure) => { + failed.set(true); + error.set(Some(workspace_notice(&failure))); } - loading.set(false); } + loading.set(false); }); } From f1d7b053c7a7e4049c5f9fa770e2c641ec9d7f08 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 20:04:07 +0800 Subject: [PATCH 14/22] fix(workspace): optimize run loading behavior in detail view - Updated the logic in the `App` function to prevent unnecessary refetching of runs when navigating to the detail page, ensuring that the existing run data is retained. - Introduced a condition to check if the detail page is accessed without any loaded runs, enhancing performance and user experience by avoiding redundant data requests. This commit aims to improve the efficiency of run data management in the workspace component. --- pchronicle-web/src/workspace.rs | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 8f4d86662..1d53e58e1 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -329,7 +329,12 @@ pub fn App() -> Element { let mut llm_config = use_signal(llm::load_config); use_effect(move || { - if !matches!(page().as_str(), "runs" | "detail") { + // Selecting a run navigates to `detail` while retaining the already + // loaded run page. Do not refetch the list just because the detail + // pane changed; only bootstrap it for a direct detail URL with no + // existing page. + let on_detail_without_runs = page() == "detail" && runs.peek().is_none(); + if page() != "runs" && !on_detail_without_runs { return; } load_runs( From 06b2012a62887d3cd69bd4f66333e6842a68ac37 Mon Sep 17 00:00:00 2001 From: Reiase Date: Wed, 16 Sep 2026 20:40:19 +0800 Subject: [PATCH 15/22] feat(worker_pool): enhance worker management and idle cleanup - Refactored the worker pool to implement a more efficient slot management system, allowing for up to 4 workers per scope and 8 workers server-wide. - Introduced a cleanup mechanism that reclaims idle workers after 120 seconds, with a reaper task running every 30 seconds to maintain optimal resource usage. - Updated the lease mechanism to allow concurrent requests in the same scope to utilize different execution workers, improving performance and resource allocation. - Enhanced documentation to reflect the new worker management capabilities and timeout settings. This commit aims to optimize worker utilization and improve the responsiveness of the worker pool in the pChronicle application. --- .../src/server/catalog_worker.rs | 252 +++++++++++++++--- .../tests/catalog_worker_contract.rs | 51 +++- docs/src/en/pchronicle/reference/cli.md | 13 +- docs/src/zh/pchronicle/reference/cli.md | 8 +- pchronicle-web/src/api.rs | 42 ++- pchronicle-web/src/main.rs | 2 +- pchronicle-web/src/requests.rs | 11 +- 7 files changed, 319 insertions(+), 60 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index 2cffed552..a8c1ac1a9 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -1,18 +1,18 @@ //! Exec workers have an immutable authenticated scope. No storage client or //! runtime is inherited from the listening process. use std::{ - collections::HashMap, io::{Read, Write}, path::PathBuf, process::Stdio, - sync::Arc, + sync::{Arc, Once}, time::Duration, }; use anyhow::{Context, Result}; use serde::{Deserialize, Serialize, de::DeserializeOwned}; use tokio::io::{AsyncReadExt, AsyncWriteExt}; -use tokio::sync::{Mutex, Semaphore}; +use tokio::sync::{Mutex, Notify, OwnedMutexGuard, Semaphore}; +use tokio::time::Instant; use tower::ServiceExt; use super::{ @@ -22,6 +22,9 @@ use super::{ const MAX_WORKERS: usize = 8; const MAX_REQUESTS: usize = 32; +const MAX_WORKERS_PER_SCOPE: usize = 4; +const WORKER_IDLE_TIMEOUT: Duration = Duration::from_secs(120); +const REAP_INTERVAL: Duration = Duration::from_secs(30); const FRAME_LIMIT: usize = 40 * 1024 * 1024; const REQUEST_TIMEOUT: Duration = Duration::from_secs(60); @@ -65,18 +68,130 @@ struct Bootstrap { mounts: Vec, } -type Slot = Arc>>; +struct SlotState { + worker: Option, + idle_since: Instant, +} + +struct Slot { + scope: String, + state: Arc>, +} + +#[derive(Default)] +struct PoolState { + slots: Mutex>, + available: Arc, +} + +// A reserved slot is never visible as idle, including while its child starts. +// Drop wakes all scopes: a released slot can satisfy a different scope by +// eviction even if the first waiter has reached its per-scope limit. +struct Lease { + guard: Option>, + available: Arc, +} + +impl Drop for Lease { + fn drop(&mut self) { + if let Some(mut guard) = self.guard.take() { + guard.idle_since = Instant::now(); + drop(guard); + } + self.available.notify_waiters(); + } +} + +impl PoolState { + async fn try_lease(&self, scope: &str) -> Result, ApiError> { + let mut slots = self.slots.lock().await; + for slot in slots.iter().filter(|slot| slot.scope == scope) { + if let Ok(guard) = slot.state.clone().try_lock_owned() { + return Ok(Some(Lease { + guard: Some(guard), + available: self.available.clone(), + })); + } + } + if slots.iter().filter(|slot| slot.scope == scope).count() >= MAX_WORKERS_PER_SCOPE { + return Ok(None); + } + if slots.len() >= MAX_WORKERS { + // Never enqueue on a busy slot. Reclaim only an idle child from + // another scope, and wait for its exit before reusing its capacity. + let idle = slots + .iter() + .enumerate() + .filter_map(|(index, slot)| { + slot.state + .clone() + .try_lock_owned() + .ok() + .map(|guard| (index, guard)) + }) + .min_by_key(|(_, guard)| guard.idle_since); + let Some((index, mut guard)) = idle else { + return Ok(None); + }; + if let Some(worker) = guard.worker.as_mut() { + worker + .child + .kill() + .await + .map_err(|error| worker_error(error.into()))?; + } + slots.remove(index); + } + let state = Arc::new(Mutex::new(SlotState { + worker: None, + idle_since: Instant::now(), + })); + let guard = state + .clone() + .try_lock_owned() + .expect("new worker slot is idle"); + slots.push(Slot { + scope: scope.to_owned(), + state, + }); + Ok(Some(Lease { + guard: Some(guard), + available: self.available.clone(), + })) + } + + async fn reap_idle(&self) { + let mut slots = self.slots.lock().await; + for index in (0..slots.len()).rev() { + let Ok(mut guard) = slots[index].state.clone().try_lock_owned() else { + continue; + }; + if guard.idle_since.elapsed() < WORKER_IDLE_TIMEOUT { + continue; + } + if let Some(worker) = guard.worker.as_mut() { + if worker.child.kill().await.is_err() { + continue; + } + } + slots.remove(index); + } + self.available.notify_waiters(); + } +} pub(super) struct WorkerPool { - slots: Mutex>, + state: Arc, requests: Semaphore, + reaper: Once, } impl Default for WorkerPool { fn default() -> Self { Self { - slots: Mutex::new(HashMap::new()), + state: Arc::new(PoolState::default()), requests: Semaphore::new(MAX_REQUESTS), + reaper: Once::new(), } } } @@ -88,30 +203,30 @@ impl WorkerPool { .map_err(|_| ApiError::unavailable().with_stage(ExecutionStage::Admission)) } - async fn slot(&self, scope: &str) -> Result { - let mut slots = self.slots.lock().await; - if let Some(slot) = slots.get(scope) { - return Ok(slot.clone()); - } - if slots.len() >= MAX_WORKERS { - // Only evict a worker with no in-flight or queued request. Wait for - // its exit before spawning a replacement, keeping the process cap. - let idle = slots - .iter() - .find(|(_, slot)| Arc::strong_count(slot) == 1) - .map(|(key, _)| key.clone()); - let Some(idle) = idle else { - return Err(ApiError::unavailable().with_stage(ExecutionStage::Admission)); - }; - if let Some(slot) = slots.remove(&idle) - && let Some(mut worker) = slot.lock().await.take() - { - let _ = worker.child.kill().await; + async fn lease(&self, scope: &str) -> Result { + self.reaper.call_once(|| { + let state = Arc::downgrade(&self.state); + tokio::spawn(async move { + loop { + tokio::time::sleep(REAP_INTERVAL).await; + let Some(state) = state.upgrade() else { + break; + }; + state.reap_idle().await; + } + }); + }); + loop { + // Register before checking capacity so completion cannot be missed + // between a failed checkout and going to sleep. + let ready = self.state.available.notified(); + tokio::pin!(ready); + ready.as_mut().enable(); + if let Some(lease) = self.state.try_lease(scope).await? { + return Ok(lease); } + ready.await; } - let slot = Arc::new(Mutex::new(None)); - slots.insert(scope.to_owned(), slot.clone()); - Ok(slot) } pub(super) async fn execute( @@ -122,19 +237,19 @@ impl WorkerPool { ) -> Result { tokio::time::timeout(REQUEST_TIMEOUT, async { super::request_progress::phase("worker_queue"); - let slot = self.slot(&scope).await?; - let mut guard = slot.lock().await; + let mut lease = self.lease(&scope).await?; + let guard = lease.guard.as_mut().expect("reserved worker slot"); // Ownership stays in this future during IPC: cancellation, timeout // or a partial frame drops/kills it rather than reusing dirty pipes. super::request_progress::phase("worker_start"); - let mut worker = match guard.take() { + let mut worker = match guard.worker.take() { Some(worker) => worker, None => Worker::start(&scope, mounts).await.map_err(worker_error)?, }; super::request_progress::phase("worker_execution"); let response = worker.exchange(&request).await.map_err(worker_error)?; let response = response.into_response().map_err(worker_error)?; - *guard = Some(worker); + guard.worker = Some(worker); Ok(response) }) .await @@ -412,6 +527,7 @@ pub(crate) fn run() -> Result<()> { #[cfg(test)] mod tests { use super::*; + use std::collections::HashMap; #[test] fn frames_reject_truncation_and_oversize_and_preserve_boundaries() { @@ -426,21 +542,67 @@ mod tests { } #[tokio::test] - async fn pool_bounds_admission_and_never_evicts_queued_scopes() { + async fn pool_scales_reuses_bounds_and_wakes_waiters() { let pool = WorkerPool::default(); let permits: Vec<_> = (0..MAX_REQUESTS).map(|_| pool.admit().unwrap()).collect(); assert!(pool.admit().is_err()); drop(permits); assert!(pool.admit().is_ok()); - let mut slots = Vec::new(); - for index in 0..MAX_WORKERS { - slots.push(pool.slot(&index.to_string()).await.unwrap()); + + let first = pool.lease("same").await.unwrap(); + let second = pool.lease("same").await.unwrap(); + assert_eq!(pool.state.slots.lock().await.len(), 2); + let first_slot = OwnedMutexGuard::mutex(first.guard.as_ref().unwrap()).clone(); + drop(first); + let reused = pool.lease("same").await.unwrap(); + assert!(Arc::ptr_eq( + &first_slot, + OwnedMutexGuard::mutex(reused.guard.as_ref().unwrap()) + )); + let mut busy = vec![second, reused]; + for _ in busy.len()..MAX_WORKERS_PER_SCOPE { + busy.push(pool.lease("same").await.unwrap()); + } + assert!(pool.state.try_lease("same").await.unwrap().is_none()); + for index in MAX_WORKERS_PER_SCOPE..MAX_WORKERS { + busy.push(pool.lease(&format!("other-{index}")).await.unwrap()); + } + assert!(pool.state.try_lease("overflow").await.unwrap().is_none()); + let waiting = pool.lease("same"); + tokio::pin!(waiting); + assert!( + tokio::time::timeout(Duration::from_millis(10), &mut waiting) + .await + .is_err() + ); + busy.remove(0); + let lease = tokio::time::timeout(Duration::from_secs(1), waiting) + .await + .unwrap() + .unwrap(); + assert_eq!(pool.state.slots.lock().await.len(), MAX_WORKERS); + drop(lease); + let replacement = pool.lease("replacement").await.unwrap(); + assert_eq!(pool.state.slots.lock().await.len(), MAX_WORKERS); + drop(replacement); + } + + #[tokio::test] + async fn reaper_removes_only_idle_expired_slots() { + let pool = WorkerPool::default(); + let mut busy = pool.lease("scope").await.unwrap(); + busy.guard.as_mut().unwrap().idle_since = Instant::now() - WORKER_IDLE_TIMEOUT; + let idle = pool.lease("scope").await.unwrap(); + drop(idle); + { + let slots = pool.state.slots.lock().await; + slots[1].state.lock().await.idle_since = Instant::now() - WORKER_IDLE_TIMEOUT; } - assert!(Arc::ptr_eq(&slots[0], &pool.slot("0").await.unwrap())); - assert!(pool.slot("overflow").await.is_err()); - slots.remove(0); - assert!(pool.slot("replacement").await.is_ok()); - assert_eq!(pool.slots.lock().await.len(), MAX_WORKERS); + pool.state.reap_idle().await; + assert_eq!(pool.state.slots.lock().await.len(), 1); + drop(busy); + pool.state.reap_idle().await; + assert_eq!(pool.state.slots.lock().await.len(), 1); } #[cfg(unix)] @@ -458,13 +620,14 @@ mod tests { let input = child.stdin.take().unwrap(); let output = child.stdout.take().unwrap(); let pool = WorkerPool::default(); - let slot = pool.slot("test").await.unwrap(); - *slot.lock().await = Some(Worker { + let mut lease = pool.lease("test").await.unwrap(); + lease.guard.as_mut().unwrap().worker = Some(Worker { child, input, output, _home: home, }); + drop(lease); let result = tokio::time::timeout( Duration::from_millis(30), pool.execute( @@ -480,7 +643,8 @@ mod tests { ) .await; assert!(result.is_err()); - assert!(slot.lock().await.is_none()); + let lease = pool.lease("test").await.unwrap(); + assert!(lease.guard.as_ref().unwrap().worker.is_none()); tokio::time::timeout(Duration::from_secs(5), async { loop { // Signal zero observes process existence without sending a signal. diff --git a/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs b/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs index add3c50c5..c8a6ccfb9 100644 --- a/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs +++ b/crates/persisting-pchronicle-cli/tests/catalog_worker_contract.rs @@ -96,7 +96,6 @@ permissions = ["read"] .map(|v| v.to_str().unwrap_or("")) .collect::>() .join(","); - assert!(timings.contains("parent_total"), "{timings}"); assert!(timings.contains("total;"), "{timings}"); let body: Value = response.json().await?; assert_eq!(body["children"].as_array().map(Vec::len), Some(1)); @@ -109,6 +108,14 @@ permissions = ["read"] .send() .await?; assert_eq!(query.status(), 200); + let timings = query + .headers() + .get_all("server-timing") + .iter() + .map(|v| v.to_str().unwrap_or("")) + .collect::>() + .join(","); + assert!(timings.contains("parent_total"), "{timings}"); let query: Value = query.json().await?; assert_eq!( query["rows"], @@ -116,6 +123,36 @@ permissions = ["read"] "{query}" ); } + // Concurrent requests in one scope may now use different exec workers. + // Each process must retain the same grants and storage identity. + let mut concurrent = tokio::task::JoinSet::new(); + for user in ["alice", "bob", "alice", "bob"] { + let client = client.clone(); + let endpoint = endpoint.to_owned(); + concurrent.spawn(async move { + let response = client + .post(format!("http://{endpoint}/api/query/evidence")) + .header("x-pchronicle-access-key", format!("{user}-ak")) + .header("x-pchronicle-secret-key", format!("{user}-sk")) + .json( + &json!({"sql":"SELECT session_id FROM runs", "max_rows":10, "max_bytes":4096}), + ) + .send() + .await?; + assert_eq!(response.status(), 200); + let body: Value = response.json().await?; + let expected = if user == "alice" { + "left-session" + } else { + "right-session" + }; + assert_eq!(body["rows"], json!([{"session_id":expected}])); + Ok::<_, anyhow::Error>(()) + }); + } + while let Some(result) = concurrent.join_next().await { + result??; + } let response = client .get(format!("{url}?dataset=right")) .header("x-pchronicle-access-key", "alice-ak") @@ -214,7 +251,17 @@ async fn exec_worker_handles_multiple_frames_and_exits_on_eof() -> Result<()> { json!({"method":"GET", "uri":"/api/health", "headers":[], "body":[]}), ) .await?; - assert_eq!(read_frame(&mut output).await?["status"], 200); + loop { + let event = read_frame(&mut output).await?; + match event["type"].as_str() { + Some("Progress") => continue, + Some("Response") => { + assert_eq!(event["value"]["status"], 200); + break; + } + _ => anyhow::bail!("unexpected worker frame: {event}"), + } + } assert_eq!(child.id(), pid); assert!(child.try_wait()?.is_none()); } diff --git a/docs/src/en/pchronicle/reference/cli.md b/docs/src/en/pchronicle/reference/cli.md index 7da3274aa..6ace554d3 100644 --- a/docs/src/en/pchronicle/reference/cli.md +++ b/docs/src/en/pchronicle/reference/cli.md @@ -341,9 +341,16 @@ an isolated exec worker containing only authorized mounts. It also enables `pchronicle serve catalog dataset add|remove|list` and `issue|grant|revoke` rewrite that file and do not start HTTP; `issue` prints the user secret once. Restart serve after changing libraries, users, or grants. -The pool allows at most 8 workers and 32 admitted requests, with serial execution -per worker, a 60-second execution/queue timeout and a 10-second body-read timeout. -Overload returns 503; timeout or IPC failure discards the worker. Worker and disk +The pool reuses idle workers in the same authenticated scope and grows on demand +when they are busy, up to 4 workers per scope and 8 workers server-wide. At capacity, +requests wait for available capacity rather than a particular busy worker; at most +32 requests are admitted. Each worker still executes serially to keep IPC isolated. +A cleanup task runs every 30 seconds and reclaims workers idle for 120 seconds; +idle workers from other scopes may be reclaimed sooner when global capacity is full. +Queueing, startup and execution share a 60-second timeout; body reads have a +10-second timeout. Overload returns 503. Cancellation during execution or IPC +failure discards that worker; cancelling a queued request does not interrupt +other requests. Worker and disk cache identity includes the user, grants and backend credential version. Caches live under `PCHRONICLE_CACHE_DIR/workers/` or the system pchronicle cache directory. Workers receive backend keys over private IPC before starting runtime threads; diff --git a/docs/src/zh/pchronicle/reference/cli.md b/docs/src/zh/pchronicle/reference/cli.md index 662354d55..050627f31 100644 --- a/docs/src/zh/pchronicle/reference/cli.md +++ b/docs/src/zh/pchronicle/reference/cli.md @@ -443,8 +443,12 @@ pchronicle serve \ 独立 exec worker 读取;不能与位置参数 Dataset、Gateway 或 Control 同时使用。配合 `dataset pin NAME catalog://127.0.0.1:PORT --ak --sk`。 `pchronicle serve catalog dataset add|remove|list` 与 `issue|grant|revoke` 只改该文件、 不启动 HTTP;`issue` 把用户 sk 只打印一次。改 library、用户或授权后必须重启 serve。 -worker 池最多 8 个进程、32 个正在处理或排队的请求;每个 worker 串行处理请求, -计算等待上限为 60 秒,请求体读取上限为 10 秒。过载返回 503,超时或 IPC 失败会淘汰进程。 +worker 池按并发压力创建进程:优先复用同一权限范围的空闲 worker,全部忙碌时按需扩容, +每个范围最多 4 个、整个服务最多 8 个进程。达到上限后等待任意可用容量, +最多接纳 32 个正在处理或排队的请求。每个 worker 内部仍串行处理,IPC 不交叉。 +空闲超过 120 秒的进程由每 30 秒执行一次的清理任务回收;全局容量不足时可提前回收 +其他范围的空闲进程。排队、启动和执行合计上限为 60 秒,请求体读取上限为 10 秒。 +过载返回 503;执行期间超时或 IPC 失败会淘汰对应进程,排队取消不会中断其他请求。 同一用户、授权范围和后端凭证版本复用 worker 及独立磁盘缓存;修改授权后重启生效。 缓存位于 `PCHRONICLE_CACHE_DIR/workers/`,未设置时使用系统 pchronicle 缓存目录。 子进程不继承父进程的 AWS 环境、profile 或用户主目录配置;登录 AK/SK 用于认证, diff --git a/pchronicle-web/src/api.rs b/pchronicle-web/src/api.rs index a443d4824..f5e8c19d3 100644 --- a/pchronicle-web/src/api.rs +++ b/pchronicle-web/src/api.rs @@ -165,7 +165,12 @@ pub async fn explorer_runs( urlencoding::encode(path), urlencoding::encode(file), ); - json_checked(with_catalog_headers(Request::get(&url)).send_tracked().await).await + json_checked( + with_catalog_headers(Request::get(&url)) + .send_tracked() + .await, + ) + .await } pub async fn explorer_tree(dataset: &str, prefix: &str) -> Result { @@ -174,7 +179,12 @@ pub async fn explorer_tree(dataset: &str, prefix: &str) -> Result Result Result { @@ -316,7 +331,12 @@ pub async fn physical_layout(dataset: &str, file: &str) -> Result) -> Element { let current = entries .iter() .rev() - .find(|e| e.transport == "running" || e.snapshot.as_ref().is_some_and(|s|s.state=="running")) + .find(|e| { + e.transport == "running" || e.snapshot.as_ref().is_some_and(|s| s.state == "running") + }) .and_then(|e| e.snapshot.as_ref()) .and_then(|s| { let s = s.worker.as_deref().unwrap_or(s); From 2007cfc6565e8b6a5614a0fe382c2d7c5d329970 Mon Sep 17 00:00:00 2001 From: Reiase Date: Thu, 17 Sep 2026 01:22:40 +0800 Subject: [PATCH 16/22] Fix search previews on run pages Unqualified text search now targets only `message_value`; field selectors carry their own columns, and `#content` is an alias for the message body. Preview SQL partitions matches per visible run so every page row gets evidence, and previews render around the hit rather than truncating the whole field. --- .../src/server/explorer.rs | 10 +- .../src/server/mod.rs | 169 ++++++++++----- .../src/server/tests.rs | 94 ++++++-- .../persisting-pchronicle/src/search/find.rs | 3 +- .../src/search/storyline.rs | 9 +- .../en/pchronicle/reference/query-model.md | 2 +- .../zh/pchronicle/reference/query-model.md | 2 +- pchronicle-web/assets/inline-trace.css | 2 +- pchronicle-web/src/workspace.rs | 205 ++++++++++++------ 9 files changed, 333 insertions(+), 163 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/explorer.rs b/crates/persisting-pchronicle-cli/src/server/explorer.rs index d142fdf81..50c958e58 100644 --- a/crates/persisting-pchronicle-cli/src/server/explorer.rs +++ b/crates/persisting-pchronicle-cli/src/server/explorer.rs @@ -423,7 +423,9 @@ pub(crate) fn run_page_with_fts( }) .collect::>(); let path_index_limit = 2_000usize; - let path_index = if records.len() > path_index_limit { + let path_index = if !needle.is_empty() { + Vec::new() + } else if records.len() > path_index_limit { // Huge compact-jsonl sources must not ship every identity into the // browser path explorer. Keep one representative per file plus a // bounded sample so WASM stays responsive. @@ -482,8 +484,12 @@ pub(crate) fn run_page_with_fts( ); RunExplorerPage { snapshot: page.snapshot, + path_index: if needle.is_empty() { + path_index + } else { + page.records.iter().map(|item| item.run.clone()).collect() + }, records: page.records, - path_index, search, } } diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 1d00bbbe6..9cab7cbf7 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -860,7 +860,6 @@ fn api_query(query: Result, QueryRejection>) -> Result } const EXPLORER_RUN_MATCH_IDENTITY_MAX_ROWS: u64 = 50_000; -const EXPLORER_RUN_MATCH_PREVIEW_LIMIT: u64 = 512; fn explorer_run_identity_sql(dataset: &str, table: &str, predicate: &str) -> String { format!( @@ -871,11 +870,54 @@ fn explorer_run_identity_sql(dataset: &str, table: &str, predicate: &str) -> Str fn explorer_run_preview_sql( dataset: &str, table: &str, - select: &str, + columns: &[&str], predicate: &str, - limit: u64, + runs: &[&RunSummary], ) -> String { - format!("SELECT {select} FROM {dataset}.{table} WHERE ({predicate}) LIMIT {limit}") + let scope = runs + .iter() + .map(|run| { + format!( + "(_file_ = {} AND document_id = {})", + crate::sql_string(&run.file), + crate::sql_string(&run.document_id), + ) + }) + .collect::>() + .join(" OR "); + let columns = columns.join(", "); + let order = if table == "steps" { + "step_id" + } else { + "document_id" + }; + // One matching row per visible run. A global step LIMIT lets one long run + // consume every preview and leaves the rest of the page without evidence. + format!( + "SELECT source_path, document_id, {columns} FROM (SELECT _file_ AS source_path, document_id, {columns}, ROW_NUMBER() OVER (PARTITION BY _file_, document_id ORDER BY {order}) AS match_rank FROM {dataset}.{table} WHERE ({predicate}) AND ({scope})) AS matched WHERE match_rank = 1 LIMIT {}", + runs.len(), + ) +} + +fn explorer_preview_columns( + expression: &persisting_pchronicle::search::FindExpr, +) -> Vec<&'static str> { + let mut predicates = Vec::new(); + crate::collect_text_predicates(expression, &mut predicates); + let mut columns = predicates + .iter() + .flat_map(|item| item.field.columns().iter().copied()) + .collect::>(); + if columns.is_empty() { + columns.extend(if expression.has_step_json() { + &["message_value", "metrics"][..] + } else { + &["task", "prompt", "notes", "agent_name", "agent_model_name"][..] + }); + } + columns.sort_unstable(); + columns.dedup(); + columns } async fn explorer_query_jsonl( @@ -1229,6 +1271,7 @@ async fn explorer_runs_inner( ) .await?; metrics.record("summary_total", started); + let mut preview_plan = None; let (fts_matches, fts_available, search_mode) = if query .q .as_deref() @@ -1265,14 +1308,6 @@ async fn explorer_runs_inner( if dataset_filter.is_some_and(|filter| dataset.mount.name != filter) { continue; } - let select = if table == "steps" { - // Keep all searchable step fields available to the preview - // selector. A COALESCE expression would hide a hit in (for - // example) reasoning_content behind a non-empty message. - "_file_ AS source_path, document_id, message_value, reasoning_content, observation, prompt, model_name" - } else { - "_file_ AS source_path, document_id, task, prompt, notes, agent_name, agent_model_name" - }; let identity_sql = explorer_run_identity_sql(&dataset.mount.name, table, &predicate); let identity_jsonl = explorer_query_jsonl( &runtime.engine, @@ -1301,46 +1336,14 @@ async fn explorer_runs_inner( let identity = format!("{}\u{1f}{}\u{1f}{}", dataset.mount.name, file, document_id); matches.entry(identity).or_insert_with(String::new); } - let preview_sql = explorer_run_preview_sql( - &dataset.mount.name, - table, - select, - &predicate, - EXPLORER_RUN_MATCH_PREVIEW_LIMIT, - ); - let preview_jsonl = explorer_query_jsonl( - &runtime.engine, - &preview_sql, - EXPLORER_RUN_MATCH_PREVIEW_LIMIT, - &request_id, - ) - .await?; - for line in preview_jsonl.lines().filter(|line| !line.trim().is_empty()) { - let row: Value = serde_json::from_str(line).map_err(|error| { - fail( - &request_id, - "explorer_runs", - anyhow::anyhow!("decode run search preview: {error}"), - ) - })?; - let Some(file) = row.get("source_path").and_then(Value::as_str) else { - continue; - }; - let Some(document_id) = row.get("document_id").and_then(Value::as_str) else { - continue; - }; - let identity = format!("{}\u{1f}{}\u{1f}{}", dataset.mount.name, file, document_id); - if !matches.contains_key(&identity) { - continue; - } - let preview = search_preview_from_row(&row, raw, table); - if let Some(existing) = matches.get_mut(&identity) - && existing.is_empty() - { - *existing = preview; - } - } } + preview_plan = Some(( + runtime, + table, + explorer_preview_columns(&expression), + predicate, + raw.to_owned(), + )); let mode = if expression.has_text() && expression.has_json() { "fts+json" } else if expression.has_text() { @@ -1375,7 +1378,7 @@ async fn explorer_runs_inner( } (BTreeMap::new(), fts_available, "none") }; - Ok(Json(explorer::run_page_with_fts( + let mut page = explorer::run_page_with_fts( summaries, &query, &fts_matches, @@ -1384,14 +1387,60 @@ async fn explorer_runs_inner( mode: search_mode, tokenizer: fts_available.then_some("jieba"), }, - ))) + ); + if let Some((runtime, table, columns, predicate, raw)) = preview_plan { + for dataset in runtime.snapshot.datasets() { + let runs = page + .records + .iter() + .filter(|item| item.run.dataset == dataset.mount.name) + .map(|item| &item.run) + .collect::>(); + if runs.is_empty() { + continue; + } + let sql = + explorer_run_preview_sql(&dataset.mount.name, table, &columns, &predicate, &runs); + let jsonl = + explorer_query_jsonl(&runtime.engine, &sql, runs.len() as u64, &request_id).await?; + for line in jsonl.lines().filter(|line| !line.trim().is_empty()) { + let row: Value = serde_json::from_str(line) + .map_err(|error| fail(&request_id, "explorer_runs_preview", error.into()))?; + let file = row.get("source_path").and_then(Value::as_str); + let document = row.get("document_id").and_then(Value::as_str); + if let Some(item) = page.records.iter_mut().find(|item| { + item.run.dataset == dataset.mount.name + && Some(item.run.file.as_str()) == file + && Some(item.run.document_id.as_str()) == document + }) { + item.search_preview = Some(search_preview_from_row(&row, &raw, table)); + } + } + } + } + Ok(Json(page)) } -fn search_preview_text(raw: &str) -> String { - // The API deliberately returns the complete normalized field. The Web - // client owns the viewport-sized excerpt so it can guarantee that the - // matched term remains visible and highlighted. - crate::find_preview_text(raw) +fn search_preview_text(raw: &str, query: &str) -> String { + const MAX_CHARS: usize = 320; + let text = crate::find_preview_text(raw); + let needle = preview_needle(query).to_ascii_lowercase(); + let hit = if needle.is_empty() { + None + } else { + text.to_ascii_lowercase().find(&needle) + }; + let start = hit + .map(|offset| text[..offset].chars().count().saturating_sub(MAX_CHARS / 3)) + .unwrap_or(0); + let mut excerpt: String = text.chars().skip(start).take(MAX_CHARS).collect(); + if start + excerpt.chars().count() < text.chars().count() { + excerpt.push('…'); + } + if start > 0 { + excerpt.insert(0, '…'); + } + excerpt } fn search_preview_from_row(row: &Value, query: &str, table: &str) -> String { @@ -1402,6 +1451,8 @@ fn search_preview_from_row(row: &Value, query: &str, table: &str) -> String { "observation", "prompt", "model_name", + "env", + "metrics", ] } else { &["task", "prompt", "notes", "agent_name", "agent_model_name"] @@ -1412,7 +1463,7 @@ fn search_preview_from_row(row: &Value, query: &str, table: &str) -> String { let Some(value) = row.get(*column).and_then(search_preview_raw_value) else { continue; }; - let preview = search_preview_text(&value); + let preview = search_preview_text(&value, query); if fallback.is_none() && !preview.is_empty() { fallback = Some(preview.clone()); } diff --git a/crates/persisting-pchronicle-cli/src/server/tests.rs b/crates/persisting-pchronicle-cli/src/server/tests.rs index 68bdbb695..94b211ded 100644 --- a/crates/persisting-pchronicle-cli/src/server/tests.rs +++ b/crates/persisting-pchronicle-cli/src/server/tests.rs @@ -51,26 +51,24 @@ fn explorer_run_identity_sql_does_not_project_step_payloads() { } #[test] -fn explorer_run_preview_sql_is_row_bounded() { - let sql = explorer_run_preview_sql( - "dataset", - "steps", - "_file_ AS source_path, document_id, message_value", - "step_id = 1", - 512, - ); - let lowered = sql.to_ascii_lowercase(); - assert!(lowered.contains("limit 512"), "{sql}"); - assert!(lowered.contains("message_value"), "{sql}"); +fn search_preview_is_bounded_around_the_hit() { + let raw = format!("{} ipython {}", "前缀 ".repeat(400), "suffix ".repeat(400)); + let preview = search_preview_text(&raw, "IPYTHON"); + assert!(preview.contains("ipython")); + assert!(preview.chars().count() <= 322); + assert!(preview.starts_with('…') && preview.ends_with('…')); } #[test] -fn search_preview_returns_the_complete_normalized_field() { - let raw = format!("{} ipython {}", "prefix ".repeat(80), "suffix ".repeat(80)); - let preview = search_preview_text(&raw); - assert!(preview.contains("ipython")); - assert!(preview.starts_with("prefix prefix")); - assert!(preview.ends_with("suffix suffix ")); +fn default_search_projects_only_message_body() { + let expression = crate::combine_match_expressions(&["rust".into()]) + .unwrap() + .unwrap(); + assert_eq!(explorer_preview_columns(&expression), ["message_value"]); + let expression = crate::combine_match_expressions(&["#observation(rust)".into()]) + .unwrap() + .unwrap(); + assert_eq!(explorer_preview_columns(&expression), ["observation"]); } #[test] @@ -2462,6 +2460,68 @@ async fn physical_api_lists_empty_sources_for_json_catalog_and_rejects_non_lance assert_eq!(body["code"], "not_found"); } +#[tokio::test] +async fn search_pages_have_body_previews_for_every_visible_run() { + use persisting_pchronicle::storage::StorylineLanceStore; + let root = tempfile::tempdir().unwrap(); + let store = StorylineLanceStore::open(root.path().join("story")) + .await + .unwrap(); + // More than the old 512-row global preview budget in the first run. + for (session, count) in [("a", 513), ("b", 1), ("c", 1), ("metadata", 1)] { + let mut document = storyline_document(session, session); + let template = document.turns[0].clone(); + document.turns = (0..count) + .map(|id| { + let mut turn = template.clone(); + turn.id = id; + turn.message = json!(if session == "metadata" { + "unrelated text".to_owned() + } else { + format!("{} rust {session}", "padding ".repeat(60)) + }); + turn.observation = Some(json!("rust metadata")); + turn.prompt = + persisting_pchronicle::model::StorylinePrompt::from_pair("rust prompt", ""); + turn + }) + .collect(); + store.replace_storyline(&document).await.unwrap(); + } + let app = router(root.path().to_string_lossy().to_string()); + let dataset = encode_query(DEFAULT_DATASET_NAME); + for (offset, expected) in [(0, vec!["a", "b"]), (2, vec!["c"])] { + let (status, body) = get_json( + &app, + &format!("/api/explorer/runs?dataset={dataset}&q=rust&offset={offset}&limit=2"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], 3, "{body}"); + assert_eq!(body["snapshot"]["has_more"], offset == 0); + let records = body["records"].as_array().unwrap(); + assert_eq!(records.len(), expected.len(), "{body}"); + assert_eq!(body["path_index"].as_array().unwrap().len(), expected.len()); + for (record, session) in records.iter().zip(expected) { + assert_eq!(record["session_id"], session); + let preview = record["search_preview"].as_str().unwrap(); + assert!(preview.contains(&format!("rust {session}")), "{record}"); + assert!(preview.chars().count() <= 322); + } + } + let (status, body) = get_json( + &app, + &format!( + "/api/explorer/runs?dataset={dataset}&q={}&limit=2", + encode_query("#observation(rust)") + ), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], 4, "{body}"); + assert_eq!(body["records"][0]["search_preview"], "rust metadata"); +} + #[tokio::test] async fn physical_api_inspects_storyline_lance_layout_file_and_page() { use persisting_pchronicle::storage::StorylineLanceStore; diff --git a/crates/persisting-pchronicle/src/search/find.rs b/crates/persisting-pchronicle/src/search/find.rs index 312909d9c..9cb05599b 100644 --- a/crates/persisting-pchronicle/src/search/find.rs +++ b/crates/persisting-pchronicle/src/search/find.rs @@ -66,12 +66,11 @@ impl FindTextField { pub fn columns(&self) -> &'static [&'static str] { match self { Self::Reasoning => &["reasoning_content"], - Self::Message | Self::User | Self::Assistant => &["message_value"], + Self::Content | Self::Message | Self::User | Self::Assistant => &["message_value"], Self::Observation => &["observation"], Self::Prompt | Self::System => &["prompt", "message_value"], Self::Model => &["model_name"], Self::Env => &["env"], - Self::Content => &["message_value", "observation", "prompt"], Self::All => &[ "message_value", "reasoning_content", diff --git a/crates/persisting-pchronicle/src/search/storyline.rs b/crates/persisting-pchronicle/src/search/storyline.rs index 8eaca396f..8f9171fe1 100644 --- a/crates/persisting-pchronicle/src/search/storyline.rs +++ b/crates/persisting-pchronicle/src/search/storyline.rs @@ -45,14 +45,7 @@ const STORYLINE_FTS_COLUMNS: &[&str] = &[ ]; /// Default Storyline step columns searched by an unqualified text query. -pub const STORYLINE_STEP_SEARCH_COLUMNS: &[&str] = &[ - "message_value", - "reasoning_content", - "model_name", - "observation", - "env", - "prompt", -]; +pub const STORYLINE_STEP_SEARCH_COLUMNS: &[&str] = &["message_value"]; /// Ensure all FTS and JSON search indexes supported by a Storyline table. pub(crate) async fn ensure_storyline_search_indexes(dataset: &mut Dataset) -> Result<()> { diff --git a/docs/src/en/pchronicle/reference/query-model.md b/docs/src/en/pchronicle/reference/query-model.md index 2cae7c447..e0c7dc169 100644 --- a/docs/src/en/pchronicle/reference/query-model.md +++ b/docs/src/en/pchronicle/reference/query-model.md @@ -79,7 +79,7 @@ uses `#field(term)`: | Selector | Meaning | | --- | --- | -| `#content` | `message_value`, `observation`, and `prompt` | +| `#content` | `message_value` (message body; the default for plain text) | | `#message` | `message_value` | | `#user` | `message_value` where `source = 'user'` | | `#assistant` | `message_value` where `source = 'agent'` (`#agent` is an alias) | diff --git a/docs/src/zh/pchronicle/reference/query-model.md b/docs/src/zh/pchronicle/reference/query-model.md index a49fe5547..c80fd475c 100644 --- a/docs/src/zh/pchronicle/reference/query-model.md +++ b/docs/src/zh/pchronicle/reference/query-model.md @@ -70,7 +70,7 @@ JOIN dataset.steps s | 选择器 | 含义 | | --- | --- | -| `#content` | `message_value`、`observation` 和 `prompt` | +| `#content` | `message_value`(消息正文;普通文本默认范围) | | `#message` | `message_value` | | `#user` | `source = 'user'` 的 `message_value` | | `#assistant` | `source = 'agent'` 的 `message_value`(`#agent` 是别名) | diff --git a/pchronicle-web/assets/inline-trace.css b/pchronicle-web/assets/inline-trace.css index 08797bd01..51ace0c2e 100644 --- a/pchronicle-web/assets/inline-trace.css +++ b/pchronicle-web/assets/inline-trace.css @@ -1328,4 +1328,4 @@ details.pc2-context-message-body pre { } .pc2-search-mode{display:inline-flex;align-items:center;height:24px;padding:0 6px;border:1px solid #d0d5dd;border-radius:999px;background:#f8fafc;color:#667085;font-size:9px;font-weight:700;white-space:nowrap}.pc2-search-mode.available{border-color:#bbf7d0;background:#f0fdf4;color:#15803d}.pc2-search-mode.unavailable{border-color:#e4e7ec;background:#f8fafc;color:#98a2b3}.pc2-filter-clear{display:inline-flex;align-items:center;justify-content:center;width:22px;height:22px;margin-right:5px;padding:0;border:0;border-radius:999px;background:transparent;color:#667085;font-size:18px;line-height:1;cursor:pointer}.pc2-filter-clear:hover{background:#eaecf0;color:#1d2939}.pc2-filterbar .pc2-filter-clear{height:22px;min-height:22px;padding:0;border:0;background:transparent;color:#667085}.pc2-search-hit{padding:0 2px;border-radius:3px;background:#fef08a;color:inherit;box-shadow:inset 0 -1px #eab308} .pc2-detail-filter-search{min-width:240px!important;height:36px!important;padding:0 10px!important;border-radius:8px!important}.pc2-detail-filter-search input{width:auto!important;height:auto!important;padding:0!important;font-size:11px!important}.pc2-detail-filter-search .pc2-filter-clear{flex:none;margin-right:0} -.pc2-run-search-preview{max-width:100%;margin-top:2px;overflow:hidden;color:#667085;font-size:9px;line-height:1.35;text-overflow:ellipsis;display:-webkit-box;-webkit-box-orient:vertical;-webkit-line-clamp:2}.pc2-run-search-preview .pc2-search-hit{background:#fef08a;color:#344054;font-weight:700} +.pc2-run-search-preview{max-width:100%;margin-top:4px;color:#475467;font-size:11px;line-height:1.45;white-space:normal;overflow-wrap:anywhere}.pc2-run-search-preview .pc2-search-hit{background:#fef08a;color:#344054;font-weight:700} diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 1d53e58e1..bad01ccf8 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -35,7 +35,7 @@ use crate::model::{ use crate::notice::{ErrorNotice, WorkspaceNotice, workspace_notice}; use crate::terminology::{ANALYSIS, ASSISTANT, DATASETS, RUNS, STEPS, STORAGE, TIMELINE}; -const SEARCH_DEBOUNCE_MS: u32 = 1_000; +const STEP_SEARCH_DEBOUNCE_MS: u32 = 1_000; const CATALOG_REFRESH_MS: u32 = 5_000; fn evidence_notice(turn_id: i64, detail: &str) -> WorkspaceNotice { @@ -248,15 +248,8 @@ pub fn App() -> Element { let runs = use_signal(|| None::); let runs_loading = use_signal(|| true); let runs_generation = use_signal(|| 0u64); - let initial_query = url_param("q").unwrap_or_default(); - let mut query = use_signal({ - let initial_query = initial_query.clone(); - move || initial_query - }); - // Keep the input text separate from the query that drives the network - // request so typing can be debounced without making the input lag. - let mut applied_query = use_signal(move || initial_query); - let mut query_debounce_id = use_signal(|| 0u64); + let mut last_runs_key = use_signal(|| None::); + let mut query = use_signal(|| url_param("q").unwrap_or_default()); let mut dataset_filter = use_signal(|| url_param("dataset_filter").unwrap_or_else(|| "all".into())); let mut status = use_signal(|| url_param("status").unwrap_or_else(|| "all".into())); @@ -337,22 +330,33 @@ pub fn App() -> Element { if page() != "runs" && !on_detail_without_runs { return; } - load_runs( - RunFilters { - query: applied_query(), - dataset: dataset_filter(), - status: status(), - sort: sort(), - direction: direction(), - path: run_path(), - file: file_prefix(), - offset: offset(), - }, - runs, - runs_loading, - runs_generation, - error, + let filters = RunFilters { + query: query(), + dataset: dataset_filter(), + status: status(), + sort: sort(), + direction: direction(), + path: run_path(), + file: file_prefix(), + offset: offset(), + }; + let key = format!( + "{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}", + filters.query, + filters.dataset, + filters.status, + filters.sort, + filters.direction, + filters.path, + filters.file, + filters.offset, + page(), ); + if last_runs_key.peek().as_deref() == Some(key.as_str()) && runs.peek().is_some() { + return; + } + last_runs_key.set(Some(key)); + load_runs(filters, runs, runs_loading, runs_generation, error); }); use_effect(move || { @@ -426,6 +430,31 @@ pub fn App() -> Element { } }); + // A filtered result should be useful immediately. Pick the first real + // match and open its parent conversation instead of leaving the hit hidden + // behind a collapsed summary row. + use_effect(move || { + let query = turn_query(); + if query.trim().is_empty() || expanded_turn_id().is_some() { + return; + } + let Some(turn) = turns().into_iter().find(|turn| turn.id >= 0) else { + return; + }; + expanded_turn_id.set(Some(turn.id)); + selected_turn.set(None); + if let Some(run) = selected_run() { + load_turn( + run, + turn.id, + expanded_turn_id, + selected_turn, + turn_loading, + error, + ); + } + }); + // Route signals are the only tracked inputs. last_place / history_ready are // written inside the helper with peek() + change-only set so this effect // cannot reschedule itself (Dioxus Signal::set always notifies). @@ -483,7 +512,6 @@ pub fn App() -> Element { file_prefix, run_path, query, - applied_query, status, sort, direction, @@ -653,7 +681,7 @@ pub fn App() -> Element { let path_runs = runs().map(|page| page.path_index).unwrap_or_default(); let selected_path = analysis().map(|value| value.run.path).or_else(|| selected_run().map(|run| run.path)).unwrap_or_default(); rsx! { div { class: "pc2-detail-layout", - PathExplorer { runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path, loading: runs_loading(), + PathExplorer { paged: !query().trim().is_empty(), runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path, loading: runs_loading(), on_path: move |value| { run_path.set(value); offset.set(0); page.set("runs".into()); }, on_view_mode: move |mode| path_list_mode.set(mode), on_select: move |run: RunSummary| { turn_query.set(query()); selected_run.set(Some(run)); analysis.set(None); turns.set(Vec::new()); turn_search.set(TurnSearchStatus::default()); selected_turn.set(None); drawer_turn.set(None); drawer_details.set(Vec::new()); drawer_turn_id.set(None); drawer_turn_ids.set(Vec::new()); drawer_title.set(String::new()); drawer_loading.set(false); expanded_turn_id.set(None); }, @@ -705,7 +733,7 @@ pub fn App() -> Element { let request_id = turn_query_debounce_id() + 1; turn_query_debounce_id.set(request_id); spawn(async move { - TimeoutFuture::new(SEARCH_DEBOUNCE_MS).await; + TimeoutFuture::new(STEP_SEARCH_DEBOUNCE_MS).await; if turn_query_debounce_id() == request_id && let Some(run) = selected_run() { @@ -724,6 +752,8 @@ pub fn App() -> Element { on_apply_query: move |value: String| { turn_query_debounce_id.set(turn_query_debounce_id() + 1); turn_query.set(value.clone()); + expanded_turn_id.set(None); + selected_turn.set(None); if let Some(run) = selected_run() { load_turns( run, @@ -808,7 +838,7 @@ pub fn App() -> Element { _ => { let path_runs = runs().map(|value| value.path_index).unwrap_or_default(); rsx! { div { class: "pc2-runs-layout", - PathExplorer { runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path: run_path(), loading: runs_loading(), + PathExplorer { paged: !query().trim().is_empty(), runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path: run_path(), loading: runs_loading(), on_path: move |value| { run_path.set(value); offset.set(0); }, on_view_mode: move |mode| path_list_mode.set(mode), on_select: move |run: RunSummary| { turn_query.set(query()); selected_run.set(Some(run)); analysis.set(None); turns.set(Vec::new()); turn_search.set(TurnSearchStatus::default()); selected_turn.set(None); drawer_turn.set(None); drawer_details.set(Vec::new()); drawer_turn_id.set(None); drawer_turn_ids.set(Vec::new()); drawer_title.set(String::new()); drawer_loading.set(false); expanded_turn_id.set(None); detail_mode.set("trace".into()); page.set("detail".into()); }, @@ -837,24 +867,8 @@ pub fn App() -> Element { datasets: catalog().map(|value| value.datasets).unwrap_or_default(), dataset: dataset_filter(), chat_sessions: assistant_index().sessions.clone(), - on_query: move |value: String| { - query.set(value.clone()); - // A new search starts from the first page; retaining a - // previous offset can make valid matches look absent. - offset.set(0); - let request_id = query_debounce_id() + 1; - query_debounce_id.set(request_id); - spawn(async move { - TimeoutFuture::new(SEARCH_DEBOUNCE_MS).await; - if query_debounce_id() == request_id { - applied_query.set(value); - } - }); - }, on_apply_query: move |value: String| { - query_debounce_id.set(query_debounce_id() + 1); - query.set(value.clone()); - applied_query.set(value); + query.set(value); offset.set(0); }, on_dataset: move |value| { dataset_filter.set(value); run_path.set(String::new()); file_prefix.set(String::new()); offset.set(0); }, @@ -1019,13 +1033,13 @@ fn load_runs( // A separate request per mounted Dataset lets a fast Lance source // paint immediately while a slower JSON source is still scanning. // Each response is merged into the same page as it arrives. - let request_limit = if all_datasets { 200 } else { 50 }; + let request_limit = 50; let mut pending = FuturesUnordered::new(); for dataset in dataset_names { let mut scoped = filters.clone(); scoped.dataset = dataset; pending.push(async move { - api::explorer_runs( + let mut value = api::explorer_runs( &scoped.query, &scoped.dataset, &scoped.status, @@ -1036,7 +1050,34 @@ fn load_runs( if all_datasets { 0 } else { scoped.offset }, request_limit, ) - .await + .await?; + // Merge only the sorted prefix needed for this global page. + // A fixed first-200 cap silently hid later matches. + while all_datasets + && value.snapshot.has_more + && value.records.len() < scoped.offset.saturating_add(request_limit) + && generation() == request_generation + { + let next = api::explorer_runs( + &scoped.query, + &scoped.dataset, + &scoped.status, + &scoped.sort, + &scoped.direction, + &scoped.path, + &scoped.file, + value.snapshot.next_offset, + request_limit, + ) + .await?; + if next.snapshot.next_offset <= value.snapshot.next_offset { + break; + } + value.snapshot = next.snapshot; + value.records.extend(next.records); + value.path_index.extend(next.path_index); + } + Ok::<_, crate::api::ApiFailure>(value) }); } @@ -1116,11 +1157,11 @@ fn merge_run_pages(pages: &[RunPage], filters: &RunFilters) -> RunPage { let all_datasets = filters.dataset.is_empty() || filters.dataset == "all"; let snapshot = if all_datasets { - let total = records.len(); + let total = pages.iter().map(|page| page.snapshot.total).sum(); let limit = 50; let offset = filters.offset.min(total); - let next_offset = (offset + limit).min(total); records = records.into_iter().skip(offset).take(limit).collect(); + let next_offset = offset + records.len(); PageSnapshot { offset, next_offset, @@ -1143,10 +1184,14 @@ fn merge_run_pages(pages: &[RunPage], filters: &RunFilters) -> RunPage { }) }; - let mut path_index = pages - .iter() - .flat_map(|page| page.path_index.iter().cloned()) - .collect::>(); + let mut path_index = if filters.query.trim().is_empty() { + pages + .iter() + .flat_map(|page| page.path_index.iter().cloned()) + .collect::>() + } else { + records.iter().map(|item| item.run.clone()).collect() + }; path_index.sort_by(|left, right| left.path.cmp(&right.path)); path_index.dedup_by(|left, right| left.query() == right.query()); @@ -1528,6 +1573,7 @@ fn PathExplorer( view_mode: PathListMode, selected_path: String, loading: bool, + paged: bool, on_path: EventHandler, on_view_mode: EventHandler, on_select: EventHandler, @@ -1536,7 +1582,7 @@ fn PathExplorer( let import_path_tree = build_import_path_tree(&runs); rsx! { aside { class: "pc2-path-explorer", header { - div { strong { "Run paths" } span { if view_mode == PathListMode::Flat { "All runs in this dataset" } else { "Tree by import path" } } } + div { strong { "Run paths" } span { if paged { "Search results on this page" } else if view_mode == PathListMode::Flat { "All runs in this dataset" } else { "Tree by import path" } } } div { class: "pc2-path-view-toggle", role: "radiogroup", aria_label: "Run list view", button { class: if view_mode == PathListMode::Flat { "active" } else { "" }, role: "radio", aria_checked: view_mode == PathListMode::Flat, onclick: move |_| on_view_mode.call(PathListMode::Flat), "Flat" } button { class: if view_mode == PathListMode::Tree { "active" } else { "" }, role: "radio", aria_checked: view_mode == PathListMode::Tree, onclick: move |_| on_view_mode.call(PathListMode::Tree), "Tree" } @@ -1544,7 +1590,7 @@ fn PathExplorer( span { "{runs.len()}" } } div { class: "pc2-path-tree", - button { class: if selected_path.is_empty() { "pc2-path-all active" } else { "pc2-path-all" }, onclick: move |_| on_path.call(String::new()), span { class: "pc2-path-icon root", "⌂" } strong { "All runs" } code { "{runs.len()}" } } + button { class: if selected_path.is_empty() { "pc2-path-all active" } else { "pc2-path-all" }, onclick: move |_| on_path.call(String::new()), span { class: "pc2-path-icon root", "⌂" } strong { if paged { "Search results" } else { "All runs" } } code { "{runs.len()}" } } if loading && runs.is_empty() { div { class: "pc2-path-loading", span { class: "spinner" } "Loading paths…" } } else if runs.is_empty() { div { class: "pc2-path-empty", "No captured run paths." } } else if view_mode == PathListMode::Flat { @@ -1557,7 +1603,7 @@ fn PathExplorer( } } } - footer { if view_mode == PathListMode::Flat { "Showing all runs in this dataset." } else { "Tree follows the imported path." } } + footer { if paged { "Showing the current search page." } else if view_mode == PathListMode::Flat { "Showing all runs in this dataset." } else { "Tree follows the imported path." } } } } } @@ -1611,7 +1657,6 @@ fn RunsExplorer( direction: String, path: String, file: String, - on_query: EventHandler, on_apply_query: EventHandler, on_dataset: EventHandler, on_status: EventHandler, @@ -1624,6 +1669,11 @@ fn RunsExplorer( on_open_chat: EventHandler, on_select: EventHandler, ) -> Element { + // Typing stays local; only submitted queries reach the page's request effect. + let mut draft_query = use_signal(|| query.clone()); + use_effect(use_reactive((&query,), move |(query,)| { + draft_query.set(query) + })); let total = page.as_ref().map_or(0, |page| page.snapshot.total); let page_offset = page.as_ref().map_or(0, |page| page.snapshot.offset); let page_limit = page.as_ref().map_or(50, |page| page.snapshot.limit); @@ -1643,7 +1693,7 @@ fn RunsExplorer( _ => "FTS unavailable", }; let search_placeholder = if search.fts_available { - "Search runs/content or JSONB (find syntax)" + "Search message body · Enter to search" } else { "Search unavailable for this Dataset" }; @@ -1654,7 +1704,7 @@ fn RunsExplorer( button { class: "button", onclick: on_refresh, "↻ Refresh" } } div { class: "pc2-filterbar", - label { class: "pc2-filter-search", span { "⌕" } input { value: "{query}", placeholder: "{search_placeholder}", aria_label: "Search runs and content", oninput: move |event| on_query.call(event.value()), onkeydown: move |event| { if event.key() == Key::Enter { event.prevent_default(); on_apply_query.call(query.clone()); } } } if !query.is_empty() { button { r#type: "button", class: "pc2-filter-clear", aria_label: "Clear run search", title: "Clear search", onclick: move |event| { event.prevent_default(); on_apply_query.call(String::new()); }, "×" } } } + label { class: "pc2-filter-search", span { "⌕" } input { value: "{draft_query}", placeholder: "{search_placeholder}", aria_label: "Search runs and content", title: "Search message body. Use #all(...) for all fields or an explicit field/JSON filter. Press Enter to search", oninput: move |event| draft_query.set(event.value()), onkeydown: move |event| { if event.key() == Key::Enter { event.prevent_default(); on_apply_query.call(draft_query()); } } } if !draft_query().is_empty() { button { r#type: "button", class: "pc2-filter-clear", aria_label: "Clear run search", title: "Clear search", onclick: move |event| { event.prevent_default(); draft_query.set(String::new()); on_apply_query.call(String::new()); }, "×" } } } select { value: "{dataset}", aria_label: "Filter by Dataset", onchange: move |event| on_dataset.call(event.value()), option { value: "all", "All Datasets" } for mounted in datasets { option { value: "{mounted.name}", "{mounted.label()}" } } @@ -1667,6 +1717,13 @@ fn RunsExplorer( span { class: if search.fts_available { "pc2-search-mode available" } else { "pc2-search-mode unavailable" }, title: "{search_label}", "{search_label}" } span { class: "pc2-result-count", "{total} runs" } } + if page.is_some() { + footer { class: "pc2-pagination", + button { disabled: page_offset == 0, onclick: move |_| on_page.call(page_offset.saturating_sub(page_limit)), "← Previous" } + span { "{page_offset + usize::from(total > 0)}–{page_next} of {total}" } + button { disabled: !page_has_more, onclick: move |_| on_page.call(page_next), "Next →" } + } + } div { class: "pc2-table-wrap", table { class: "pc2-run-table", thead { tr { th { "Session" } th { "Agent / model" } th { "Status" } th { "Events" } th { "Root" } } } @@ -1685,13 +1742,7 @@ fn RunsExplorer( } } } - if page.is_some() { - footer { class: "pc2-pagination", - button { disabled: page_offset == 0, onclick: move |_| on_page.call(page_offset.saturating_sub(page_limit)), "← Previous" } - span { "{page_offset + usize::from(total > 0)}–{page_next} of {total}" } - button { disabled: !page_has_more, onclick: move |_| on_page.call(page_next), "Next →" } - } - } + } } } @@ -1716,7 +1767,7 @@ fn RunTableRow( .map(|value| search_preview_excerpt(value, &highlight_query)); rsx! { tr { tabindex: "0", onclick: move |_| on_select.call(run.clone()), onkeydown: move |event| if event.key() == Key::Enter { on_select.call(keyboard_run.clone()) }, - td { div { class: "pc2-session-cell", div { class: "pc2-session-heading", strong { if compact { "Record · " } HighlightedText { text: item.run.session_id.clone(), query: query.clone() } } if has_chat && !compact { ChatMarker { run: run.clone(), on_open_chat } } } span { if compact { "1 JSON record · {item.run.file}" } else { "{item.run.row_count} captured rows" } } if let Some(preview) = preview { div { class: "pc2-run-search-preview", title: "Matched content preview", HighlightedText { text: preview, query: highlight_query.clone() } } } } } + td { div { class: "pc2-session-cell", div { class: "pc2-session-heading", strong { if compact { "Record · " } HighlightedText { text: item.run.session_id.clone(), query: query.clone() } } if has_chat && !compact { ChatMarker { run: run.clone(), on_open_chat } } } span { if compact { "1 JSON record · {item.run.file}" } else { "{item.run.row_count} captured rows" } } if let Some(preview) = preview { div { class: "pc2-run-search-preview", title: "{preview}", span { "Match: " } HighlightedText { text: preview.clone(), query: highlight_query.clone() } } } } } td { div { class: "pc2-session-cell", strong { if compact { "Compact JSONL" } else { HighlightedText { text: item.run.agent_id.clone(), query: query.clone() } } } span { if !compact { HighlightedText { text: model_text.clone(), query: query.clone() } } } } } td { StatusBadge { value: item.run.status.clone() } } td { class: "pc2-number", "{item.run.row_count}" } @@ -3067,7 +3118,6 @@ fn apply_workspace_search( mut file_prefix: Signal, mut run_path: Signal, mut query: Signal, - mut applied_query: Signal, mut status: Signal, mut sort: Signal, mut direction: Signal, @@ -3092,8 +3142,7 @@ fn apply_workspace_search( file_prefix.set(query_value(search, "file_prefix").unwrap_or_default()); run_path.set(query_value(search, "path").unwrap_or_default()); let next_query = query_value(search, "q").unwrap_or_default(); - query.set(next_query.clone()); - applied_query.set(next_query); + query.set(next_query); status.set(query_value(search, "status").unwrap_or_else(|| "all".into())); sort.set(query_value(search, "sort").unwrap_or_else(|| "session".into())); direction.set(query_value(search, "direction").unwrap_or_else(|| "asc".into())); @@ -3482,6 +3531,18 @@ mod tests { assert!(!merged.snapshot.has_more); } + #[test] + fn merged_search_keeps_remote_totals_and_only_visible_paths() { + let first = server_page(0, 300, &["a", "b"]); + let second = server_page(0, 400, &["c", "d"]); + let mut filters = run_filters("all", 0); + filters.query = "rust".into(); + let merged = merge_run_pages(&[first, second], &filters); + assert_eq!(merged.snapshot.total, 700); + assert!(merged.snapshot.has_more); + assert_eq!(merged.path_index.len(), merged.records.len()); + } + #[test] fn analyze_workspace_url_retains_only_the_session_id() { assert_eq!( From 8d6d1de86648186c53e0128e72af1c7e5325019c Mon Sep 17 00:00:00 2001 From: Reiase Date: Thu, 17 Sep 2026 01:49:42 +0800 Subject: [PATCH 17/22] Parallelize remote object store reads Increase the remote concurrency defaults and overlap adjacent block fetches to reduce latency when reading uncached objects. Batch FTS predicate searches across datasets and open all storyline datasets concurrently. --- crates/persisting-pchronicle-cli/src/lib.rs | 74 +++++++++++-------- .../src/store/blockcache/adapter.rs | 62 +++++++++------- .../src/store/object_store_io_gate.rs | 4 +- .../src/store/storyline/datafusion.rs | 27 +++---- 4 files changed, 87 insertions(+), 80 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/lib.rs b/crates/persisting-pchronicle-cli/src/lib.rs index a03a7d154..b02327e73 100644 --- a/crates/persisting-pchronicle-cli/src/lib.rs +++ b/crates/persisting-pchronicle-cli/src/lib.rs @@ -3588,6 +3588,7 @@ pub(crate) async fn find_expression_predicate_for_dataset( for predicate in text_predicates { let mut matches = Vec::::new(); let mut searched = false; + let mut jobs = Vec::new(); for dataset in snapshot.datasets() { if dataset_filter.is_some_and(|filter| dataset.mount.name != filter) { continue; @@ -3599,40 +3600,49 @@ pub(crate) async fn find_expression_predicate_for_dataset( { continue; } - let Some(paths) = + if let Some(paths) = snapshot.storyline_table_paths(&dataset.mount.name, &source.file)? - else { - continue; - }; - match search_storyline_step_matches_fts_in_columns( - &paths, - &predicate.query, - predicate.field.columns(), - ) - .await { - Ok(step_matches) => { - available = true; - searched = true; - let source_predicate = predicate - .field - .source_predicate() - .map(|value| format!(" AND ({value})")) - .unwrap_or_default(); - matches.extend(step_matches.into_iter().map(|(document_id, step_id)| { - format!( - "(_file_ = {} AND document_id = {} AND step_id = {}{})", - sql_string(&source.file), - sql_string(&document_id), - step_id, - source_predicate, - ) - })); - } - Err(error) => errors.push(format!( - "FTS unavailable for {} / {}: {error:#}", - dataset.mount.name, source.file - )), + jobs.push((dataset.mount.name.clone(), source.file.clone(), paths)); + } + } + } + let query = predicate.query.clone(); + let columns = predicate.field.columns(); + let results = stream::iter(jobs) + .map(|(dataset, file, paths)| { + let query = query.clone(); + async move { + let result = + search_storyline_step_matches_fts_in_columns(&paths, &query, columns).await; + (dataset, file, result) + } + }) + .buffer_unordered(4) + .collect::>() + .await; + for (dataset, file, result) in results { + match result { + Ok(step_matches) => { + available = true; + searched = true; + let source_predicate = predicate + .field + .source_predicate() + .map(|value| format!(" AND ({value})")) + .unwrap_or_default(); + matches.extend(step_matches.into_iter().map(|(document_id, step_id)| { + format!( + "(_file_ = {} AND document_id = {} AND step_id = {}{})", + sql_string(&file), + sql_string(&document_id), + step_id, + source_predicate, + ) + })); + } + Err(error) => { + errors.push(format!("FTS unavailable for {dataset} / {file}: {error:#}")) } } } diff --git a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs index 5a9db89fb..44caa2395 100644 --- a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs +++ b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs @@ -290,34 +290,40 @@ impl ObjectStore for CachedObjectStore { let this = self.clone(); let path = p.clone(); let meta = head.meta.clone(); - let end = range.end; - let stream = futures::stream::try_unfold(range.start, move |start| { - let (this, path, meta, key, mut options) = ( - this.clone(), - path.clone(), - meta.clone(), - key.clone(), - options.clone(), - ); - async move { - if start >= end { - return Ok(None); - } - let block_start = (start / this.block_size_bytes) * this.block_size_bytes; - let block_end = block_start - .saturating_add(this.block_size_bytes) - .min(meta.size); - options.range = Some((block_start..block_end).into()); - let bytes = this - .read_block(&path, &meta, options, &key, block_start..block_end) - .await?; - let next = block_end.min(end); - Ok(Some(( - bytes.slice((start - block_start) as usize..(next - block_start) as usize), - next, - ))) - } - }); + let block_size = self.block_size_bytes; + let blocks = (range.start / block_size + ..range.end.saturating_add(block_size - 1) / block_size) + .map(|block| { + let block_start = block * block_size; + let block_end = block_start.saturating_add(block_size).min(meta.size); + let visible_start = range.start.max(block_start); + let visible_end = range.end.min(block_end); + (block_start, block_end, visible_start, visible_end) + }) + .collect::>(); + // Keep at most four misses in flight. The stream remains lazy, while + // adjacent uncached blocks overlap their S3 requests. + let stream = futures::stream::iter(blocks) + .map( + move |(block_start, block_end, visible_start, visible_end)| { + let this = this.clone(); + let path = path.clone(); + let meta = meta.clone(); + let key = key.clone(); + let mut options = options.clone(); + async move { + options.range = Some((block_start..block_end).into()); + let bytes = this + .read_block(&path, &meta, options, &key, block_start..block_end) + .await?; + Ok::<_, object_store::Error>(bytes.slice( + (visible_start - block_start) as usize + ..(visible_end - block_start) as usize, + )) + } + }, + ) + .buffered(4); Ok(GetResult { payload: GetResultPayload::Stream(stream.boxed()), meta: head.meta, diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 39611dd50..3440eb5bc 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -15,8 +15,8 @@ use std::time::{Duration, Instant}; use tokio::sync::{OwnedSemaphorePermit, Semaphore}; -const DEFAULT_REMOTE_CONCURRENCY: usize = 1; -const MAX_REMOTE_CONCURRENCY: usize = 2; +const DEFAULT_REMOTE_CONCURRENCY: usize = 4; +const MAX_REMOTE_CONCURRENCY: usize = 8; const MAX_RETAINED_SCOPES: usize = 1024; const SCOPE_IDLE_TTL: Duration = Duration::from_secs(300); const MAX_DELAY_MS: u64 = 30_000; diff --git a/crates/persisting-pchronicle/src/store/storyline/datafusion.rs b/crates/persisting-pchronicle/src/store/storyline/datafusion.rs index 42ef80d07..aa8a4210d 100644 --- a/crates/persisting-pchronicle/src/store/storyline/datafusion.rs +++ b/crates/persisting-pchronicle/src/store/storyline/datafusion.rs @@ -435,24 +435,15 @@ impl StorylineDataSource { paths: StorylineTablePaths, options: StorylineDataSourceOptions, ) -> Result { - let remote = paths.runs.to_string_lossy().contains("://") - && !paths.runs.to_string_lossy().starts_with("file:"); - let (runs, steps, tool_calls, objects) = if remote { - // Avoid four concurrent Lance opens against flaky S3 gateways. - ( - open_dataset(&paths.runs, paths.runs_version).await?, - open_dataset(&paths.steps, paths.steps_version).await?, - open_dataset(&paths.tool_calls, paths.tool_calls_version).await?, - open_objects(&paths.objects, paths.objects_version).await?, - ) - } else { - tokio::try_join!( - open_dataset(&paths.runs, paths.runs_version), - open_dataset(&paths.steps, paths.steps_version), - open_dataset(&paths.tool_calls, paths.tool_calls_version), - open_objects(&paths.objects, paths.objects_version), - )? - }; + // Each open is independently gated by endpoint/bucket AIMD. Running + // them together lets metadata HEADs overlap without bypassing the S3 + // safety limits. + let (runs, steps, tool_calls, objects) = tokio::try_join!( + open_dataset(&paths.runs, paths.runs_version), + open_dataset(&paths.steps, paths.steps_version), + open_dataset(&paths.tool_calls, paths.tool_calls_version), + open_objects(&paths.objects, paths.objects_version), + )?; let objects = Arc::new(objects); Ok(Self { paths, From 2e1b7386d2244402adef46b00a9163dee456f8af Mon Sep 17 00:00:00 2001 From: Reiase Date: Fri, 18 Sep 2026 13:31:48 +0800 Subject: [PATCH 18/22] Enhance object store management and retry mechanisms - Introduced `RetryPatience` enum to define retry behavior for interactive and batch operations, allowing for better control over request timeouts. - Updated `with_object_store_retries` to utilize the new patience settings, optimizing retry strategies based on workload type. - Enhanced `PersistentCache` to include compaction logic, reducing fragmentation and improving write performance. - Added methods for managing foreground demand in object store operations, ensuring that interactive requests are prioritized over background tasks. This commit aims to improve the efficiency and reliability of object store interactions in the pChronicle application. --- crates/persisting-pchronicle-cli/src/lib.rs | 20 +- .../src/server/catalog_worker.rs | 112 +++++- .../src/server/explorer.rs | 149 ++++++-- .../src/server/mod.rs | 351 +++++++++++------- .../src/server/problem.rs | 38 ++ .../src/server/request_log.rs | 12 + .../src/server/tests.rs | 250 ++++++++++++- .../src/server/ui_cache.rs | 7 +- crates/persisting-pchronicle/src/storage.rs | 6 +- .../src/store/blockcache/adapter.rs | 236 ++++++++++-- .../src/store/catalog/manifest_cache.rs | 218 ++++++++++- crates/persisting-pchronicle/src/store/mod.rs | 1 - .../src/store/object_store_io_gate.rs | 256 ++++++++++++- .../src/store/opendal_store.rs | 37 +- .../src/store/persistent_cache.rs | 285 ++++++++++++-- .../src/store/storyline/datafusion.rs | 32 +- .../src/store/storyline/mod.rs | 27 ++ .../src/store/storyline/tests.rs | 80 ++++ .../tests/storyline_lance_roundtrip.rs | 40 ++ pchronicle-web/assets/path-explorer.css | 2 +- pchronicle-web/assets/workbench.css | 2 +- pchronicle-web/src/api.rs | 28 +- pchronicle-web/src/model.rs | 2 + pchronicle-web/src/requests.rs | 230 +++++++++--- pchronicle-web/src/workspace.rs | 290 ++++++++++++--- 25 files changed, 2333 insertions(+), 378 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/lib.rs b/crates/persisting-pchronicle-cli/src/lib.rs index b02327e73..2655160c7 100644 --- a/crates/persisting-pchronicle-cli/src/lib.rs +++ b/crates/persisting-pchronicle-cli/src/lib.rs @@ -121,7 +121,9 @@ impl Cli { /// invocations return None and retain their existing execution path. pub fn run_catalog_worker_before_runtime(cli: &Cli) -> Option> { match &cli.command { - Command::Serve(args) if args.catalog_query_worker => Some(server::catalog_worker::run()), + Command::Serve(args) if args.catalog_query_worker => { + Some(server::catalog_worker::run(cli.log_level)) + } _ => None, } } @@ -172,6 +174,18 @@ pub enum LogLevel { Debug, } +impl LogLevel { + /// The `--log-level` value that parses back to this variant. + pub(crate) fn as_arg(self) -> &'static str { + match self { + Self::Error => "error", + Self::Warn => "warn", + Self::Info => "info", + Self::Debug => "debug", + } + } +} + struct DiagnosticWriter<'a> { level: LogLevel, inner: &'a mut dyn Write, @@ -3596,7 +3610,9 @@ pub(crate) async fn find_expression_predicate_for_dataset( for source in &dataset.sources { if source.status != CatalogSourceStatus::Ready || source.kind != CatalogSourceKind::Store - || source_filter.is_some_and(|filter| filter != source.file) + || source_filter.is_some_and(|filter| { + filter != source.file && !source.file.starts_with(&format!("{filter}/")) + }) { continue; } diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index a8c1ac1a9..07fa1955f 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -273,9 +273,12 @@ fn command( exe: PathBuf, home: &std::path::Path, cache: &std::path::Path, + blocks: &std::path::Path, ) -> tokio::process::Command { let mut command = tokio::process::Command::new(exe); command + .arg("--log-level") + .arg(super::request_log::initialized_log_level().as_arg()) .arg("serve") .arg("--catalog-query-worker") .env_clear() @@ -285,6 +288,10 @@ fn command( .env("XDG_CONFIG_HOME", home) .env("XDG_CACHE_HOME", home) .env("PCHRONICLE_CACHE_DIR", cache) + // `env_clear` plus a throwaway HOME makes the Lance block cache resolve + // under a directory that dies with the worker, so every worker refetched + // the same index pages from the object store. Name it explicitly. + .env("PCHRONICLE_LANCE_CACHE_DIR", blocks) .env("AWS_EC2_METADATA_DISABLED", "true") .env("RAYON_NUM_THREADS", "2") .env("AWS_CONFIG_FILE", home.join("no-aws-config")) @@ -306,6 +313,10 @@ fn command( "SSL_CERT_FILE", "SSL_CERT_DIR", "PCHRONICLE_QUERY_MEMORY_LIMIT", + "PCHRONICLE_LANCE_CACHE_CAPACITY_BYTES", + // Workers issue the reads, so admission tuning that never reaches them + // tunes nothing. + "PCHRONICLE_OBJECT_STORE_CONCURRENCY", "RUST_LOG", ] { if let Some(value) = std::env::var_os(key) { @@ -322,7 +333,13 @@ impl Worker { .map(PathBuf::from) .or_else(|| dirs::cache_dir().map(|p| p.join("pchronicle"))) .context("no catalog worker cache directory")?; - let cache = std::path::absolute(root)?.join("workers").join(scope); + let root = std::path::absolute(root)?; + let cache = root.join("workers").join(scope); + // Blocks are keyed by store, object version and size, so every worker + // and every scope can share them. Keeping them beside the per-scope + // caches rather than inside one means a reader does not refetch what + // another worker already paid for. + let blocks = root.join("blocks"); let mut builder = std::fs::DirBuilder::new(); builder.recursive(true); #[cfg(unix)] @@ -331,7 +348,9 @@ impl Worker { builder.mode(0o700); } builder.create(&cache)?; - let mut child = command(std::env::current_exe()?, home.path(), &cache).spawn()?; + builder.create(&blocks)?; + let mut child = + command(std::env::current_exe()?, home.path(), &cache, &blocks).spawn()?; let input = child.stdin.take().context("worker stdin missing")?; let output = child.stdout.take().context("worker stdout missing")?; let mut worker = Self { @@ -424,7 +443,11 @@ pub(super) fn validate_backends(mounts: &[CatalogLibrary]) -> Result<()> { /// Called before main constructs any runtime or threads. Credentials arrive /// only over stdin, and remain fixed for the lifetime of this process. -pub(crate) fn run() -> Result<()> { +pub(crate) fn run(level: crate::LogLevel) -> Result<()> { + // Handlers execute here, so `ApiError::internal` emits its `root_cause` + // line in this process. Without a subscriber the inherited stderr stayed + // empty and every worker-side failure reached the browser as a bare 500. + super::request_log::init_warehouse_tracing(level); let mut input = std::io::stdin().lock(); let mut output = std::io::stdout().lock(); let bootstrap: Bootstrap = read_frame(&mut input)?.context("missing worker bootstrap")?; @@ -661,7 +684,12 @@ mod tests { #[test] fn exec_environment_excludes_ambient_credentials() { let home = tempfile::tempdir().unwrap(); - let cmd = command(PathBuf::from("pchronicle"), home.path(), home.path()); + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + home.path(), + ); let env: HashMap<_, _> = cmd.as_std().get_envs().collect(); for key in [ "AWS_ACCESS_KEY_ID", @@ -683,6 +711,82 @@ mod tests { ); } + #[test] + fn exec_environment_names_a_surviving_block_cache() { + let home = tempfile::tempdir().unwrap(); + let blocks = tempfile::tempdir().unwrap(); + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + blocks.path(), + ); + let env: HashMap<_, _> = cmd.as_std().get_envs().collect(); + // Without this the cache resolves under the worker's throwaway HOME, so + // each worker refetches every index page the last one already read. + assert_eq!( + env[std::ffi::OsStr::new("PCHRONICLE_LANCE_CACHE_DIR")], + Some(blocks.path().as_os_str()) + ); + assert_ne!( + env[std::ffi::OsStr::new("PCHRONICLE_LANCE_CACHE_DIR")], + Some(home.path().as_os_str()) + ); + } + + #[test] + fn exec_environment_forwards_object_store_admission_tuning() { + let home = tempfile::tempdir().unwrap(); + // Workers issue the object-store reads, and `command` clears the + // environment. A knob missing from the allowlist silently tunes only + // the parent, which reads almost nothing. + // SAFETY: single-threaded test asserting how `command` forwards it. + unsafe { std::env::set_var("PCHRONICLE_OBJECT_STORE_CONCURRENCY", "8") }; + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + home.path(), + ); + let env: HashMap<_, _> = cmd.as_std().get_envs().collect(); + assert_eq!( + env[std::ffi::OsStr::new("PCHRONICLE_OBJECT_STORE_CONCURRENCY")], + Some(std::ffi::OsStr::new("8")) + ); + unsafe { std::env::remove_var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") }; + } + + #[test] + fn exec_arguments_carry_the_serve_log_level() { + let home = tempfile::tempdir().unwrap(); + let cmd = command( + PathBuf::from("pchronicle"), + home.path(), + home.path(), + home.path(), + ); + let args: Vec<_> = cmd + .as_std() + .get_args() + .map(|arg| arg.to_string_lossy().into_owned()) + .collect(); + // Handlers run in the child, so it must install a subscriber at the + // serve level; otherwise `root_cause` diagnostics are dropped there. + let level = args + .iter() + .position(|arg| arg == "--log-level") + .map(|index| args[index + 1].clone()); + assert_eq!( + level.as_deref(), + Some(super::super::request_log::initialized_log_level().as_arg()), + "{args:?}" + ); + assert!( + args.contains(&"--catalog-query-worker".to_owned()), + "{args:?}" + ); + } + #[test] fn mixed_s3_credentials_are_rejected_instead_of_using_first_key() { let library = CatalogLibrary { diff --git a/crates/persisting-pchronicle-cli/src/server/explorer.rs b/crates/persisting-pchronicle-cli/src/server/explorer.rs index 50c958e58..a5be52612 100644 --- a/crates/persisting-pchronicle-cli/src/server/explorer.rs +++ b/crates/persisting-pchronicle-cli/src/server/explorer.rs @@ -1,4 +1,4 @@ -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use persisting_pchronicle::model::EventRecord; use persisting_pchronicle::storage::{ @@ -217,6 +217,8 @@ pub(crate) struct TurnSearchStatus { #[derive(Clone, Debug, Serialize)] pub(crate) struct TurnExplorerPage { + #[serde(skip_serializing_if = "Option::is_none")] + pub(crate) analysis: Option, pub(crate) snapshot: PageSnapshot, pub(crate) records: Vec, pub(crate) search: TurnSearchStatus, @@ -519,15 +521,13 @@ pub(crate) fn analyze( let mut kinds = BTreeMap::::new(); let mut model_groups = BTreeMap::::new(); let mut error_count = 0usize; + let index = EventIndex::new(events); for item in turns { if let Some(timestamp) = item.turn.timestamp.as_ref() { timestamps.push(timestamp.clone()); } - let linked = events - .iter() - .filter(|event| item.event_seqs.contains(&event.seq)) - .collect::>(); + let linked = index.linked(&item.event_seqs, events); let mut values = Vec::new(); if let Some(value) = &item.turn.metrics { values.push(value); @@ -681,53 +681,89 @@ pub(crate) fn analyze( } } -pub(crate) fn turn_page( +pub(crate) fn turn_list_with_search( turns: &[TrajectoryTurnView], events: &[EventRecord], q: Option<&str>, source: Option<&str>, - offset: usize, - limit: usize, -) -> ExplorerPage { + search: TurnSearchStatus, +) -> TurnExplorerPage { let needle = q.unwrap_or_default().trim().to_ascii_lowercase(); + let index = EventIndex::new(events); let records = turns .iter() - .filter(|item| source.is_none_or(|source| source == "all" || item.turn.source == source)) - .filter(|item| needle.is_empty() || searchable_turn(item).contains(&needle)) - .map(|item| turn_summary(item, events)) - .collect(); - paginate(records, offset, limit.clamp(1, 500)) -} - -pub(crate) fn turn_page_with_search( - turns: &[TrajectoryTurnView], - events: &[EventRecord], - q: Option<&str>, - source: Option<&str>, - offset: usize, - limit: usize, - search: TurnSearchStatus, -) -> TurnExplorerPage { - let page = turn_page(turns, events, q, source, offset, limit); + .filter(|item| { + source.is_none_or(|source| { + source.is_empty() || source == "all" || item.turn.source == source + }) && (needle.is_empty() || searchable_turn(item).contains(&needle)) + }) + .map(|item| turn_summary(item, events, &index)) + .collect::>(); + let total = records.len(); TurnExplorerPage { - snapshot: page.snapshot, - records: page.records, + analysis: None, + // Preserve the response envelope for existing clients; turns are + // returned in full, with no offset or per-request row cap. + snapshot: PageSnapshot { + offset: 0, + next_offset: total, + total, + has_more: false, + limit: total, + }, + records, search, } } +/// Positions of every event sequence inside a trajectory's event list. +/// +/// Turns reference their events by sequence, so resolving them by scanning the +/// whole list once per turn costs `turns × events` and dominates long +/// trajectories. Repeated sequences are all kept: a source may emit an id more +/// than once, and collapsing them would change token and tool aggregates. +struct EventIndex(HashMap>); + +impl EventIndex { + fn new(events: &[EventRecord]) -> Self { + let mut positions = HashMap::>::with_capacity(events.len()); + for (position, event) in events.iter().enumerate() { + positions.entry(event.seq).or_default().push(position); + } + Self(positions) + } + + fn linked<'a>(&self, seqs: &[u64], events: &'a [EventRecord]) -> Vec<&'a EventRecord> { + let mut positions = seqs + .iter() + .filter_map(|seq| self.0.get(seq)) + .flatten() + .copied() + .collect::>(); + // A turn may list its sequences out of order or repeat one; callers + // expect each event once, in event-list order, as a scan produced them. + positions.sort_unstable(); + positions.dedup(); + positions + .into_iter() + .map(|position| &events[position]) + .collect() + } +} + pub(crate) fn turn_detail( item: &TrajectoryTurnView, events: &[EventRecord], event_provenance: CatalogEventProvenance, ) -> TurnDetail { - let linked = events - .iter() - .filter(|event| item.event_seqs.contains(&event.seq)) + let index = EventIndex::new(events); + let linked = index + .linked(&item.event_seqs, events) + .into_iter() .cloned() .collect::>(); TurnDetail { - summary: turn_summary(item, events), + summary: turn_summary(item, events, &index), turn: item.turn.clone(), wire_tool_calls: item.wire_tool_calls.clone(), event_provenance, @@ -735,11 +771,12 @@ pub(crate) fn turn_detail( } } -fn turn_summary(item: &TrajectoryTurnView, events: &[EventRecord]) -> TurnSummary { - let linked = events - .iter() - .filter(|event| item.event_seqs.contains(&event.seq)) - .collect::>(); +fn turn_summary( + item: &TrajectoryTurnView, + events: &[EventRecord], + index: &EventIndex, +) -> TurnSummary { + let linked = index.linked(&item.event_seqs, events); let mut values = Vec::new(); if let Some(value) = &item.turn.metrics { values.push(value); @@ -1243,6 +1280,44 @@ mod tests { ); } + #[test] + fn event_index_matches_a_full_scan_for_duplicate_and_unordered_sequences() { + let events = [7u64, 3, 7, 5, 3, 9] + .into_iter() + .map(|seq| EventRecord { + identity: Default::default(), + seq, + source: "test".into(), + kind: "note".into(), + timestamp: None, + session_id: None, + agent_id: None, + parent_uuid: None, + trace_id: None, + call_id: None, + subagent_id: None, + parent_agent_id: None, + branch: None, + parent_call_id: None, + payload: serde_json::json!({ "seq": seq }), + }) + .collect::>(); + let index = EventIndex::new(&events); + for seqs in [ + vec![], + vec![3], + vec![9, 3, 7], + vec![5, 5], + vec![3, 4, 7, 11], + ] { + let scanned = events + .iter() + .filter(|event| seqs.contains(&event.seq)) + .collect::>(); + assert_eq!(index.linked(&seqs, &events), scanned, "{seqs:?}"); + } + } + #[test] fn percentiles_report_coverage_without_inventing_missing_samples() { let stats = metric_stats(vec![10.0, 20.0, 30.0, 40.0], 8); diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 9cab7cbf7..93d9ecaef 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -12,9 +12,9 @@ pub(crate) mod request_log; mod request_progress; mod ui_cache; -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use std::net::SocketAddr; -use std::sync::Arc; +use std::sync::{Arc, Mutex, Weak}; use std::time::{Duration, Instant}; use anyhow::Context; @@ -36,6 +36,7 @@ use persisting_pchronicle::storage::StoryCoords; use persisting_pchronicle::storage::{ CatalogConsistency, CatalogErrorPolicy, CatalogEventProvenance, CatalogSnapshotOptions, CatalogStorylineKey, DEFAULT_DATASET_NAME, DatasetCatalogSnapshot, DatasetMount, + with_background_object_store_io, }; use serde::{Deserialize, Serialize}; use serde_json::{Value, json}; @@ -61,7 +62,8 @@ struct AppState { /// Serializes global refreshes and stores the next automatic retry time. catalog_refresh: Arc>, catalog_refresh_interval: Duration, - trajectory_cache: Arc>>, + trajectory_cache: Arc)>>>, + trajectory_flights: Arc>>>, /// Gateway-backed Warehouses read canonical events from the latest /// manifest for single-trace observation, independent of projection idle. live_reads: bool, @@ -77,16 +79,22 @@ struct AppState { const DEFAULT_CATALOG_REFRESH_INTERVAL: Duration = Duration::from_secs(30); // Return a structured timeout before the catalog worker's 60s transport cutoff. const RUNS_SCAN_TIMEOUT: Duration = Duration::from_secs(50); +const TRAJECTORY_READ_TIMEOUT: Duration = Duration::from_secs(50); +// Backstop for the cached scan, which also runs as a background refresh with no +// request deadline. Kept above `RUNS_SCAN_TIMEOUT` so a foreground request +// reports the endpoint's own timeout instead of this inner one. +const RUNS_SCAN_BACKSTOP: Duration = Duration::from_secs(55); -async fn with_runs_deadline( +async fn with_deadline( timeout: Duration, request_id: &RequestId, + timeout_error: fn() -> ApiError, operation: impl std::future::Future>, ) -> Result { tokio::time::timeout(timeout, operation) .await .map_err(|_| { - ApiError::runs_timeout() + timeout_error() .with_request_id(request_id.as_str()) .with_stage(ExecutionStage::Query) })? @@ -236,6 +244,7 @@ fn app_state_with_catalog_refresh_interval( catalog_refresh: Arc::new(tokio::sync::Mutex::new(Instant::now())), catalog_refresh_interval, trajectory_cache: Arc::new(tokio::sync::RwLock::new(None)), + trajectory_flights: Arc::new(Mutex::new(HashMap::new())), live_reads: false, catalog_acl: None, catalog_query_worker: false, @@ -741,9 +750,10 @@ async fn runs( metrics: RequestMetrics, ) -> Result>, ApiError> { let started = Instant::now(); - let summaries = with_runs_deadline( + let summaries = with_deadline( RUNS_SCAN_TIMEOUT, &request_id, + ApiError::runs_timeout, load_run_summaries(&state, None, None, &request_id, Some(&metrics)), ) .await?; @@ -784,7 +794,7 @@ async fn load_run_summaries( }; let query_metrics = metrics.cloned(); let execute = move |background: bool| async move { - let summaries = tokio::time::timeout(RUNS_SCAN_TIMEOUT, async { + let summaries = tokio::time::timeout(RUNS_SCAN_BACKSTOP, async { // Background work must not be attributed to the triggering HTTP request. let metrics = query_metrics.filter(|_| !background); let phase = Instant::now(); @@ -1233,9 +1243,10 @@ async fn explorer_runs( fts: FtsDiagnostics, query: Result, QueryRejection>, ) -> Result, ApiError> { - with_runs_deadline( + with_deadline( RUNS_SCAN_TIMEOUT, &request_id, + ApiError::runs_timeout, explorer_runs_inner(State(state), request_id.clone(), metrics, fts, query), ) .await @@ -1277,7 +1288,17 @@ async fn explorer_runs_inner( .as_deref() .is_some_and(|value| !value.trim().is_empty()) { - let runtime = current_catalog(&state, &request_id).await?; + let file_filter = query + .file + .as_deref() + .map(str::trim) + .filter(|file| !file.is_empty()); + let runtime = match (dataset_filter, file_filter) { + (Some(dataset), Some(file)) => { + catalog_for_source(&state, Some(dataset), file, &request_id).await? + } + _ => current_catalog(&state, &request_id).await?, + }; let raw = query.q.as_deref().unwrap_or_default().trim(); let expression = crate::combine_match_expressions(&[raw.to_owned()]) .map_err(|error| ApiError::invalid_request(error.to_string()))? @@ -1285,7 +1306,11 @@ async fn explorer_runs_inner( let (predicate, fts_available, fts_errors) = crate::find_expression_predicate_for_dataset( &runtime.snapshot, &expression, - None, + query + .file + .as_deref() + .map(str::trim) + .filter(|file| !file.is_empty()), dataset_filter, ) .await @@ -1680,14 +1705,14 @@ async fn try_resolve_on_demand_storyline_run( let Some(store) = open_storyline_source(state, dataset_name, file, request_id).await? else { return Ok(None); }; - let Some((_generation, ids)) = store - .document_ids_snapshot() + let Some((_generation, present)) = store + .contains_document(session_id) .await .map_err(|error| fail(request_id, "resolve_run", error))? else { return Ok(None); }; - if !ids.iter().any(|id| id == session_id) { + if !present { return Ok(None); } let path = explorer::explorer_run_path(dataset_name, file, session_id, session_id, None, None); @@ -2047,12 +2072,14 @@ struct LoadedTrajectory { turns: Vec, } +type TrajectoryFlight = tokio::sync::OnceCell>; + async fn load_trajectory( state: &AppState, query: &SessionQuery, request_id: &RequestId, metrics: &RequestMetrics, -) -> Result { +) -> Result, ApiError> { let phase = Instant::now(); let run = resolve_run_summary(state, query, request_id, Some(metrics)).await?; metrics.record("resolve", phase); @@ -2060,115 +2087,149 @@ async fn load_trajectory( let runtime = catalog_for_source(state, Some(&run.dataset), &run.file, request_id).await?; metrics.record("catalog", phase); let cache_key = format!( - "{}\u{1f}{}\u{1f}{}\u{1f}{}", + "{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}", runtime.snapshot.snapshot_id(), run.dataset, run.file, - run.session_id + run.session_id, + run.document_id ); - if !state.live_reads - && let Some((_, loaded)) = state - .trajectory_cache - .read() - .await - .as_ref() - .filter(|(key, _)| key == &cache_key) + if let Some((_, loaded)) = state + .trajectory_cache + .read() + .await + .as_ref() + .filter(|(key, _)| key == &cache_key) { metrics.record("trajectory_cache", Instant::now()); return Ok(loaded.clone()); } - if run.format.as_deref() == Some("compact-jsonl/v1") { - return Ok(LoadedTrajectory { - runtime, - run, - event_provenance: CatalogEventProvenance::SyntheticFromStoryline, - records: Vec::new(), - turns: Vec::new(), - }); - } - let phase = Instant::now(); - let bundle = catalog_or_on_demand_trajectory_bundle( - state, - &runtime, - &run, - request_id, - "load_trajectory", - ) - .await?; - metrics.record("trajectory_read", phase); - let event_provenance = bundle.event_view.provenance; - let records = bundle.event_view.document.events; - let document = bundle.storyline; - // ACTF step records carry the first user input at document level when it - // is the baseline prompt. Preserve it on the first turn so Explorer can - // render the user side of the conversation without changing storage. - let document_prompt = document.prompt.clone(); - let mut by_call = BTreeMap::>::new(); - for event in &records { - if let Some(call_id) = event.call_id.as_ref().filter(|id| !id.is_empty()) { - by_call.entry(call_id.clone()).or_default().push(event.seq); - } - } - let turns = document - .turns - .into_iter() - .enumerate() - .map(|(turn_index, mut turn)| { - if turn_index == 0 && turn.source == "agent" && turn.prompt.is_none() { - turn.prompt = document_prompt.clone(); + let flight = { + let mut flights = state + .trajectory_flights + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + flights.retain(|_, entry| entry.strong_count() > 0); + if let Some(flight) = flights.get(&cache_key).and_then(Weak::upgrade) { + flight + } else { + let flight = Arc::new(tokio::sync::OnceCell::new()); + if flights.len() < 64 { + flights.insert(cache_key.clone(), Arc::downgrade(&flight)); } - let call_id = turn_call_id(&turn); - let event_seqs = event_seqs_for_turn(&turn, &by_call); - let mut wire_tool_calls = Vec::new(); - for event in records - .iter() - .filter(|event| event_seqs.contains(&event.seq)) - { - collect_wire_tool_calls(&event.payload, &mut wire_tool_calls); + flight + } + }; + // Cancelled initialization can be taken over by a waiter. Failures are not cached. + let loaded = flight + .get_or_try_init(|| async { + if run.format.as_deref() == Some("compact-jsonl/v1") { + return Ok(Arc::new(LoadedTrajectory { + runtime, + run, + event_provenance: CatalogEventProvenance::SyntheticFromStoryline, + records: Vec::new(), + turns: Vec::new(), + })); } - let mut seen = BTreeSet::new(); - wire_tool_calls.retain(|call| { - seen.insert(( - call.id.clone(), - call.name.clone(), - serde_json::to_string(&call.arguments).unwrap_or_default(), - )) - }); - // Tool outputs are canonicalized on the Storyline tool call, not - // necessarily on the event payload that supplied the call. Carry - // that result onto the wire call so AgenticMD can render it next - // to the matching command even when no observation envelope - // exists. - if let Some(native_calls) = turn.tool_calls.as_ref() { - for wire_call in &mut wire_tool_calls { - if wire_call.result.is_none() { - wire_call.result = wire_call.id.as_deref().and_then(|id| { - native_calls - .iter() - .find(|call| call.tool_call_id == id) - .and_then(|call| call.result.clone()) - }); - } + let phase = Instant::now(); + let bundle = catalog_or_on_demand_trajectory_bundle( + state, + &runtime, + &run, + request_id, + "load_trajectory", + ) + .await?; + metrics.record("trajectory_read", phase); + let event_provenance = bundle.event_view.provenance; + let records = bundle.event_view.document.events; + let document = bundle.storyline; + // ACTF step records carry the first user input at document level when it + // is the baseline prompt. Preserve it on the first turn so Explorer can + // render the user side of the conversation without changing storage. + let document_prompt = document.prompt.clone(); + let mut by_call = BTreeMap::>::new(); + let mut by_seq = HashMap::>::with_capacity(records.len()); + for (position, event) in records.iter().enumerate() { + if let Some(call_id) = event.call_id.as_ref().filter(|id| !id.is_empty()) { + by_call.entry(call_id.clone()).or_default().push(event.seq); } + by_seq.entry(event.seq).or_default().push(position); } - TrajectoryTurnView { - turn, - call_id, - event_seqs, - wire_tool_calls, - } + let turns = document + .turns + .into_iter() + .enumerate() + .map(|(turn_index, mut turn)| { + if turn_index == 0 && turn.source == "agent" && turn.prompt.is_none() { + turn.prompt = document_prompt.clone(); + } + let call_id = turn_call_id(&turn); + let event_seqs = event_seqs_for_turn(&turn, &by_call); + // Resolve this turn's events through the sequence index. + // Scanning every record per turn is quadratic and dominates + // long trajectories. + let mut positions = event_seqs + .iter() + .filter_map(|seq| by_seq.get(seq)) + .flatten() + .copied() + .collect::>(); + positions.sort_unstable(); + positions.dedup(); + let mut wire_tool_calls = Vec::new(); + for position in positions { + collect_wire_tool_calls(&records[position].payload, &mut wire_tool_calls); + } + let mut seen = BTreeSet::new(); + wire_tool_calls.retain(|call| { + seen.insert(( + call.id.clone(), + call.name.clone(), + serde_json::to_string(&call.arguments).unwrap_or_default(), + )) + }); + // Tool outputs are canonicalized on the Storyline tool call, not + // necessarily on the event payload that supplied the call. Carry + // that result onto the wire call so AgenticMD can render it next + // to the matching command even when no observation envelope + // exists. + if let Some(native_calls) = turn.tool_calls.as_ref() { + for wire_call in &mut wire_tool_calls { + if wire_call.result.is_none() { + wire_call.result = wire_call.id.as_deref().and_then(|id| { + native_calls + .iter() + .find(|call| call.tool_call_id == id) + .and_then(|call| call.result.clone()) + }); + } + } + } + TrajectoryTurnView { + turn, + call_id, + event_seqs, + wire_tool_calls, + } + }) + .collect(); + let loaded = Arc::new(LoadedTrajectory { + runtime, + run, + event_provenance, + records, + turns, + }); + Ok::<_, ApiError>(loaded) }) - .collect(); - let loaded = LoadedTrajectory { - runtime, - run, - event_provenance, - records, - turns, - }; - if !state.live_reads { - *state.trajectory_cache.write().await = Some((cache_key, loaded.clone())); - } + .await? + .clone(); + // Memoize even under live catalog reads. The turns request and the follow-up + // /run statistics request share a worker when the pool is idle; without this + // memo the second request re-scans the same document. + *state.trajectory_cache.write().await = Some((cache_key, loaded.clone())); Ok(loaded) } @@ -2191,11 +2252,11 @@ async fn trajectory_view( .map(|calls| calls.len()) .sum(); Ok(Json(TrajectoryView { - run: loaded.run, + run: loaded.run.clone(), event_provenance: loaded.event_provenance, event_kind_counts, tool_call_count, - turns: loaded.turns, + turns: loaded.turns.clone(), })) } @@ -2206,13 +2267,31 @@ async fn explorer_run( query: Result, QueryRejection>, ) -> Result, ApiError> { let query = api_query(query)?; - let loaded = load_trajectory(&state, &query, &request_id, &metrics).await?; - Ok(Json(explorer::analyze( - loaded.run, - &loaded.turns, - &loaded.records, - loaded.event_provenance, - ))) + with_deadline( + TRAJECTORY_READ_TIMEOUT, + &request_id, + ApiError::trajectory_timeout, + async { + // Statistics read the whole run, so they admit through the + // background gate. Sharing the foreground gate lets this scan + // occupy every slot and starve the step page the reader is + // actually looking at, which both requests then time out behind. + let loaded = with_background_object_store_io(load_trajectory( + &state, + &query, + &request_id, + &metrics, + )) + .await?; + Ok(Json(explorer::analyze( + loaded.run.clone(), + &loaded.turns, + &loaded.records, + loaded.event_provenance, + ))) + }, + ) + .await } #[derive(Debug, Serialize)] @@ -2244,6 +2323,8 @@ async fn explorer_record( #[derive(Debug, Deserialize)] struct TurnsQuery { + #[serde(default)] + include_analysis: bool, dataset: Option, file: Option, run_id: Option, @@ -2252,8 +2333,6 @@ struct TurnsQuery { root_session_id: Option, q: Option, source: Option, - offset: Option, - limit: Option, } impl TurnsQuery { @@ -2277,8 +2356,26 @@ async fn explorer_turns( metrics: RequestMetrics, fts: FtsDiagnostics, query: Result, QueryRejection>, +) -> Result, ApiError> { + with_deadline( + TRAJECTORY_READ_TIMEOUT, + &request_id, + ApiError::trajectory_timeout, + explorer_turns_inner(State(state), request_id.clone(), metrics, fts, query), + ) + .await +} + +async fn explorer_turns_inner( + State(state): State, + request_id: RequestId, + metrics: RequestMetrics, + fts: FtsDiagnostics, + query: Result, QueryRejection>, ) -> Result, ApiError> { let query = api_query(query)?; + // Timeline and statistics are separate HTTP requests. Embedding analysis + // here forces the step list to wait on the same full-run scan. let session = query.session(); let loaded = load_trajectory(&state, &session, &request_id, &metrics).await?; let phase = Instant::now(); @@ -2395,19 +2492,25 @@ async fn explorer_turns( (loaded.turns.clone(), query.q.as_deref()) }; let phase = Instant::now(); - let page = explorer::turn_page_with_search( + let mut page = explorer::turn_list_with_search( &turns, &loaded.records, search_query, query.source.as_deref(), - query.offset.unwrap_or(0), - query.limit.unwrap_or(100), explorer::TurnSearchStatus { fts_available, mode: search_mode, tokenizer: fts_available.then_some("jieba"), }, ); + if query.include_analysis { + page.analysis = Some(explorer::analyze( + loaded.run.clone(), + &loaded.turns, + &loaded.records, + loaded.event_provenance, + )); + } metrics.record("turn_projection", phase); Ok(Json(page)) } diff --git a/crates/persisting-pchronicle-cli/src/server/problem.rs b/crates/persisting-pchronicle-cli/src/server/problem.rs index f3f65b5a4..b93e0fccb 100644 --- a/crates/persisting-pchronicle-cli/src/server/problem.rs +++ b/crates/persisting-pchronicle-cli/src/server/problem.rs @@ -137,6 +137,15 @@ impl ApiError { let request_id = request_id.as_ref(); let deeper = error.source().is_some(); let root_cause = truncate_utf8(&error.root_cause().to_string(), ROOT_CAUSE_LIMIT); + // An elapsed internal budget is a timeout, not a defect. Reporting it + // as `internal` hid both the retry advice and the fact that the read + // was merely too slow for the request's time budget. + if error + .chain() + .any(|cause| cause.is::()) + { + return Self::deadline_exceeded(request_id, handler, error); + } let api = if let Some(boundary) = error.downcast_ref::() { Self::from_boundary(request_id, boundary.code, boundary.message.clone()) } else { @@ -229,6 +238,35 @@ impl ApiError { ) } + /// A budget inside the handler elapsed. Operators still need the chain, so + /// log it like an internal failure but answer with the retryable status. + fn deadline_exceeded(request_id: &str, handler: &'static str, error: anyhow::Error) -> Self { + tracing::warn!( + target: LOG_TARGET, + request_id = %request_id, + code = "unavailable", + handler = %handler, + root_cause = %truncate_utf8(&error.root_cause().to_string(), ROOT_CAUSE_LIMIT), + chain = %truncate_utf8(&format!("{error:#}"), CHAIN_LIMIT), + "warehouse request exceeded its deadline" + ); + Self::public( + StatusCode::GATEWAY_TIMEOUT, + BoundaryCode::Unavailable, + "Request exceeded its time budget; narrow the dataset or file scope and retry", + ) + .with_request_id(request_id) + .with_stage(ExecutionStage::Query) + } + + pub(super) fn trajectory_timeout() -> Self { + Self::public( + StatusCode::GATEWAY_TIMEOUT, + BoundaryCode::Unavailable, + "Trajectory request timed out; retry this run", + ) + } + pub(super) fn input(issue: InputIssue) -> Self { let message = issue.message().to_owned(); match issue.kind() { diff --git a/crates/persisting-pchronicle-cli/src/server/request_log.rs b/crates/persisting-pchronicle-cli/src/server/request_log.rs index f2512a39c..f73b9ed3a 100644 --- a/crates/persisting-pchronicle-cli/src/server/request_log.rs +++ b/crates/persisting-pchronicle-cli/src/server/request_log.rs @@ -304,7 +304,19 @@ pub(crate) fn tracing_filter(level: crate::LogLevel) -> String { } } +/// Level the process initialized tracing with, so spawned catalog workers can +/// be started at the same verbosity instead of defaulting to their own. +static INITIALIZED_LEVEL: std::sync::OnceLock = std::sync::OnceLock::new(); + +pub(crate) fn initialized_log_level() -> crate::LogLevel { + INITIALIZED_LEVEL + .get() + .copied() + .unwrap_or(crate::LogLevel::Info) +} + pub(crate) fn init_warehouse_tracing(level: crate::LogLevel) { + let _ = INITIALIZED_LEVEL.set(level); // Synchronous stderr is enough: lines are short. Do not wrap this in an // async logger while `main` holds `stdout.lock()` — on macOS `Stderr` // writes take that same lock and deadlock Tokio workers. diff --git a/crates/persisting-pchronicle-cli/src/server/tests.rs b/crates/persisting-pchronicle-cli/src/server/tests.rs index 94b211ded..a1388abc3 100644 --- a/crates/persisting-pchronicle-cli/src/server/tests.rs +++ b/crates/persisting-pchronicle-cli/src/server/tests.rs @@ -332,6 +332,59 @@ async fn internal_error_logs_root_cause_and_redacts_json() { ); } +#[tokio::test] +async fn elapsed_internal_budget_reports_a_retryable_timeout() { + let events = std::sync::Arc::new(std::sync::Mutex::new(Vec::::new())); + let _guard = tracing::subscriber::set_default(CapturingSubscriber::new(events.clone())); + let elapsed = tokio::time::timeout(Duration::from_millis(1), std::future::pending::<()>()) + .await + .expect_err("pending future must time out"); + let error = anyhow::Error::new(elapsed) + .context("runs scan timed out") + .context("cached acceleration build failure"); + let response = super::fail( + &RequestId("rid-deadline".into()), + "load_run_summaries", + error, + ) + .into_response(); + + assert_eq!(response.status(), StatusCode::GATEWAY_TIMEOUT); + let body = response_json(response).await; + assert_eq!(body["code"], "unavailable"); + assert_eq!(body["stage"], "query"); + assert_eq!(body["request_id"], "rid-deadline"); + assert!( + body["message"].as_str().unwrap().contains("time budget"), + "{body}" + ); + + let logged = events.lock().unwrap().clone(); + assert!( + logged + .iter() + .all(|event| event.level != tracing::Level::ERROR), + "a timeout must not be logged as an internal defect: {logged:?}" + ); + let warning = logged + .iter() + .find(|event| event.message.contains("exceeded its deadline")) + .unwrap_or_else(|| panic!("{logged:?}")); + assert!( + warning + .fields + .get("chain") + .unwrap() + .contains("runs scan timed out"), + "{:?}", + warning.fields + ); + assert_eq!( + warning.fields.get("handler").map(String::as_str), + Some("load_run_summaries") + ); +} + #[tokio::test] async fn middleware_echoes_request_id_on_json_errors() { use tower::ServiceExt; @@ -2847,9 +2900,10 @@ async fn runs_deadline_includes_catalog_lock_and_releases_cancelled_work() { let state = app_state(config); let held = state.catalog_refresh.lock().await; let request_id = RequestId("runs-timeout-test".into()); - let error = with_runs_deadline( + let error = with_deadline( Duration::from_millis(20), &request_id, + ApiError::runs_timeout, load_run_summaries(&state, None, None, &request_id, None), ) .await @@ -2860,12 +2914,204 @@ async fn runs_deadline_includes_catalog_lock_and_releases_cancelled_work() { assert_eq!(body["request_id"], "runs-timeout-test"); assert_eq!(body["stage"], "query"); drop(held); - let summaries = with_runs_deadline( + let summaries = with_deadline( Duration::from_secs(5), &request_id, + ApiError::runs_timeout, load_run_summaries(&state, None, None, &request_id, None), ) .await .unwrap(); assert_eq!(summaries.len(), 1); } + +#[tokio::test] +async fn full_turn_lists_and_scoped_search_keep_source_boundaries() { + use persisting_pchronicle::storage::StorylineLanceStore; + let root = tempfile::tempdir().unwrap(); + for (file, session) in [ + ("group/one", "one"), + ("group/two", "two"), + ("group-other", "other"), + ] { + let store = StorylineLanceStore::open(root.path().join(file)) + .await + .unwrap(); + let mut story = storyline_document(session, session); + let template = story.turns[0].clone(); + story.turns = (0..5) + .map(|index| { + let mut turn = template.clone(); + turn.id = 10 + index * 3; + turn.source = if index % 2 == 0 { "user" } else { "agent" }.into(); + turn.message = json!(format!("hello {session} step {index}")); + turn + }) + .collect(); + store.replace_storyline(&story).await.unwrap(); + } + let state = app_state( + ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.path().to_string_lossy()).unwrap(), + ]) + .unwrap(), + ); + let request = RequestId("search-scope-test".into()); + let runtime = current_catalog(&state, &request).await.unwrap(); + let app = finish_routes(state); + for (file, expected) in [("", 3), ("group", 2), ("group/one", 1)] { + let (status, body) = get_json( + &app, + &format!("/api/explorer/runs?dataset={DEFAULT_DATASET_NAME}&file={file}&q=hello"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], expected, "{body}"); + } + let coords = format!( + "dataset={DEFAULT_DATASET_NAME}&file=group/one&agent_id=agent&session_id=one&run_id=one" + ); + for (suffix, total, id) in [ + ("offset=1&limit=2", 5, 10), + ("source=agent&offset=1&limit=1", 2, 13), + ("q=hello&offset=2&limit=1", 5, 10), + ] { + let (status, body) = + get_json(&app, &format!("/api/explorer/turns?{coords}&{suffix}")).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], total, "{body}"); + assert_eq!(body["records"][0]["id"], id, "{body}"); + } + let (status, body) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&offset=99&limit=2"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["snapshot"]["total"], 5); + assert_eq!(body["snapshot"]["has_more"], false); + assert_eq!(body["records"].as_array().unwrap().len(), 5); + let (status, body) = get_json(&app, &format!("/api/explorer/turn?{coords}&turn_id=19")).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["turn"]["id"], 19, "{body}"); + // An unreadable sibling must never be opened by scoped FTS. + let sibling = runtime + .snapshot + .storyline_table_paths(DEFAULT_DATASET_NAME, "group-other") + .unwrap() + .unwrap(); + std::fs::remove_dir_all(&sibling.steps).unwrap(); + let expression = crate::combine_match_expressions(&["hello".into()]) + .unwrap() + .unwrap(); + let (_, available, errors) = crate::find_expression_predicate_for_dataset( + &runtime.snapshot, + &expression, + Some("group"), + Some(DEFAULT_DATASET_NAME), + ) + .await + .unwrap(); + assert!(available); + assert!(errors.is_empty(), "{errors:?}"); +} + +#[tokio::test] +async fn overlapping_trajectory_loads_share_immutable_result() { + let root = json_dataset_root(); + let state = app_state( + ChronicleServerConfig::mounted(vec![ + DatasetMount::default(root.to_string_lossy()).unwrap(), + ]) + .unwrap(), + ); + let request = RequestId("coalesced-trajectory".into()); + let metrics = RequestMetrics::default(); + let summaries = load_run_summaries(&state, None, None, &request, None) + .await + .unwrap(); + let run = &summaries[0]; + let query = SessionQuery { + dataset: Some(run.dataset.clone()), + file: Some(run.file.clone()), + run_id: run.run_id.clone(), + agent_id: run.agent_id.clone(), + session_id: run.session_id.clone(), + root_session_id: run.root_session_id.clone(), + offset: None, + limit: None, + }; + let (left, right) = tokio::join!( + load_trajectory(&state, &query, &request, &metrics), + load_trajectory(&state, &query, &request, &metrics) + ); + assert!(Arc::ptr_eq(&left.unwrap(), &right.unwrap())); +} + +#[tokio::test] +async fn explorer_loads_all_turns_with_offloaded_unknown_fields() { + use persisting_pchronicle::storage::StorylineLanceStore; + let root = tempfile::tempdir().unwrap(); + let store = StorylineLanceStore::open(root.path().join("story")) + .await + .unwrap(); + let mut story = storyline_document("all-turns", "all-turns"); + for index in 0..16 { + story + .unknown_fields + .insert( + "codex", + "source", + format!("/events/{index}"), + json!({"text": "x".repeat(8192)}), + ) + .unwrap(); + } + story.refresh_unknown_key_counts().unwrap(); + let template = story.turns[0].clone(); + story.turns = (0..501) + .map(|index| { + let mut turn = template.clone(); + turn.id = index; + turn + }) + .collect(); + store.replace_storyline(&story).await.unwrap(); + let app = router(root.path().to_string_lossy().into_owned()); + let coords = "dataset=dataset&file=story&agent_id=agent&session_id=all-turns&run_id=all-turns"; + // Even legacy pagination parameters must not truncate the result. + let (status, body) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&offset=100&limit=1"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + let records = body["records"].as_array().unwrap(); + assert_eq!(records.len(), 501); + assert_eq!(records[0]["id"], 0); + assert_eq!(records[500]["id"], 500); + assert_eq!(body["snapshot"]["has_more"], false); + assert!(body.get("analysis").is_none()); + let (status, combined) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&include_analysis=true"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{combined}"); + assert_eq!(combined["records"], body["records"]); + let (status, body) = get_json(&app, &format!("/api/explorer/run?{coords}")).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["turn_count"], 501); + assert_eq!(combined["analysis"], body); + let (status, filtered) = get_json( + &app, + &format!("/api/explorer/turns?{coords}&include_analysis=true&source=nonexistent"), + ) + .await; + assert_eq!(status, StatusCode::OK, "{filtered}"); + assert!(filtered["records"].as_array().unwrap().is_empty()); + assert_eq!( + filtered["analysis"], body, + "statistics must cover the entire run" + ); +} diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index 5ed8ef274..b7a358a2b 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -9,7 +9,7 @@ use std::time::{Duration, Instant}; use anyhow::{Context, Result}; use persisting_pchronicle::storage::{ CatalogConsistency, CatalogState, DatasetLocation, DatasetMount, ManifestCache, PathListKind, - with_background_object_store_io, + wait_for_foreground_object_store_idle, with_background_object_store_io, }; use serde::{Deserialize, Serialize}; use tokio::sync::{RwLock, mpsc, oneshot}; @@ -659,6 +659,11 @@ async fn run_worker( } else { None }; + // Interactive turns/run hold foreground object-store demand; pause + // browse refresh so it does not share the S3 pipe with the user. + if is_background { + wait_for_foreground_object_store_idle().await; + } let work = async { tokio::time::timeout( Duration::from_secs(10), diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 78c34cdbd..4db64032b 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -3,6 +3,7 @@ pub type Result = anyhow::Result; #[cfg(feature = "lance-store")] pub use crate::store::opendal_store::StoreConfig; +pub use crate::store::opendal_store::{RetryPatience, set_retry_patience}; /// Parse an integer byte size with binary IEC suffixes. pub fn parse_byte_size(value: &str) -> std::result::Result { @@ -65,8 +66,9 @@ pub use crate::store::index_build_progress::{ pub use crate::store::object_store_io_gate::{ IoKind as ObjectStoreIoKind, ObjectStoreGateSnapshot, ObjectStoreThrottleEvent, ObjectStoreThrottleHookGuard, format_aimd_flow_label as format_object_store_aimd_flow_label, - install_throttle_hook as install_object_store_throttle_hook, - snapshot as object_store_gate_snapshot, with_background_object_store_io, + foreground_object_store_demand, install_throttle_hook as install_object_store_throttle_hook, + snapshot as object_store_gate_snapshot, wait_for_foreground_object_store_idle, + with_background_object_store_io, }; #[cfg(feature = "lance-store")] diff --git a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs index 44caa2395..a844ce8d4 100644 --- a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs +++ b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs @@ -7,11 +7,19 @@ use async_trait::async_trait; use bytes::Bytes; use futures::{StreamExt, stream::BoxStream}; use object_store::{ - CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, ObjectMeta, - ObjectStore, PutMultipartOptions, PutOptions, PutPayload, PutResult, Result as ObjectResult, - UploadPart, path::Path, + Attributes, CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, + ObjectMeta, ObjectStore, PutMultipartOptions, PutOptions, PutPayload, PutResult, + Result as ObjectResult, UploadPart, path::Path, }; -use std::{ops::Range, sync::Arc}; +use std::{ + ops::Range, + sync::{Arc, Mutex}, + time::{Duration, Instant}, +}; + +/// A round trip slower than this is worth naming next to the request that +/// caused it; the enclosing span already carries the path and byte range. +const SLOW_REQUEST: Duration = Duration::from_millis(200); /// Admission and feedback live at the object-request boundary. The backend /// already retries transport errors; do not replay an entire Lance open here. @@ -20,7 +28,18 @@ async fn remote_request( request: impl std::future::Future>, ) -> ObjectResult { let _permit = io_gate::acquire(uri, IoKind::Read).await; + // Time the transfer alone. Admission queueing is reported by the gate, so + // separating the two keeps a slow backend from reading as contention. + let started = Instant::now(); let result = request.await; + let elapsed = started.elapsed(); + if elapsed >= SLOW_REQUEST { + tracing::debug!( + target: "pchronicle.object_store_gate", + elapsed_ms = elapsed.as_millis() as u64, + "remote object read" + ); + } match &result { Ok(_) => io_gate::note_success(uri), Err(error) if io_gate::is_transient_error(error) => { @@ -140,6 +159,53 @@ pub fn lance_store_params(capacity_bytes: u64) -> lance_io::object_store::Object } } +/// How long a ranged read may reuse metadata it already fetched. +/// +/// A ranged read cannot name its cache block until it knows the object's +/// version, etag and size, so it HEADs the backend first. Asking every time +/// means a cache hit still pays a full round trip and the block cache buys no +/// latency at all. Lance never reuses a path for different bytes -- data files, +/// index pages and manifests are all named per commit -- so the answer is +/// stable for far longer than this. Keep the window short anyway so a backend +/// that does rewrite a path recovers on its own. +const HEAD_TTL: Duration = Duration::from_secs(60); + +/// Objects to remember at once. A reader touches the fragments and index pages +/// of a handful of datasets, so this holds a whole working set. +const HEAD_CAPACITY: usize = 8192; + +/// Metadata for objects this store has already looked up. +#[derive(Debug, Default)] +struct HeadMemo { + entries: std::collections::HashMap, +} + +/// A path plus the version asked for, which is all that changes the answer; +/// remaining conditions are checked against the metadata by the caller. +type HeadKey = (String, Option); + +impl HeadMemo { + fn get(&self, key: &HeadKey) -> Option<(ObjectMeta, Attributes)> { + let (meta, attributes, fetched) = self.entries.get(key)?; + (fetched.elapsed() < HEAD_TTL).then(|| (meta.clone(), attributes.clone())) + } + + fn forget(&mut self, path: &str) { + self.entries.retain(|(known, _), _| known != path); + } + + fn insert(&mut self, key: HeadKey, meta: ObjectMeta, attributes: Attributes) { + if self.entries.len() >= HEAD_CAPACITY { + self.entries + .retain(|_, (_, _, fetched)| fetched.elapsed() < HEAD_TTL); + if self.entries.len() >= HEAD_CAPACITY { + self.entries.clear(); + } + } + self.entries.insert(key, (meta, attributes, Instant::now())); + } +} + #[derive(Debug, Clone)] pub struct CachedObjectStore { inner: Arc, @@ -147,6 +213,7 @@ pub struct CachedObjectStore { block_size_bytes: u64, store_uri: String, io_scope: String, + heads: Arc>, } impl CachedObjectStore { @@ -160,6 +227,7 @@ impl CachedObjectStore { block_size_bytes, io_scope: io_gate::scope_key(&store_uri), store_uri, + heads: Arc::new(Mutex::new(HeadMemo::default())), } } @@ -190,6 +258,12 @@ impl CachedObjectStore { .as_ref() .is_some_and(|tag| result.meta.e_tag.as_ref() != Some(tag)) { + // Remembered metadata describes bytes the backend no + // longer serves; the next read must ask again. + self.heads + .lock() + .expect("head memo") + .forget(path.as_ref()); return Err(object_store::Error::Precondition { path: path.to_string(), source: "object changed while reading cached block".into(), @@ -230,27 +304,40 @@ impl ObjectStore for CachedObjectStore { if o.head || o.range.is_none() { return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; } - // Fetch metadata for the requested version with all caller conditions. - let head = remote_request( - &self.io_scope, - self.inner.get_opts( - p, - GetOptions { - head: true, - range: None, - ..o.clone() - }, - ), - ) - .await?; - o.check_preconditions(&head.meta)?; - let version = head - .meta + // Metadata for the requested version. Conditions beyond the version are + // checked locally below, so remembered metadata reaches the same verdict + // a conditional round trip would have returned. + let head_key = (p.to_string(), o.version.clone()); + let remembered = self.heads.lock().expect("head memo").get(&head_key); + let (head_meta, head_attributes) = match remembered { + Some(head) => head, + None => { + let head = remote_request( + &self.io_scope, + self.inner.get_opts( + p, + GetOptions { + head: true, + range: None, + ..o.clone() + }, + ), + ) + .await?; + let head = (head.meta, head.attributes); + self.heads + .lock() + .expect("head memo") + .insert(head_key.clone(), head.0.clone(), head.1.clone()); + head + } + }; + o.check_preconditions(&head_meta)?; + let version = head_meta .version .clone() .filter(|v| !v.is_empty() && v != "null"); - let etag = head - .meta + let etag = head_meta .e_tag .clone() .filter(|v| !v.is_empty() && !v.starts_with("W/")); @@ -259,14 +346,14 @@ impl ObjectStore for CachedObjectStore { } // A backend that did not identify the requested version cannot safely // populate a version-keyed cache. Preserve its native GET semantics. - if o.version.is_some() && o.version != head.meta.version { + if o.version.is_some() && o.version != head_meta.version { return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; } let range = o .range .as_ref() .unwrap() - .as_range(head.meta.size) + .as_range(head_meta.size) .map_err(|source| object_store::Error::Generic { store: "pchronicle-cache", source: Box::new(source), @@ -278,7 +365,7 @@ impl ObjectStore for CachedObjectStore { p.as_ref(), &version, &etag, - head.meta.size, + head_meta.size, self.block_size_bytes, )) .unwrap(); @@ -289,7 +376,7 @@ impl ObjectStore for CachedObjectStore { }; let this = self.clone(); let path = p.clone(); - let meta = head.meta.clone(); + let meta = head_meta.clone(); let block_size = self.block_size_bytes; let blocks = (range.start / block_size ..range.end.saturating_add(block_size - 1) / block_size) @@ -326,9 +413,9 @@ impl ObjectStore for CachedObjectStore { .buffered(4); Ok(GetResult { payload: GetResultPayload::Stream(stream.boxed()), - meta: head.meta, + meta: head_meta, range, - attributes: head.attributes, + attributes: head_attributes, }) } fn delete_stream( @@ -425,4 +512,97 @@ mod tests { Bytes::from_static(b"0123456789") ); } + + /// Counts what actually leaves for the backend. + #[derive(Debug)] + struct Counted { + inner: Arc, + gets: Arc, + } + + impl std::fmt::Display for Counted { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "counted") + } + } + + #[async_trait] + impl ObjectStore for Counted { + async fn put_opts( + &self, + p: &Path, + b: PutPayload, + o: PutOptions, + ) -> ObjectResult { + self.inner.put_opts(p, b, o).await + } + async fn put_multipart_opts( + &self, + p: &Path, + o: PutMultipartOptions, + ) -> ObjectResult> { + self.inner.put_multipart_opts(p, o).await + } + async fn get_opts(&self, p: &Path, o: GetOptions) -> ObjectResult { + self.gets + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + self.inner.get_opts(p, o).await + } + fn delete_stream( + &self, + p: BoxStream<'static, ObjectResult>, + ) -> BoxStream<'static, ObjectResult> { + self.inner.delete_stream(p) + } + fn list(&self, p: Option<&Path>) -> BoxStream<'static, ObjectResult> { + self.inner.list(p) + } + async fn list_with_delimiter(&self, p: Option<&Path>) -> ObjectResult { + self.inner.list_with_delimiter(p).await + } + async fn copy_opts(&self, a: &Path, b: &Path, o: CopyOptions) -> ObjectResult<()> { + self.inner.copy_opts(a, b, o).await + } + } + + #[tokio::test] + async fn a_cached_range_read_reaches_the_backend_no_more() { + let root = tempfile::tempdir().unwrap(); + let memory = Arc::new(InMemory::new()); + let path = Path::from("dataset/data.lance"); + memory + .put(&path, Bytes::from_static(b"0123456789").into()) + .await + .unwrap(); + let gets = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let cached = CachedObjectStore::new( + Arc::new(Counted { + inner: memory, + gets: Arc::clone(&gets), + }), + CacheConfig::new(root.path().into(), 1024, 4), + "s3://test-bucket".into(), + ); + + assert_eq!( + cached.get_range(&path, 3..8).await.unwrap(), + Bytes::from_static(b"34567") + ); + let after_miss = gets.load(std::sync::atomic::Ordering::Relaxed); + assert!(after_miss > 0, "the first read must fill the cache"); + + // Re-reading served blocks used to re-HEAD the object, so a hit still + // cost a round trip and the cache bought no latency. + for _ in 0..5 { + assert_eq!( + cached.get_range(&path, 3..8).await.unwrap(), + Bytes::from_static(b"34567") + ); + } + assert_eq!( + gets.load(std::sync::atomic::Ordering::Relaxed), + after_miss, + "a fully cached range must not reach the backend" + ); + } } diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index 338e92076..c5a9aa86d 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -13,6 +13,11 @@ use serde::{Deserialize, Serialize}; use tokio::sync::RwLock; const MAX_REFRESH_DIRECTORIES: usize = 10_000; +/// Observations per disk commit during a mount walk. Bounded so a long walk +/// still publishes progress, and so a crash forfeits at most this many reads. +const PERSIST_BATCH_DIRECTORIES: usize = 256; +/// How long a browse observation may wait for neighbours to join its commit. +const PERSIST_MAX_DELAY: Duration = Duration::from_secs(5); const REFRESH_MOUNT_DEADLINE: Duration = Duration::from_secs(120); const REFRESH_DIRECTORY_TIMEOUT: Duration = Duration::from_secs(20); @@ -48,12 +53,27 @@ pub enum ManifestReadMode { Fresh, } +/// One directory observation that is live in memory but not yet on disk. +struct ObservedDirectory { + key: String, + listing: ManifestListing, + removed: Vec, +} + +/// Observations published in memory and still owed to disk. +#[derive(Default)] +struct PendingWrites { + observed: Vec, + since: Option, +} + /// Persistent manifest cache. Keys are caller-owned stable identities, e.g. /// `mount-uri\0relative-prefix`; this keeps the cache independent of UI types. #[derive(Clone)] pub struct ManifestCache { disk: Arc>, values: Arc>>, + pending: Arc>, } impl ManifestCache { @@ -66,7 +86,11 @@ impl ManifestCache { .filter_map(|(key, value)| serde_json::from_value(value).ok().map(|v| (key, v))) .collect(), )); - Self { disk, values } + Self { + disk, + values, + pending: Arc::new(tokio::sync::Mutex::new(PendingWrites::default())), + } } pub async fn get(&self, key: &str) -> Option { @@ -100,27 +124,70 @@ impl ManifestCache { location: &DatasetLocation, prefix: &str, ) -> Result { - self.refresh_impl(key.into(), location, prefix, false).await + // A foreground miss is rare and its caller is waiting on the answer, so + // it writes through rather than joining the browse walk's batch. + let observed = self.observe(key.into(), location, prefix, false).await?; + let listing = observed.listing.clone(); + self.persist(std::slice::from_ref(&observed), prefix).await; + Ok(listing) } /// Observe remote child names now; the browse worker resolves each child's /// type and statistics when visiting that prefix, without delaying its parent. + /// + /// The observation is live in memory when this returns, but its disk write + /// rides with its neighbours: a browse walk visits thousands of prefixes, + /// and a Lance commit per prefix costs more than the listing it records. pub async fn refresh_for_browse( &self, key: impl Into, location: &DatasetLocation, prefix: &str, ) -> Result { - self.refresh_impl(key.into(), location, prefix, true).await + let observed = self.observe(key.into(), location, prefix, true).await?; + let listing = observed.listing.clone(); + self.enqueue(observed).await; + Ok(listing) + } + + /// Commit every observation that is only in memory. + /// + /// A browse walk publishes as it goes, so callers that need the disk copy + /// to agree with memory — a shutdown, a test, a read-after-write — must ask. + pub async fn flush(&self) { + let batch = std::mem::take(&mut *self.pending.lock().await); + self.persist(&batch.observed, "").await; + } + + async fn enqueue(&self, observed: ObservedDirectory) { + let due = { + let mut pending = self.pending.lock().await; + pending.since.get_or_insert_with(tokio::time::Instant::now); + pending.observed.push(observed); + let elapsed = pending + .since + .is_some_and(|since| since.elapsed() >= PERSIST_MAX_DELAY); + if pending.observed.len() >= PERSIST_BATCH_DIRECTORIES || elapsed { + Some(std::mem::take(&mut *pending)) + } else { + None + } + }; + if let Some(batch) = due { + self.persist(&batch.observed, "").await; + } } - async fn refresh_impl( + /// Read one level and publish it in memory, returning what the disk cache + /// still owes. Split from persistence so a mount walk can commit its + /// observations together instead of once per directory. + async fn observe( &self, key: String, location: &DatasetLocation, prefix: &str, browse: bool, - ) -> Result { + ) -> Result { let refresh_gate = crate::store::root_write_lock::for_root(&serde_json::to_string(&(self.path(), &key))?); let _guard = refresh_gate.lock().await; @@ -167,18 +234,43 @@ impl ManifestCache { values.insert(key.clone(), listing.clone()); removed }; + // `key` contains NUL separators and is an internal cache identity; + // logging it makes journald truncate the record at the dataset name. + tracing::debug!(target: "pchronicle.serve", prefix, entries = listing.entries.len(), browse, "manifest cache updated"); + Ok(ObservedDirectory { + key, + listing, + removed, + }) + } + + /// Commit observations to disk. The cache is rebuildable, so a failure or + /// a crash mid-walk only costs the unwritten observations. + async fn persist(&self, observed: &[ObservedDirectory], prefix: &str) { + let payloads = observed + .iter() + .map(|item| serde_json::to_value(&item.listing).map(|value| (item.key.clone(), value))) + .collect::, _>>(); + let payloads = match payloads { + Ok(payloads) => payloads, + Err(error) => { + tracing::warn!(target: "pchronicle.serve", prefix, %error, + "manifest cache encoding failed; using memory"); + return; + } + }; + let removed = observed + .iter() + .flat_map(|item| item.removed.iter().cloned()) + .collect::>(); if let Err(error) = self .disk - .upsert(&key, &serde_json::to_value(&listing)?, &removed) + .upsert_many(payloads.iter().map(|(key, value)| (key, value)), &removed) .await { tracing::warn!(target: "pchronicle.serve", prefix, %error, "manifest cache persistence failed; using memory"); } - // `key` contains NUL separators and is an internal cache identity; - // logging it makes journald truncate the record at the dataset name. - tracing::debug!(target: "pchronicle.serve", prefix, entries = listing.entries.len(), browse, "manifest cache updated"); - Ok(listing) } pub fn writable(&self) -> bool { @@ -315,6 +407,7 @@ impl ManifestCache { let mut queue = VecDeque::from([String::new()]); let mut seen = HashSet::from([String::new()]); let mut report = ManifestRefreshReport::default(); + let mut pending = Vec::::new(); let deadline = tokio::time::Instant::now() + budget; while let Some(prefix) = queue.pop_front() { if report.refreshed_directories >= max_directories @@ -330,9 +423,9 @@ impl ManifestCache { }; let directory_deadline = deadline.min(tokio::time::Instant::now() + REFRESH_DIRECTORY_TIMEOUT); - let listing = match tokio::time::timeout_at( + let observed = match tokio::time::timeout_at( directory_deadline, - self.refresh(key, location, &prefix), + self.observe(key, location, &prefix, false), ) .await { @@ -342,6 +435,12 @@ impl ManifestCache { break; } }; + let listing = observed.listing.clone(); + pending.push(observed); + if pending.len() >= PERSIST_BATCH_DIRECTORIES { + self.persist(&pending, &prefix).await; + pending.clear(); + } report.refreshed_directories += 1; for child in listing .entries @@ -360,6 +459,7 @@ impl ManifestCache { } tokio::task::yield_now().await; } + self.persist(&pending, key_prefix).await; // Publish completeness alongside the cached root, including across restarts. // Do not hold the values lock over persistence. let root = { @@ -428,6 +528,9 @@ mod tests { .unwrap(); assert_eq!(cache.summary("mount").await.datasets, 0); assert!(cache.get("mount\0nested").await.is_none()); + // A browse observation is published in memory first; the disk copy only + // has to agree once the pending writes are asked for. + cache.flush().await; drop(cache); assert_eq!( ManifestCache::open(path) @@ -439,6 +542,97 @@ mod tests { ); } + #[tokio::test] + async fn mount_walk_commits_its_observations_together() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + let directories = 40; + for index in 0..directories { + let leaf = source.join(format!("dir-{index:03}/leaf")); + std::fs::create_dir_all(&leaf).unwrap(); + crate::storage::write_compact_jsonl_manifest(&leaf, 1, 1).unwrap(); + } + let path = temp.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + let report = cache.refresh_mount("mount", &location).await.unwrap(); + // The root and every `dir-*`; a manifest makes `leaf` a dataset entry. + assert_eq!(report.refreshed_directories, directories + 1); + drop(cache); + + // Each Lance commit re-reads the cache to find matches, so a commit per + // directory makes a mount walk quadratic in the directory count. + let versions = lance::Dataset::open(path.to_str().unwrap()) + .await + .unwrap() + .version() + .version; + assert!( + versions <= 4, + "a {directories}-directory walk took {versions} commits" + ); + assert_eq!( + ManifestCache::open(path).await.summary_under("mount").await, + LocationSummary { + partial: false, + datasets: directories as u64, + trajectories: directories as u64, + } + ); + } + + #[tokio::test] + async fn browse_observations_share_one_commit_and_stay_readable() { + let temp = tempfile::tempdir().unwrap(); + let source = temp.path().join("source"); + let prefixes = 40; + for index in 0..prefixes { + let leaf = source.join(format!("dir-{index:03}/leaf")); + std::fs::create_dir_all(&leaf).unwrap(); + crate::storage::write_compact_jsonl_manifest(&leaf, 1, 1).unwrap(); + } + let path = temp.path().join("cache.lance"); + let cache = ManifestCache::open(path.clone()).await; + let location = DatasetLocation::parse(source.to_str().unwrap()).unwrap(); + // The browse coordinator visits one prefix per call, exactly like this. + cache + .refresh_for_browse("mount", &location, "") + .await + .unwrap(); + for index in 0..prefixes { + let prefix = format!("dir-{index:03}"); + cache + .refresh_for_browse(format!("mount\0{prefix}"), &location, &prefix) + .await + .unwrap(); + } + // Every observation must answer from memory before it reaches disk. + assert_eq!(cache.summary_under("mount").await.datasets, prefixes as u64); + assert!(cache.get("mount\0dir-017").await.is_some()); + + cache.flush().await; + drop(cache); + // Each commit builds a DataFusion context and re-reads the target, so a + // commit per visited prefix costs far more than the listings it records. + let versions = lance::Dataset::open(path.to_str().unwrap()) + .await + .unwrap() + .version() + .version; + assert!( + versions <= 3, + "{prefixes} browse observations took {versions} commits" + ); + assert_eq!( + ManifestCache::open(path).await.summary_under("mount").await, + LocationSummary { + partial: false, + datasets: prefixes as u64, + trajectories: prefixes as u64, + } + ); + } + #[tokio::test] async fn bounded_walk_reports_and_persists_partial_then_recovers() { let dir = tempfile::tempdir().unwrap(); diff --git a/crates/persisting-pchronicle/src/store/mod.rs b/crates/persisting-pchronicle/src/store/mod.rs index 6e7c77d1e..47ccb7e5c 100644 --- a/crates/persisting-pchronicle/src/store/mod.rs +++ b/crates/persisting-pchronicle/src/store/mod.rs @@ -51,7 +51,6 @@ pub(crate) mod opendal_store; pub mod persistent_cache; #[cfg(feature = "lance-store")] mod query_engine; -#[cfg(feature = "lance-store")] pub use persistent_cache::PersistentCache; #[cfg(feature = "lance-store")] mod root_write_lock; diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 3440eb5bc..55c9241a7 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -3,9 +3,12 @@ //! Lance opens and table writes against flaky S3-compatible gateways amplify //! timeouts when several datasets race (list `_versions/`, retries, AIMD inside //! object_store). This gate: -//! 1. caps concurrent remote ops per endpoint + bucket (default 1); +//! 1. caps concurrent remote ops per endpoint + bucket (default 4); //! 2. after a transient failure, forces a shared cooldown + growing delay; -//! 3. decays the delay after a streak of successes. +//! 3. decays the delay after a streak of successes; +//! 4. keeps interactive (foreground) work ahead of browse/maintenance +//! (background): background acquires yield while the same scope has +//! foreground demand, so shared S3 bandwidth is not split evenly. //! //! Local `file://` paths bypass the gate entirely. @@ -16,11 +19,13 @@ use std::time::{Duration, Instant}; use tokio::sync::{OwnedSemaphorePermit, Semaphore}; const DEFAULT_REMOTE_CONCURRENCY: usize = 4; +const DEFAULT_BACKGROUND_CONCURRENCY: usize = 1; const MAX_REMOTE_CONCURRENCY: usize = 8; const MAX_RETAINED_SCOPES: usize = 1024; const SCOPE_IDLE_TTL: Duration = Duration::from_secs(300); const MAX_DELAY_MS: u64 = 30_000; const SUCCESS_STREAK_TO_DECAY: u32 = 4; +const BACKGROUND_YIELD_POLL: Duration = Duration::from_millis(25); /// Whether the gated op is primarily reading metadata/objects or writing them. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -152,10 +157,15 @@ impl Default for AimdState { } struct Gate { + /// `foreground` or `background`, so a log line says which lane queued. + lane: &'static str, concurrency: usize, states: Mutex>, } +/// Admission waits shorter than this are noise next to a remote round trip. +const WAIT_LOG_THRESHOLD: Duration = Duration::from_millis(100); + fn state_for<'a>( states: &'a mut HashMap, key: &str, @@ -206,21 +216,122 @@ pub async fn with_background_object_store_io(work: F) -> BACKGROUND_IO.scope((), work).await } +/// Process-wide interactive demand per admission scope (holders + in-flight +/// acquires). Background work polls this so browse refresh does not share the +/// pipe with turns/run while a user request is active. +fn foreground_demand_map() -> &'static Mutex> { + static FOREGROUND_DEMAND: OnceLock>> = OnceLock::new(); + FOREGROUND_DEMAND.get_or_init(|| Mutex::new(HashMap::new())) +} + +fn enter_foreground_demand(key: &str) { + let Ok(mut map) = foreground_demand_map().lock() else { + return; + }; + *map.entry(key.to_owned()).or_insert(0) += 1; +} + +fn leave_foreground_demand(key: &str) { + let Ok(mut map) = foreground_demand_map().lock() else { + return; + }; + let Some(count) = map.get_mut(key) else { + return; + }; + *count = count.saturating_sub(1); + if *count == 0 { + map.remove(key); + } +} + +fn scope_foreground_demand(key: &str) -> u32 { + foreground_demand_map() + .lock() + .ok() + .and_then(|map| map.get(key).copied()) + .unwrap_or(0) +} + +/// Total interactive object-store acquires currently in flight (any scope). +pub fn foreground_object_store_demand() -> u32 { + foreground_demand_map() + .lock() + .ok() + .map(|map| map.values().copied().sum()) + .unwrap_or(0) +} + +/// Block until no interactive object-store work is admitted. Browse refresh +/// calls this before starting a background walk so user requests go first. +pub async fn wait_for_foreground_object_store_idle() { + let mut logged = false; + let started = Instant::now(); + loop { + if foreground_object_store_demand() == 0 { + if logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + yielded_ms = started.elapsed().as_millis() as u64, + "background resumed after interactive object-store idle" + ); + } + return; + } + if !logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + demand = foreground_object_store_demand(), + "background yielding to interactive object-store demand" + ); + logged = true; + } + tokio::time::sleep(BACKGROUND_YIELD_POLL).await; + } +} + +struct ForegroundDemandGuard { + key: String, +} + +impl ForegroundDemandGuard { + fn enter(key: &str) -> Self { + enter_foreground_demand(key); + Self { + key: key.to_owned(), + } + } +} + +impl Drop for ForegroundDemandGuard { + fn drop(&mut self) { + leave_foreground_demand(&self.key); + } +} + fn gate() -> &'static Gate { static FOREGROUND_GATE: OnceLock = OnceLock::new(); static BACKGROUND_GATE: OnceLock = OnceLock::new(); - let slot = if BACKGROUND_IO.try_with(|_| ()).is_ok() { + let background = BACKGROUND_IO.try_with(|_| ()).is_ok(); + let slot = if background { &BACKGROUND_GATE } else { &FOREGROUND_GATE }; slot.get_or_init(|| { - let concurrency = std::env::var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") + let configured = std::env::var("PCHRONICLE_OBJECT_STORE_CONCURRENCY") .ok() .and_then(|value| value.parse::().ok()) .unwrap_or(DEFAULT_REMOTE_CONCURRENCY) .clamp(1, MAX_REMOTE_CONCURRENCY); + // Background stays single-flight: even when interactive is idle, browse + // should not open a second S3 pipeline beside itself. + let concurrency = if background { + DEFAULT_BACKGROUND_CONCURRENCY.min(configured) + } else { + configured + }; Gate { + lane: if background { "background" } else { "foreground" }, concurrency, states: Mutex::new(HashMap::new()), } @@ -388,17 +499,60 @@ pub fn format_aimd_flow_label(snap: &ObjectStoreGateSnapshot, event: Option<&str pub(crate) struct Permit { _permit: Option, + /// Held for the full foreground acquire+hold window so background yields. + _demand: Option, } /// Acquire admission for a Lance/object-store operation on `uri`. pub(crate) async fn acquire(uri: &str, kind: IoKind) -> Permit { if !is_remote_uri(uri) { - return Permit { _permit: None }; + return Permit { + _permit: None, + _demand: None, + }; } acquire_scoped(gate(), &scope_key(uri), kind).await } +async fn yield_to_foreground(key: &str, kind: IoKind) { + let mut logged = false; + let started = Instant::now(); + loop { + if scope_foreground_demand(key) == 0 { + if logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + scope = key, + kind = kind.as_str(), + yielded_ms = started.elapsed().as_millis() as u64, + "background admission resumed after interactive demand cleared" + ); + } + return; + } + if !logged { + tracing::debug!( + target: "pchronicle.object_store_gate", + scope = key, + kind = kind.as_str(), + demand = scope_foreground_demand(key), + "background admission yielding to interactive object-store demand" + ); + logged = true; + } + tokio::time::sleep(BACKGROUND_YIELD_POLL).await; + } +} + async fn acquire_scoped(g: &Gate, key: &str, kind: IoKind) -> Permit { + let background = g.lane == "background"; + // Interactive demand covers the whole wait+hold window so browse cannot + // race into the same S3 endpoint between admit and first byte. + let demand = if background { + None + } else { + Some(ForegroundDemandGuard::enter(key)) + }; let semaphore = { let mut states = g .states @@ -409,18 +563,40 @@ async fn acquire_scoped(g: &Gate, key: &str, kind: IoKind) -> Permit { Arc::clone(&state.semaphore) }; loop { + if background { + yield_to_foreground(key, kind).await; + } wait_out_degradation(g, key, kind).await; let permit = match semaphore.clone().try_acquire_owned() { Ok(permit) => permit, Err(_) => { + // A stall here is admission queueing, not the backend. Without + // a log line the two are indistinguishable in a trace, and + // every remote read looks like a slow round trip. + let queued = Instant::now(); let _wait = WaitGuard::new(g, key, kind, "admit", 0); - match semaphore.clone().acquire_owned().await { + let permit = match semaphore.clone().acquire_owned().await { Ok(permit) => permit, Err(error) => { tracing::error!(?error, "object-store I/O semaphore closed unexpectedly"); - return Permit { _permit: None }; + return Permit { + _permit: None, + _demand: demand, + }; } + }; + let waited = queued.elapsed(); + if waited >= WAIT_LOG_THRESHOLD { + tracing::debug!( + lane = g.lane, + scope = key, + kind = kind.as_str(), + waited_ms = waited.as_millis() as u64, + concurrency = g.concurrency, + "object-store admission queued" + ); } + permit } }; // A failure may have started a new cooldown while admission was @@ -429,8 +605,15 @@ async fn acquire_scoped(g: &Gate, key: &str, kind: IoKind) -> Permit { drop(permit); continue; } + // Interactive arrived while we waited for a background slot: give the + // permit back and yield instead of holding endpoint bandwidth. + if background && scope_foreground_demand(key) > 0 { + drop(permit); + continue; + } return Permit { _permit: Some(permit), + _demand: demand, }; } } @@ -534,6 +717,7 @@ async fn wait_out_degradation(g: &Gate, key: &str, kind: IoKind) { // already logged by `note_failure`, so each waiter need not be WARN noise. tracing::debug!( target: "pchronicle.object_store_gate", + lane = g.lane, kind = kind.as_str(), wait_ms, delay_ms, @@ -661,23 +845,71 @@ mod tests { use super::*; #[tokio::test] - async fn foreground_and_background_have_independent_admission_and_feedback() { + async fn background_yields_while_foreground_holds_admission() { let uri = scope_for_endpoint("s3://workload-isolation/path", "http://localhost:18060"); let mut foreground_permits = Vec::new(); for _ in 0..gate().concurrency { foreground_permits.push(acquire(&uri, IoKind::Read).await); } + assert!(foreground_object_store_demand() > 0); + with_background_object_store_io(async { + let raced = + tokio::time::timeout(Duration::from_millis(80), acquire(&uri, IoKind::Read)).await; + assert!( + raced.is_err(), + "background must yield while interactive holds the endpoint" + ); + }) + .await; + drop(foreground_permits); + assert_eq!(foreground_object_store_demand(), 0); + with_background_object_store_io(async { + let _permit = + tokio::time::timeout(Duration::from_millis(200), acquire(&uri, IoKind::Read)) + .await + .expect("background proceeds once interactive is idle"); + note_failure(&uri, IoKind::Read); + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + // Foreground AIMD is independent of the background failure above. + assert_eq!( + gate() + .states + .lock() + .unwrap() + .get(&uri) + .map(|s| s.failures) + .unwrap_or(0), + 0 + ); + let _permit = tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) + .await + .expect("background cooldown must not delay foreground requests"); + note_success(&uri); + with_background_object_store_io(async { + assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); + }) + .await; + } + + #[tokio::test] + async fn foreground_and_background_have_independent_admission_and_feedback() { + let uri = scope_for_endpoint("s3://aimd-isolation/path", "http://localhost:18061"); + // Idle interactive: background may admit on its own lane immediately. with_background_object_store_io(async { let _permit = tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) .await - .expect("background must not wait on foreground admission"); + .expect("background must not wait when interactive is idle"); note_failure(&uri, IoKind::Read); assert_eq!(gate().states.lock().unwrap()[&uri].failures, 1); }) .await; - assert_eq!(gate().states.lock().unwrap()[&uri].failures, 0); - drop(foreground_permits); + assert_eq!( + gate().states.lock().unwrap().get(&uri).map(|s| s.failures), + None + ); let _permit = tokio::time::timeout(Duration::from_millis(100), acquire(&uri, IoKind::Read)) .await .expect("background cooldown must not delay foreground requests"); @@ -732,6 +964,7 @@ mod tests { #[tokio::test] async fn busy_backend_does_not_block_other_endpoints_or_buckets() { let g = Gate { + lane: "foreground", concurrency: 1, states: Mutex::new(HashMap::new()), }; @@ -779,6 +1012,7 @@ mod tests { #[tokio::test] async fn cooldown_after_queued_admission_releases_capacity_and_cancellation_clears_waiters() { let g = Arc::new(Gate { + lane: "foreground", concurrency: 1, states: Mutex::new(HashMap::new()), }); diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 0bee48479..46394af78 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -48,10 +48,39 @@ impl StoreConfig { } } +/// How long object-store retries may run before the caller gives up. +/// +/// A long import can wait out a multi-minute outage and should. A request +/// serving a browser cannot: retries that outlast its deadline turn every +/// unreachable endpoint into a timeout, so the operator sees "too slow" +/// where the truth is "cannot connect". The two workloads run in separate +/// processes, so each declares its own patience at startup. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RetryPatience { + /// Bounded well inside an interactive request budget. + Interactive, + /// Waits out long outages; the default. + Batch, +} + +static PATIENCE: OnceLock = OnceLock::new(); + +/// Declare this process's retry patience. Only the first call takes effect, +/// and it must happen before any object store is opened. +pub fn set_retry_patience(patience: RetryPatience) { + let _ = PATIENCE.set(patience); +} + /// Retries for transient object-store failures (DNS blips, connect resets, -/// 5xx, rate limits). Tuned for long imports over flaky endpoints: up to 8 -/// retries with exponential backoff + jitter, capped at 30s. +/// 5xx, rate limits): exponential backoff with jitter, bounded by the +/// process's declared patience. fn with_object_store_retries(operator: Operator) -> Operator { + let (max_delay, max_times) = match PATIENCE.get().copied().unwrap_or(RetryPatience::Batch) { + // 0.5s + 1s + 2s + 4s, so a failing operation still reports its own + // error well inside the request budget that wraps it. + RetryPatience::Interactive => (Duration::from_secs(4), 4), + RetryPatience::Batch => (Duration::from_secs(30), 8), + }; operator.layer( RetryLayer::new() .with_notify(|event: opendal::layers::RetryEvent<'_>| { @@ -67,8 +96,8 @@ fn with_object_store_retries(operator: Operator) -> Operator { .with_jitter() .with_factor(2.0) .with_min_delay(Duration::from_millis(500)) - .with_max_delay(Duration::from_secs(30)) - .with_max_times(8), + .with_max_delay(max_delay) + .with_max_times(max_times), ) } diff --git a/crates/persisting-pchronicle/src/store/persistent_cache.rs b/crates/persisting-pchronicle/src/store/persistent_cache.rs index 6fa988282..592ad9353 100644 --- a/crates/persisting-pchronicle/src/store/persistent_cache.rs +++ b/crates/persisting-pchronicle/src/store/persistent_cache.rs @@ -2,6 +2,7 @@ use anyhow::{Context, Result}; use futures::TryStreamExt; use lance::Dataset; +use lance::dataset::optimize::{CompactionOptions, compact_files}; use lance::dataset::{InsertBuilder, MergeInsertBuilder, WhenMatched, WhenNotMatched}; use lance::deps::arrow_array::{Array, RecordBatch, RecordBatchIterator, StringArray}; use lance::deps::arrow_schema::{DataType, Field, Schema}; @@ -11,6 +12,9 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use tokio::sync::RwLock; const SCHEMA_VERSION: &str = "persistent-cache-v1"; +/// Fragments tolerated before a write folds them back together. +const MAX_FRAGMENTS: usize = 32; +const COMPACTION_TARGET_ROWS: usize = 16 * 1024; pub struct PersistentCache { path: PathBuf, values: RwLock>, @@ -63,6 +67,10 @@ where } if let Some(values) = loaded { *cache.values.get_mut() = values; + if let Err(error) = cache.rewrite_if_fragmented().await { + tracing::warn!(target: "pchronicle.serve", error = %error, + "persistent cache compaction failed"); + } } else if let Some(error) = last_error { tracing::warn!(target: "pchronicle.serve", error = %error, writable, attempts, "persistent cache unreadable"); @@ -90,49 +98,141 @@ where self.values.read().await.clone() } pub async fn upsert(&self, key: &K, value: &V, removed: &[K]) -> Result<()> { + self.upsert_many([(key, value)], removed).await + } + + /// Persist many entries in one merge-insert. + /// + /// Every merge-insert is a Lance commit that re-reads the target to find + /// matches, so writing a bulk observation one entry at a time costs + /// `entries × cache size`. Callers that produce a whole observation should + /// hand it over together. + pub async fn upsert_many<'a, I>(&self, entries: I, removed: &[K]) -> Result<()> + where + I: IntoIterator, + K: 'a, + V: 'a, + { if !self.writable() { return Ok(()); } + let mut keys = Vec::new(); + let mut payloads = Vec::new(); + for (key, value) in entries { + keys.push(serde_json::to_string(key)?); + payloads.push(serde_json::to_string(value)?); + } + // Merge-insert rejects a source that repeats a key, and a later + // observation supersedes an earlier one in the same batch. + let mut seen = std::collections::HashSet::with_capacity(keys.len()); + for index in (0..keys.len()).rev() { + if !seen.insert(keys[index].clone()) { + keys.remove(index); + payloads.remove(index); + } + } + // An entry present in this batch was just observed, so a retirement + // recorded earlier in the same batch must not delete it. + let removed = removed + .iter() + .map(|key| serde_json::to_string(key)) + .collect::, _>>()? + .into_iter() + .filter(|key| !seen.contains(key)) + .collect::>(); + if keys.is_empty() && removed.is_empty() { + return Ok(()); + } let _write = self.write_gate.lock().await; - let schema = Arc::new(Schema::new(vec![ - Field::new("key", DataType::Utf8, false), - Field::new("payload", DataType::Utf8, false), - Field::new("schema_version", DataType::Utf8, false), - ])); - let batch = RecordBatch::try_new( - schema.clone(), - vec![ - Arc::new(StringArray::from(vec![serde_json::to_string(key)?])) as _, - Arc::new(StringArray::from(vec![serde_json::to_string(value)?])) as _, - Arc::new(StringArray::from(vec![SCHEMA_VERSION])) as _, - ], - )?; + let batch = cache_batch(keys, payloads)?; + let schema = batch.schema(); if self.path.exists() { let mut dataset = Dataset::open(self.path.to_string_lossy().as_ref()).await?; for chunk in removed.chunks(128) { let keys = chunk .iter() - .map(|key| { - Ok(format!( - "'{}'", - serde_json::to_string(key)?.replace("'", "''") - )) - }) - .collect::>>()? + .map(|key| format!("'{}'", key.replace("'", "''"))) + .collect::>() .join(","); dataset.delete(&format!("key IN ({keys})")).await?; } - MergeInsertBuilder::try_new(Arc::new(dataset), vec!["key".into()])? - .when_matched(WhenMatched::UpdateAll) - .when_not_matched(WhenNotMatched::InsertAll) - .try_build()? - .execute_reader(Box::new(RecordBatchIterator::new(vec![Ok(batch)], schema))) + if batch.num_rows() > 0 { + MergeInsertBuilder::try_new(Arc::new(dataset), vec!["key".into()])? + .when_matched(WhenMatched::UpdateAll) + .when_not_matched(WhenNotMatched::InsertAll) + .try_build()? + .execute_reader(Box::new(RecordBatchIterator::new(vec![Ok(batch)], schema))) + .await?; + } + self.compact_if_fragmented().await?; + } else if batch.num_rows() > 0 { + InsertBuilder::new(self.path.to_string_lossy().as_ref()) + .execute(vec![batch]) .await?; - } else { + } + Ok(()) + } + + /// Fold accumulated fragments back together. + /// + /// A merge-insert reads every fragment of the target to find matches, so a + /// cache grown one write at a time makes each later write scan every row + /// ever written. Compaction keeps that drift bounded between restarts. + async fn compact_if_fragmented(&self) -> Result<()> { + let mut dataset = Dataset::open(self.path.to_string_lossy().as_ref()).await?; + if dataset.get_fragments().len() <= MAX_FRAGMENTS { + return Ok(()); + } + compact_files( + &mut dataset, + CompactionOptions { + target_rows_per_fragment: COMPACTION_TARGET_ROWS, + max_rows_per_group: 1024, + num_threads: Some(1), + ..Default::default() + }, + None, + ) + .await?; + Ok(()) + } + + /// Replace a badly fragmented cache with one file holding what we just read. + /// + /// Caches written one entry at a time before batching landed hold a + /// fragment and a version per entry — tens of thousands of them, gigabytes + /// for a directory listing. Compacting that many fragments is far slower + /// than rewriting the contents we already have in memory, and only a + /// rewrite reclaims the superseded files. The cache is rebuildable, so + /// losing it to a crash mid-rewrite costs a refresh, nothing more. + async fn rewrite_if_fragmented(&self) -> Result<()> { + if !self.writable() || !self.path.exists() { + return Ok(()); + } + let fragments = Dataset::open(self.path.to_string_lossy().as_ref()) + .await? + .get_fragments() + .len(); + if fragments <= MAX_FRAGMENTS { + return Ok(()); + } + let values = self.values.read().await; + let mut keys = Vec::with_capacity(values.len()); + let mut payloads = Vec::with_capacity(values.len()); + for (key, value) in values.iter() { + keys.push(serde_json::to_string(key)?); + payloads.push(serde_json::to_string(value)?); + } + let entries = keys.len(); + let batch = cache_batch(keys, payloads)?; + std::fs::remove_dir_all(&self.path)?; + if batch.num_rows() > 0 { InsertBuilder::new(self.path.to_string_lossy().as_ref()) .execute(vec![batch]) .await?; } + tracing::info!(target: "pchronicle.serve", fragments, entries, + path = %self.path.display(), "rewrote fragmented persistent cache"); Ok(()) } async fn load(&self) -> Result> { @@ -165,6 +265,139 @@ where Ok(values) } } +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn repeated_single_writes_stay_compact_and_readable() -> Result<()> { + let temp = tempfile::tempdir()?; + let path = temp.path().join("cache.lance"); + let cache = PersistentCache::::open(path.clone()).await; + assert!(cache.writable()); + for index in 0..80 { + cache + .upsert(&format!("key-{index:03}"), &format!("value-{index}"), &[]) + .await?; + } + let fragments = Dataset::open(path.to_str().context("cache path")?) + .await? + .get_fragments() + .len(); + assert!(fragments <= MAX_FRAGMENTS, "{fragments} fragments"); + + // Compaction rewrites fragments, so every entry must survive it. + cache.upsert(&"key-007".into(), &"rewritten".into(), &[]).await?; + cache.upsert(&"gone".into(), &String::new(), &[]).await?; + cache + .upsert(&"key-000".into(), &"kept".into(), &["gone".into()]) + .await?; + drop(cache); + let values = PersistentCache::::open(path).await.values().await; + assert_eq!(values.len(), 80); + assert_eq!(values["key-007"], "rewritten"); + assert_eq!(values["key-000"], "kept"); + assert!(!values.contains_key("gone")); + Ok(()) + } + + #[tokio::test] + async fn opening_rewrites_a_cache_fragmented_by_earlier_writes() -> Result<()> { + let temp = tempfile::tempdir()?; + let path = temp.path().join("cache.lance"); + // Reproduce the pre-batching shape: one fragment and one version per + // entry, written without ever folding them back together. + let cache = PersistentCache::::open(path.clone()).await; + let schema = cache_batch(vec!["seed".into()], vec!["\"seed\"".into()])?.schema(); + for index in 0..(MAX_FRAGMENTS * 2) { + let batch = cache_batch( + vec![format!("\"key-{index:03}\"")], + vec![format!("\"value-{index}\"")], + )?; + if index == 0 { + InsertBuilder::new(path.to_str().context("cache path")?) + .execute(vec![batch]) + .await?; + } else { + let dataset = Dataset::open(path.to_str().context("cache path")?).await?; + MergeInsertBuilder::try_new(Arc::new(dataset), vec!["key".into()])? + .when_matched(WhenMatched::UpdateAll) + .when_not_matched(WhenNotMatched::InsertAll) + .try_build()? + .execute_reader(Box::new(RecordBatchIterator::new( + vec![Ok(batch)], + schema.clone(), + ))) + .await?; + } + } + drop(cache); + let before = Dataset::open(path.to_str().context("cache path")?) + .await? + .get_fragments() + .len(); + assert!(before > MAX_FRAGMENTS, "{before} fragments"); + + let cache = PersistentCache::::open(path.clone()).await; + let after = Dataset::open(path.to_str().context("cache path")?) + .await? + .get_fragments() + .len(); + assert!(after < before, "{before} fragments became {after}"); + assert!(after <= MAX_FRAGMENTS, "{after} fragments"); + let values = cache.values().await; + assert_eq!(values.len(), MAX_FRAGMENTS * 2); + assert_eq!(values["key-005"], "value-5"); + // Superseded files must go with the rewrite, not linger as versions. + let versions = std::fs::read_dir(path.join("_versions"))?.count(); + assert!(versions <= 2, "{versions} versions"); + Ok(()) + } + + #[tokio::test] + async fn a_batch_supersedes_its_own_repeats_and_keeps_what_it_writes() -> Result<()> { + let temp = tempfile::tempdir()?; + let path = temp.path().join("cache.lance"); + let cache = PersistentCache::::open(path.clone()).await; + cache.upsert(&"stale".into(), &"old".into(), &[]).await?; + let entries = [ + ("a".to_owned(), "first".to_owned()), + ("a".to_owned(), "last".to_owned()), + ("stale".to_owned(), "revived".to_owned()), + ]; + cache + .upsert_many( + entries.iter().map(|(key, value)| (key, value)), + &["stale".to_owned(), "absent".to_owned()], + ) + .await?; + drop(cache); + let values = PersistentCache::::open(path).await.values().await; + assert_eq!(values["a"], "last"); + // A key this batch writes was just observed; its earlier retirement + // inside the same batch must not delete it. + assert_eq!(values["stale"], "revived"); + Ok(()) + } +} + +fn cache_batch(keys: Vec, payloads: Vec) -> Result { + let schema = Arc::new(Schema::new(vec![ + Field::new("key", DataType::Utf8, false), + Field::new("payload", DataType::Utf8, false), + Field::new("schema_version", DataType::Utf8, false), + ])); + let versions = vec![SCHEMA_VERSION; keys.len()]; + Ok(RecordBatch::try_new( + schema, + vec![ + Arc::new(StringArray::from(keys)) as _, + Arc::new(StringArray::from(payloads)) as _, + Arc::new(StringArray::from(versions)) as _, + ], + )?) +} + fn text_column(batch: &RecordBatch, name: &str) -> Result> { let array = batch .column(batch.schema().index_of(name)?) diff --git a/crates/persisting-pchronicle/src/store/storyline/datafusion.rs b/crates/persisting-pchronicle/src/store/storyline/datafusion.rs index aa8a4210d..97e9e6b21 100644 --- a/crates/persisting-pchronicle/src/store/storyline/datafusion.rs +++ b/crates/persisting-pchronicle/src/store/storyline/datafusion.rs @@ -27,7 +27,7 @@ use lance::Dataset; use lance::deps::arrow_schema::{Schema as ArrowSchema, SchemaRef}; use super::content::{ - content_columns, hydrate_selected_batches, open_objects, preview_selected_batches, + content_columns, hydrate_batches, native_json_columns, open_objects, preview_selected_batches, }; use super::{StorylineLanceStore, StorylineTablePaths}; use crate::store::datafusion_bridge::{from_datafusion, into_datafusion}; @@ -184,7 +184,12 @@ impl TableProvider for StorylineTableProvider { .create_plan() .await .map_err(|error| into_datafusion(error.into()))?; - let selected = selected_content_columns(self.kind, projection, &self.schema); + let selected = selected_content_columns( + self.kind, + projection, + &self.schema, + self.options.content_read_mode == StorylineContentReadMode::Full, + ); if selected.is_empty() { Ok(plan) } else { @@ -193,7 +198,7 @@ impl TableProvider for StorylineTableProvider { selected, match self.options.content_read_mode { StorylineContentReadMode::Full => { - ContentMaterializationMode::Full(self.objects.clone()) + ContentMaterializationMode::Full(self.objects.clone(), self.kind) } StorylineContentReadMode::Preview => ContentMaterializationMode::Preview, }, @@ -239,6 +244,7 @@ fn selected_content_columns( kind: StorylineTableKind, projection: Option<&Vec>, schema: &SchemaRef, + include_native_json: bool, ) -> HashSet<&'static str> { let projected = projection.map(|projection| { projection @@ -248,11 +254,17 @@ fn selected_content_columns( }); content_columns(kind) .iter() - .filter_map(|(name, _)| { + .map(|(name, _)| *name) + .chain( + native_json_columns(kind) + .iter() + .copied() + .filter(|_| include_native_json), + ) + .filter(|name| { projected .as_ref() .is_none_or(|projected| projected.contains(name)) - .then_some(*name) }) .collect() } @@ -267,7 +279,7 @@ struct ContentHydrationExec { #[derive(Debug, Clone)] enum ContentMaterializationMode { - Full(Arc), + Full(Arc, StorylineTableKind), Preview, } @@ -296,7 +308,7 @@ impl DisplayAs for ContentHydrationExec { let mut selected = self.selected.iter().copied().collect::>(); selected.sort_unstable(); let mode = match self.mode { - ContentMaterializationMode::Full(_) => "full", + ContentMaterializationMode::Full(..) => "full", ContentMaterializationMode::Preview => "preview", }; write!( @@ -351,8 +363,10 @@ impl ExecutionPlan for ContentHydrationExec { async move { let batch = batch?; let mut batches = match mode { - ContentMaterializationMode::Full(objects) => { - hydrate_selected_batches(&objects, vec![batch], &selected).await + ContentMaterializationMode::Full(objects, kind) => { + // Match point reads: native JSON envelopes can contain + // offloaded values too, including unknown_fields. + hydrate_batches(&objects, vec![batch], kind).await } ContentMaterializationMode::Preview => { preview_selected_batches(vec![batch], &selected) diff --git a/crates/persisting-pchronicle/src/store/storyline/mod.rs b/crates/persisting-pchronicle/src/store/storyline/mod.rs index a9146448f..cb984ede1 100644 --- a/crates/persisting-pchronicle/src/store/storyline/mod.rs +++ b/crates/persisting-pchronicle/src/store/storyline/mod.rs @@ -711,6 +711,33 @@ impl StorylineLanceStore { Ok(Some((paths.generation, ids))) } + /// Report whether one document is present, without materializing every + /// identity in the source. + /// + /// A source can hold tens of thousands of runs, so reading the whole + /// `document_id` column to answer "does this run live here" dominated + /// Explorer navigation. `document_id` carries a scalar index, so the + /// filtered read is a point lookup. + pub async fn contains_document(&self, document_id: &str) -> Result> { + let Some(paths) = self.current_table_paths().await? else { + return Ok(None); + }; + let predicate = format!("document_id = '{}'", document_id.replace('\'', "''")); + let batches = read_projected_batches( + &paths.runs, + paths.runs_version, + &["document_id"], + Some(&predicate), + ) + .await?; + let matched = batches.iter().map(RecordBatch::num_rows).sum::(); + anyhow::ensure!( + matched <= 1, + "duplicate document_id in committed Storyline snapshot" + ); + Ok(Some((paths.generation, matched == 1))) + } + pub(crate) async fn resolve_current_table_paths(&self) -> Result> { let current = self.read_current_control().await?; let Some(pointer) = current.control.committed else { diff --git a/crates/persisting-pchronicle/src/store/storyline/tests.rs b/crates/persisting-pchronicle/src/store/storyline/tests.rs index da70d62aa..fadb01538 100644 --- a/crates/persisting-pchronicle/src/store/storyline/tests.rs +++ b/crates/persisting-pchronicle/src/store/storyline/tests.rs @@ -379,6 +379,86 @@ async fn unknown_content_ref_magic_string_round_trips_as_literal() { ); } +#[tokio::test] +async fn datafusion_hydrates_offloaded_native_json_columns() { + let temporary = tempfile::tempdir().unwrap(); + let store = StorylineLanceStore::open(temporary.path()).await.unwrap(); + let large = serde_json::json!({"payload": "x".repeat(128 * 1024)}); + let mut expected = story("native-json-content"); + expected + .unknown_fields + .insert("codex", "source", "/events/10", large.clone()) + .unwrap(); + expected.refresh_unknown_key_counts().unwrap(); + expected.turns[0].extra = Some(large.clone()); + let mut envelope = story("native-json-envelope"); + for index in 0..16 { + envelope + .unknown_fields + .insert( + "codex", + "source", + format!("/events/{index}"), + serde_json::json!({"text": "y".repeat(8192)}), + ) + .unwrap(); + } + envelope.refresh_unknown_key_counts().unwrap(); + store + .replace_storylines(&[expected.clone(), envelope.clone()]) + .await + .unwrap(); + + let source = super::datafusion::StorylineDataSource::from_store(&store) + .await + .unwrap(); + let context = source.session_context().unwrap(); + // Project only native JSON: hydration must not depend on also selecting + // an ordinary content column such as notes or message_value. + let batches = context + .sql("SELECT unknown_fields FROM runs WHERE session_id = 'native-json-content'") + .await + .unwrap() + .collect() + .await + .unwrap(); + let actual = lance_arrow::json::decode_json( + batches[0] + .column(0) + .as_any() + .downcast_ref::() + .unwrap() + .value(0), + ); + let actual: crate::formats::unknown_fields::StorylineUnknownFields = + serde_json::from_str(&actual).unwrap(); + assert!( + actual == expected.unknown_fields, + "unknown field payload was not hydrated" + ); + let batches = context + .sql("SELECT * FROM runs WHERE session_id = 'native-json-envelope'") + .await + .unwrap() + .collect() + .await + .unwrap(); + let runs = story_runs_from_batch(&batches[0]).unwrap(); + assert!( + runs[0].unknown_fields == envelope.unknown_fields, + "unknown field envelope was not hydrated" + ); + let batches = context + .sql("SELECT * FROM steps WHERE session_id = 'native-json-content' ORDER BY turn_ordinal") + .await + .unwrap() + .collect() + .await + .unwrap(); + let steps = story_steps_from_batch(&batches[0]).unwrap(); + assert!(steps[0].extra == Some(large), "step extra was not hydrated"); +} + #[tokio::test] async fn default_store_accepts_large_compressible_unknown_value() { let mut expected = story("large-logical-unknown"); diff --git a/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs b/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs index 7bf10be58..f47ec891d 100644 --- a/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs +++ b/crates/persisting-pchronicle/tests/storyline_lance_roundtrip.rs @@ -175,6 +175,46 @@ async fn atif_singleton_array_shape_is_lossless_through_storyline_lance() -> Res Ok(()) } +#[tokio::test] +async fn single_document_lookup_answers_without_listing_the_source() -> Result<()> { + let temporary = tempfile::tempdir()?; + let store = StorylineLanceStore::open(temporary.path()).await?; + let documents = (0..64) + .map(|index| { + let id = format!("run-{index:03}"); + let mut document = StorylineDocument::new(&id, "agent"); + document.trajectory_id = Some(id); + document + }) + .collect::>(); + store.replace_storylines(&documents).await?; + + let (generation, ids) = store + .document_ids_snapshot() + .await? + .expect("committed snapshot"); + assert_eq!(ids.len(), documents.len()); + + // Resolving one run must not depend on the source's document count: the + // Explorer path used to read every identity to answer this question. + let (lookup_generation, present) = store + .contains_document("run-017") + .await? + .expect("committed snapshot"); + assert_eq!(lookup_generation, generation); + assert!(present); + assert_eq!( + store.contains_document("run-999").await?, + Some((generation.clone(), false)) + ); + // A quoted identity must stay a value, never become SQL. + assert_eq!( + store.contains_document("run-017' OR '1'='1").await?, + Some((generation, false)) + ); + Ok(()) +} + #[tokio::test] async fn incremental_storyline_replace_preserves_global_collection_order() -> Result<()> { let temporary = tempfile::tempdir()?; diff --git a/pchronicle-web/assets/path-explorer.css b/pchronicle-web/assets/path-explorer.css index 8933dbe2e..c87dcc912 100644 --- a/pchronicle-web/assets/path-explorer.css +++ b/pchronicle-web/assets/path-explorer.css @@ -1,4 +1,4 @@ -.pc2-runs-layout{min-height:0;display:grid;grid-template-columns:270px minmax(0,1fr);gap:12px;flex:1}.pc2-path-explorer{min-width:0;min-height:0;display:flex;flex-direction:column;border:1px solid #dfe3e8;border-radius:10px;background:#fff;overflow:hidden;box-shadow:0 1px 2px #10182808}.pc2-path-explorer>header{min-height:55px;display:flex;align-items:center;justify-content:space-between;gap:8px;padding:9px 11px;border-bottom:1px solid #eceef1;background:#f8fafc}.pc2-path-explorer>header>div{min-width:0;display:flex;flex-direction:column;gap:2px}.pc2-path-explorer>header strong{color:#1d2939;font-size:11px}.pc2-path-explorer>header div span{color:#98a2b3;font-size:8px}.pc2-path-explorer>header>span{min-width:23px;padding:3px 6px;border:1px solid #d0d5dd;border-radius:999px;background:#fff;color:#667085;font:600 8px ui-monospace,SFMono-Regular,Menlo,monospace;text-align:center}.pc2-path-explorer>footer{padding:8px 10px;border-top:1px solid #eef0f3;background:#fafbfc;color:#98a2b3;font-size:7px;line-height:1.4}.pc2-path-tree{min-height:0;flex:1;padding:7px 6px 14px;overflow:auto;scrollbar-gutter:stable}.pc2-path-all,.pc2-path-row{width:100%;min-height:31px;display:grid;grid-template-columns:20px minmax(0,1fr) 8px 28px;gap:3px;align-items:center;padding:2px 5px;border:0;border-radius:6px;background:transparent;color:#475467;text-align:left}.pc2-path-all{grid-template-columns:20px minmax(0,1fr) 28px;margin-bottom:4px;cursor:pointer}.pc2-path-all:hover,.pc2-path-row:hover{background:#f2f6fc}.pc2-path-all.active,.pc2-path-row.active{background:#eaf2ff;color:#1d4ed8}.pc2-path-row.branch{background:#f7faff}.pc2-path-all strong{overflow:hidden;font-size:9px;text-overflow:ellipsis;white-space:nowrap}.pc2-path-all code,.pc2-path-row>code{color:#98a2b3;font-size:8px;text-align:right}.pc2-path-all.active code,.pc2-path-row.active>code{color:#2563eb}.pc2-path-toggle{width:19px;height:23px;padding:0;border:0;border-radius:4px;background:transparent;color:#98a2b3;font-size:13px;cursor:pointer}.pc2-path-toggle:hover{background:#e4eaf2;color:#475467}.pc2-path-toggle.leaf{display:block}.pc2-path-name{min-width:0;height:27px;display:flex;align-items:center;gap:6px;padding:0;border:0;background:transparent;color:inherit;text-align:left;cursor:pointer}.pc2-path-name>span:last-child{overflow:hidden;font-size:9px;font-weight:600;text-overflow:ellipsis;white-space:nowrap}.pc2-path-icon{position:relative;width:13px;height:11px;flex:none;color:#94a3b8}.pc2-path-icon.root{display:grid;place-items:center;font-size:11px}.pc2-path-icon.folder:before{position:absolute;inset:2px 0 0;border:1px solid currentColor;border-radius:2px;background:#f8fafc;content:""}.pc2-path-icon.folder:after{position:absolute;top:0;left:1px;width:6px;height:4px;border:1px solid currentColor;border-bottom:0;border-radius:2px 2px 0 0;background:#f8fafc;content:""}.pc2-path-icon.run:before{position:absolute;inset:0 1px;border:1px solid currentColor;border-radius:2px;background:#fff;content:""}.pc2-path-icon.run:after{position:absolute;top:3px;left:4px;width:5px;height:1px;background:currentColor;box-shadow:0 3px currentColor;content:""}.pc2-path-row.active .pc2-path-icon{color:#2563eb}.pc2-path-health{width:6px;height:6px;border-radius:50%;background:#98a2b3}.pc2-path-health.good{background:#22c55e}.pc2-path-health.live{background:#3b82f6}.pc2-path-health.bad{background:#ef4444}.pc2-path-children{margin-left:10px;padding-left:5px;border-left:1px solid #dfe3e8}.pc2-path-loading,.pc2-path-empty{min-height:120px;display:flex;align-items:center;justify-content:center;gap:6px;padding:16px;color:#98a2b3;font-size:8px;text-align:center}.pc2-path-loading .spinner{width:12px;height:12px;margin:0}.pc2-path-filter{max-width:180px;height:30px!important;overflow:hidden;padding:0 8px;border-color:#bfdbfe!important;background:#eff6ff!important;color:#1d4ed8!important;font-size:8px!important;text-overflow:ellipsis;white-space:nowrap;cursor:pointer}.pc2-path-all:focus-visible,.pc2-path-toggle:focus-visible,.pc2-path-name:focus-visible{outline:2px solid #60a5fa;outline-offset:1px}@media(max-width:1150px){.pc2-runs-layout{grid-template-columns:230px minmax(0,1fr)}.pc2-filter-search{min-width:240px}}@media(max-width:850px){.pc2-runs-layout{display:flex;overflow:auto;flex-direction:column}.pc2-path-explorer{min-height:220px;max-height:260px}.pc2-path-explorer>footer{display:none}} +.pc2-runs-layout{min-height:0;display:grid;grid-template-columns:270px minmax(0,1fr);gap:12px;flex:1}.pc2-path-explorer{min-width:0;min-height:0;display:flex;flex-direction:column;border:1px solid #dfe3e8;border-radius:10px;background:#fff;overflow:hidden;box-shadow:0 1px 2px #10182808}.pc2-path-explorer>header{min-height:55px;display:flex;align-items:center;justify-content:space-between;gap:8px;padding:9px 11px;border-bottom:1px solid #eceef1;background:#f8fafc}.pc2-path-explorer>header>div{min-width:0;display:flex;flex-direction:column;gap:2px}.pc2-path-explorer>header strong{color:#1d2939;font-size:11px}.pc2-path-explorer>header div span{color:#98a2b3;font-size:8px}.pc2-path-explorer>header>span{min-width:23px;padding:3px 6px;border:1px solid #d0d5dd;border-radius:999px;background:#fff;color:#667085;font:600 8px ui-monospace,SFMono-Regular,Menlo,monospace;text-align:center}.pc2-path-explorer>footer{padding:8px 10px;border-top:1px solid #eef0f3;background:#fafbfc;color:#98a2b3;font-size:7px;line-height:1.4}.pc2-path-explorer>footer.paged{display:flex;align-items:center;justify-content:space-between;gap:8px}.pc2-path-page{min-width:0;display:flex;align-items:center;gap:5px;color:#667085;font-size:8px}.pc2-path-page select{max-width:132px;height:22px;padding:0 4px;border:1px solid #d0d5dd;border-radius:5px;background:#fff;color:#344054;font:600 8px ui-monospace,SFMono-Regular,Menlo,monospace}.pc2-path-page select:focus-visible{outline:2px solid #60a5fa;outline-offset:1px}.pc2-path-tree{min-height:0;flex:1;padding:7px 6px 14px;overflow:auto;scrollbar-gutter:stable}.pc2-path-all,.pc2-path-row{width:100%;min-height:31px;display:grid;grid-template-columns:20px minmax(0,1fr) 8px 28px;gap:3px;align-items:center;padding:2px 5px;border:0;border-radius:6px;background:transparent;color:#475467;text-align:left}.pc2-path-all{grid-template-columns:20px minmax(0,1fr) 28px;margin-bottom:4px;cursor:pointer}.pc2-path-all:hover,.pc2-path-row:hover{background:#f2f6fc}.pc2-path-all.active,.pc2-path-row.active{background:#eaf2ff;color:#1d4ed8}.pc2-path-row.branch{background:#f7faff}.pc2-path-all strong{overflow:hidden;font-size:9px;text-overflow:ellipsis;white-space:nowrap}.pc2-path-all code,.pc2-path-row>code{color:#98a2b3;font-size:8px;text-align:right}.pc2-path-all.active code,.pc2-path-row.active>code{color:#2563eb}.pc2-path-toggle{width:19px;height:23px;padding:0;border:0;border-radius:4px;background:transparent;color:#98a2b3;font-size:13px;cursor:pointer}.pc2-path-toggle:hover{background:#e4eaf2;color:#475467}.pc2-path-toggle.leaf{display:block}.pc2-path-name{min-width:0;height:27px;display:flex;align-items:center;gap:6px;padding:0;border:0;background:transparent;color:inherit;text-align:left;cursor:pointer}.pc2-path-name>span:last-child{overflow:hidden;font-size:9px;font-weight:600;text-overflow:ellipsis;white-space:nowrap}.pc2-path-icon{position:relative;width:13px;height:11px;flex:none;color:#94a3b8}.pc2-path-icon.root{display:grid;place-items:center;font-size:11px}.pc2-path-icon.folder:before{position:absolute;inset:2px 0 0;border:1px solid currentColor;border-radius:2px;background:#f8fafc;content:""}.pc2-path-icon.folder:after{position:absolute;top:0;left:1px;width:6px;height:4px;border:1px solid currentColor;border-bottom:0;border-radius:2px 2px 0 0;background:#f8fafc;content:""}.pc2-path-icon.run:before{position:absolute;inset:0 1px;border:1px solid currentColor;border-radius:2px;background:#fff;content:""}.pc2-path-icon.run:after{position:absolute;top:3px;left:4px;width:5px;height:1px;background:currentColor;box-shadow:0 3px currentColor;content:""}.pc2-path-row.active .pc2-path-icon{color:#2563eb}.pc2-path-health{width:6px;height:6px;border-radius:50%;background:#98a2b3}.pc2-path-health.good{background:#22c55e}.pc2-path-health.live{background:#3b82f6}.pc2-path-health.bad{background:#ef4444}.pc2-path-children{margin-left:10px;padding-left:5px;border-left:1px solid #dfe3e8}.pc2-path-loading,.pc2-path-empty{min-height:120px;display:flex;align-items:center;justify-content:center;gap:6px;padding:16px;color:#98a2b3;font-size:8px;text-align:center}.pc2-path-loading .spinner{width:12px;height:12px;margin:0}.pc2-path-filter{max-width:180px;height:30px!important;overflow:hidden;padding:0 8px;border-color:#bfdbfe!important;background:#eff6ff!important;color:#1d4ed8!important;font-size:8px!important;text-overflow:ellipsis;white-space:nowrap;cursor:pointer}.pc2-path-all:focus-visible,.pc2-path-toggle:focus-visible,.pc2-path-name:focus-visible{outline:2px solid #60a5fa;outline-offset:1px}@media(max-width:1150px){.pc2-runs-layout{grid-template-columns:230px minmax(0,1fr)}.pc2-filter-search{min-width:240px}}@media(max-width:850px){.pc2-runs-layout{display:flex;overflow:auto;flex-direction:column}.pc2-path-explorer{min-height:220px;max-height:260px}.pc2-path-explorer>footer{display:none}} .pc2-path-row{grid-template-columns:20px minmax(0,1fr) 24px 8px 28px} .pc2-path-explorer>header>.pc2-path-view-toggle{flex:0 0 auto;display:flex;flex-direction:row;gap:2px;padding:2px;border:1px solid #d0d5dd;border-radius:6px;background:#fff}.pc2-path-view-toggle button{height:22px;padding:0 5px;border:0;border-radius:4px;background:transparent;color:#667085;font-size:8px;cursor:pointer}.pc2-path-view-toggle button.active{background:#eaf2ff;color:#1d4ed8;font-weight:700}.pc2-path-view-toggle button:focus-visible{outline:2px solid #60a5fa;outline-offset:1px} .pc2-path-explorer>header>div:first-child{flex:1} diff --git a/pchronicle-web/assets/workbench.css b/pchronicle-web/assets/workbench.css index e7d53e075..850c9a782 100644 --- a/pchronicle-web/assets/workbench.css +++ b/pchronicle-web/assets/workbench.css @@ -152,6 +152,6 @@ .pc2-form select{width:100%;height:42px;padding:0 12px;border:1px solid #cbd5e1;border-radius:9px;background:#fff;color:#344054;font:inherit}.pc2-identity-list{display:flex;flex-direction:column;gap:8px}.pc2-identity-list .button{width:100%;text-align:left}.pc2-settings .button.danger{border-color:#fecaca;color:#b42318;background:#fff}.pc2-settings .button.danger:hover{background:#fff1f0} /* Request diagnostics uses the existing workspace rail and scroll surface. */ -.request-indicator{display:flex;align-items:center;gap:7px;width:100%;padding:10px 14px;border:0;border-radius:8px;background:#ffffff0b;color:#cbd5e1;text-align:left;font-size:12px;cursor:pointer}.request-indicator .spinner{margin:0;width:12px;height:12px;flex:none}.requests-panel{padding:24px;overflow:auto;height:100%;box-sizing:border-box}.requests-panel header p{color:var(--muted)}.requests-panel form{display:flex;gap:8px;margin:16px 0}.requests-panel input{min-width:260px;padding:8px;border:1px solid var(--border);border-radius:6px}.requests-grid{display:grid;grid-template-columns:minmax(240px,1fr) minmax(400px,2fr);gap:20px}.requests-grid aside{max-height:65vh;overflow:auto}.request-row{display:flex;flex-direction:column;gap:6px;width:100%;padding:12px;border:1px solid var(--border);background:white;text-align:left;cursor:pointer;overflow-wrap:anywhere}.request-row span{font-size:11px;color:var(--muted)}.request-row.active{background:#eff6ff;border-color:#2563eb}.requests-grid article{min-width:0;padding:20px;border:1px solid var(--border);border-radius:10px;background:white;overflow-wrap:anywhere}.request-phases{width:100%;border-collapse:collapse;font-size:13px}.request-phases caption{text-align:left;font-weight:600;padding:14px 0}.request-phases td,.request-phases th{text-align:left;padding:10px 6px;border-bottom:1px solid var(--border)}.request-error{color:#b42318}.request-indicator:focus-visible,.request-row:focus-visible{outline:2px solid #60a5fa;outline-offset:2px}@media(max-width:850px){.requests-grid{grid-template-columns:1fr}.requests-grid aside{max-height:220px}} +.request-indicator{display:flex;align-items:center;gap:7px;width:100%;padding:10px 14px;border:0;border-radius:8px;background:#ffffff0b;color:#cbd5e1;text-align:left;font-size:12px;cursor:pointer}.request-indicator .spinner{margin:0;width:12px;height:12px;flex:none}.requests-panel{padding:24px;overflow:auto;height:100%;box-sizing:border-box}.requests-panel header p{color:var(--muted)}.requests-panel form{display:flex;gap:8px;margin:16px 0}.requests-panel input{min-width:260px;padding:8px;border:1px solid var(--border);border-radius:6px}.requests-grid{display:grid;grid-template-columns:minmax(240px,1fr) minmax(400px,2fr);gap:20px;align-items:start}.requests-grid aside{max-height:65vh;overflow:auto}.request-row{display:flex;flex-direction:column;gap:4px;width:100%;padding:12px;border:1px solid var(--border);background:white;text-align:left;cursor:pointer;overflow-wrap:anywhere}.request-row span{font-size:11px;color:var(--muted)}.request-row-summary{color:#344054!important;font-weight:600}.request-row-id{font:500 10px ui-monospace,SFMono-Regular,Menlo,monospace}.request-row.active{background:#eff6ff;border-color:#2563eb}.requests-grid article{min-width:0;max-height:65vh;padding:20px;border:1px solid var(--border);border-radius:10px;background:white;overflow:auto;overflow-wrap:anywhere}.request-phases{width:100%;border-collapse:collapse;font-size:13px}.request-phases caption{text-align:left;font-weight:600;padding:14px 0 8px}.request-phases td,.request-phases th{text-align:left;padding:10px 6px;border-bottom:1px solid var(--border)}.request-error{color:#b42318}.request-indicator:focus-visible,.request-row:focus-visible{outline:2px solid #60a5fa;outline-offset:2px}@media(max-width:850px){.requests-grid{grid-template-columns:1fr}.requests-grid aside{max-height:220px}.requests-grid article{max-height:none}} .request-indicator{flex-wrap:wrap;box-sizing:border-box}.request-indicator small{width:100%;font-size:10px;overflow-wrap:anywhere}.request-row strong{font-size:13px}.requests-panel input:focus-visible{outline:2px solid #2563eb;outline-offset:2px}@media(max-width:850px){.request-indicator{font-size:9px;padding:8px 2px;overflow-wrap:anywhere}.request-indicator small{display:none}.requests-panel input{min-width:0;width:100%}} diff --git a/pchronicle-web/src/api.rs b/pchronicle-web/src/api.rs index f5e8c19d3..fb5a7704f 100644 --- a/pchronicle-web/src/api.rs +++ b/pchronicle-web/src/api.rs @@ -199,15 +199,6 @@ pub async fn explorer_tree_anonymous( json_checked(Request::get(&url).send_tracked().await).await } -pub async fn run_analysis(run: &RunSummary) -> Result { - json_checked( - with_catalog_headers(Request::get(&format!("/api/explorer/run?{}", run.query()))) - .send_tracked() - .await, - ) - .await -} - pub async fn compact_record(run: &RunSummary) -> Result { json_checked( with_catalog_headers(Request::get(&format!( @@ -220,9 +211,14 @@ pub async fn compact_record(run: &RunSummary) -> Result Result { +pub async fn turns( + run: &RunSummary, + q: &str, + source: &str, + include_analysis: bool, +) -> Result { let url = format!( - "/api/explorer/turns?{}&q={}&source={}&offset=0&limit=500", + "/api/explorer/turns?{}&q={}&source={}&include_analysis={include_analysis}", run.query(), urlencoding::encode(q), urlencoding::encode(source), @@ -235,6 +231,16 @@ pub async fn turns(run: &RunSummary, q: &str, source: &str) -> Result Result { + let url = format!("/api/explorer/run?{}", run.query()); + json_checked( + with_catalog_headers(Request::get(&url)) + .send_tracked() + .await, + ) + .await +} + pub async fn turn_detail(run: &RunSummary, turn_id: i64) -> Result { json_checked( with_catalog_headers(Request::get(&format!( diff --git a/pchronicle-web/src/model.rs b/pchronicle-web/src/model.rs index 1b5ad7822..b63c2c258 100644 --- a/pchronicle-web/src/model.rs +++ b/pchronicle-web/src/model.rs @@ -590,6 +590,8 @@ pub struct TurnSummary { #[derive(Clone, Debug, PartialEq, Deserialize)] pub struct TurnPage { + #[serde(default)] + pub analysis: Option, pub snapshot: PageSnapshot, pub records: Vec, #[serde(default)] diff --git a/pchronicle-web/src/requests.rs b/pchronicle-web/src/requests.rs index 01459cf9b..be1b7790f 100644 --- a/pchronicle-web/src/requests.rs +++ b/pchronicle-web/src/requests.rs @@ -32,6 +32,7 @@ pub struct Entry { pub path: String, pub transport: String, pub started: Instant, + pub finished_ms: Option, pub snapshot: Option, pub diagnostic_error: Option, } @@ -68,6 +69,7 @@ impl Drop for CancelOnDrop { update(&self.0, |e| { if e.transport == "running" { e.transport = "cancelled".into(); + e.finished_ms = Some(e.started.elapsed().as_millis() as u64); } }); } @@ -114,6 +116,7 @@ impl TrackedSend for Request { path, transport: "running".into(), started: Instant::now(), + finished_ms: None, snapshot: None, diagnostic_error: None, }); @@ -121,6 +124,7 @@ impl TrackedSend for Request { let _cancel = CancelOnDrop(id.clone()); let result = self.send().await; update(&id, |e| { + e.finished_ms = Some(e.started.elapsed().as_millis() as u64); e.transport = match &result { Ok(r) if r.ok() => "completed", _ => "failed", @@ -134,36 +138,78 @@ impl TrackedSend for Request { } } -pub fn use_request_polling() { - use_future(|| async { +fn selected_entry<'a>(entries: &'a [Entry], selected: &str) -> Option<&'a Entry> { + if selected.is_empty() { + entries.last() + } else { + entries.iter().find(|entry| entry.id == selected) + } +} + +fn needs_diagnostics(entry: &Entry) -> bool { + entry.diagnostic_error.is_none() && entry.snapshot.as_ref().is_none_or(|s| s.state == "running") +} + +fn use_request_polling(selected: Signal) { + use_future(move || async move { loop { - let pending = REQUESTS - .peek() - .iter() - .filter(|e| { - e.diagnostic_error.is_none() - && e.snapshot.as_ref().is_none_or(|s| s.state == "running") - }) - .map(|e| (e.id.clone(), e.started)) - .collect::>(); - if let Some(token) = observer() { - let queries=pending.into_iter().map(|(id,started)| { let token=token.clone(); async move { - let fetch=async { - let response=Request::get(&format!("/api/requests/{id}")).header(OBSERVER,&token).send().await.map_err(|_|"Diagnostics connection failed".to_string())?; - if response.status()==404 && started.elapsed().as_secs()<10 { return Ok(None); } - if !response.ok() { return Err("Diagnostics expired or are unavailable; the original request may still be running".into()); } - response.json::().await.map(Some).map_err(|_|"Invalid diagnostics response".into()) - }; - let result= match futures_util::future::select(Box::pin(fetch),Box::pin(TimeoutFuture::new(3000))).await { - futures_util::future::Either::Left((r,_))=>r, - _=>Err("Diagnostics timed out; server status is unknown".into()), - }; - update(&id,|e|match result { Ok(Some(s))=>{ - if e.path == "Lookup" { e.method=s.method.clone(); e.path=s.path.clone(); } - e.snapshot=Some(s); - }, Ok(None)=>{}, Err(error)=>e.diagnostic_error=Some(error) }); - }}); - futures_util::future::join_all(queries).await; + // Only the visible inspector needs server progress. Transport status + // for the sidebar is already maintained by send_tracked. + if web_sys::window() + .and_then(|window| window.document()) + .is_none_or(|document| document.hidden()) + { + TimeoutFuture::new(1000).await; + continue; + } + let pending = { + let entries = REQUESTS.peek(); + let selected = selected.peek(); + selected_entry(&entries, &selected) + .filter(|entry| needs_diagnostics(entry)) + .map(|entry| (entry.id.clone(), entry.started)) + }; + if let Some((id, started)) = pending + && let Some(token) = observer() + { + let fetch = async { + let response = Request::get(&format!("/api/requests/{id}")) + .header(OBSERVER, &token) + .send() + .await + .map_err(|_| "Diagnostics connection failed".to_string())?; + if response.status() == 404 && started.elapsed().as_secs() < 10 { + return Ok(None); + } + if !response.ok() { + return Err("Diagnostics expired or are unavailable; the original request may still be running".into()); + } + response + .json::() + .await + .map(Some) + .map_err(|_| "Invalid diagnostics response".into()) + }; + let result = match futures_util::future::select( + Box::pin(fetch), + Box::pin(TimeoutFuture::new(3000)), + ) + .await + { + futures_util::future::Either::Left((r, _)) => r, + _ => Err("Diagnostics timed out; server status is unknown".into()), + }; + update(&id, |e| match result { + Ok(Some(s)) => { + if e.path == "Lookup" { + e.method = s.method.clone(); + e.path = s.path.clone(); + } + e.snapshot = Some(s); + } + Ok(None) => {} + Err(error) => e.diagnostic_error = Some(error), + }); } TimeoutFuture::new(1000).await; } @@ -188,13 +234,42 @@ fn label(name: &str) -> &str { _ => name, } } + +fn entry_elapsed_ms(entry: &Entry) -> u64 { + entry.finished_ms.unwrap_or_else(|| { + entry + .snapshot + .as_ref() + .map(|snapshot| snapshot.elapsed_ms) + .unwrap_or_else(|| entry.started.elapsed().as_millis() as u64) + }) +} + +fn entry_summary(entry: &Entry) -> String { + let mut parts = vec![entry.transport.clone()]; + if let Some(status) = entry.snapshot.as_ref().and_then(|snapshot| snapshot.status) { + parts.push(format!("HTTP {status}")); + } + parts.push(format!("{} ms", entry_elapsed_ms(entry))); + parts.join(" · ") +} + +fn short_request_id(id: &str) -> String { + if id.len() <= 16 { + id.to_owned() + } else { + format!("{}…{}", &id[..8], &id[id.len().saturating_sub(4)..]) + } +} #[component] pub fn RequestIndicator(on_open: EventHandler<()>) -> Element { let entries = REQUESTS.read(); let running = entries .iter() .filter(|e| { - e.transport == "running" || e.snapshot.as_ref().is_some_and(|s| s.state == "running") + e.transport == "running" + || (e.transport == "unknown" + && e.snapshot.as_ref().is_some_and(|s| s.state == "running")) }) .count(); let failed = entries @@ -207,7 +282,9 @@ pub fn RequestIndicator(on_open: EventHandler<()>) -> Element { .iter() .rev() .find(|e| { - e.transport == "running" || e.snapshot.as_ref().is_some_and(|s| s.state == "running") + e.transport == "running" + || (e.transport == "unknown" + && e.snapshot.as_ref().is_some_and(|s| s.state == "running")) }) .and_then(|e| e.snapshot.as_ref()) .and_then(|s| { @@ -236,17 +313,29 @@ fn Phases(snapshot: Snapshot) -> Element { if let Some(worker)=snapshot.worker { h3 { "Worker execution" } Phases { snapshot:*worker } } } } +#[component] +fn RequestRow(entry: Entry, active: bool, on_select: EventHandler) -> Element { + let summary = entry_summary(&entry); + let short_id = short_request_id(&entry.id); + let id = entry.id.clone(); + rsx! { + button { + class: if active { "request-row active" } else { "request-row" }, + onclick: move |_| on_select.call(id.clone()), + strong { "{entry.method} {entry.path}" } + span { class: "request-row-summary", "{summary}" } + span { class: "request-row-id", title: "{entry.id}", "{short_id}" } + } + } +} + #[component] pub fn RequestsPanel() -> Element { let mut selected = use_signal(String::new); let mut lookup = use_signal(String::new); + use_request_polling(selected); let entries = REQUESTS.read().clone(); - let active = if selected().is_empty() { - entries.last() - } else { - entries.iter().find(|e| e.id == selected()) - } - .cloned(); + let active = selected_entry(&entries, &selected()).cloned(); rsx! { section { class:"requests-panel", header { h1 { "Requests" } p { "Inspect this browser’s recent requests, execution stages and failures. Server history is retained for up to 10 minutes." } form { onsubmit:move |event| { @@ -258,7 +347,7 @@ pub fn RequestsPanel() -> Element { if let Some(entry)=entries.iter_mut().find(|e|e.id==id) { entry.diagnostic_error=None; } else { if entries.len()>=LIMIT { entries.remove(0); } - entries.push(Entry { id,method:"GET".into(),path:"Lookup".into(),transport:"unknown".into(),started:Instant::now()-std::time::Duration::from_secs(10),snapshot:None,diagnostic_error:None }); + entries.push(Entry { id,method:"GET".into(),path:"Lookup".into(),transport:"unknown".into(),started:Instant::now()-std::time::Duration::from_secs(10),finished_ms:None,snapshot:None,diagnostic_error:None }); } }, input { aria_label:"Request ID", placeholder:"Find by request ID", value:"{lookup}", oninput:move |e|lookup.set(e.value()) } @@ -268,10 +357,14 @@ pub fn RequestsPanel() -> Element { if !selected().is_empty() && !entries.iter().any(|e|e.id==selected()) { p { role:"status", "This request is not in this browser’s recent history." } } div { class:"requests-grid", aside { aria_label:"Recent requests", - for entry in entries.iter().rev() { button { class:if active.as_ref().is_some_and(|e|e.id==entry.id) {"request-row active"} else {"request-row"}, - onclick:{let id=entry.id.clone();move |_|selected.set(id.clone())}, - strong { "{entry.method} {entry.path}" } span { "{entry.transport} · {entry.id}" } - } } + for entry in entries.iter().rev() { + RequestRow { + key: "{entry.id}", + entry: entry.clone(), + active: active.as_ref().is_some_and(|e| e.id == entry.id), + on_select: move |id| selected.set(id), + } + } } article { if let Some(entry)=active { @@ -291,3 +384,56 @@ pub fn RequestsPanel() -> Element { } } } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn diagnostics_stop_on_terminal_snapshot_or_error() { + let mut entry = Entry { + id: "test".into(), + method: "GET".into(), + path: "test".into(), + transport: "completed".into(), + started: Instant::now(), + finished_ms: None, + snapshot: None, + diagnostic_error: None, + }; + let entries = (0..40) + .map(|id| Entry { + id: id.to_string(), + ..entry.clone() + }) + .collect::>(); + assert_eq!(selected_entry(&entries, "").unwrap().id, "39"); + assert_eq!(selected_entry(&entries, "5").unwrap().id, "5"); + assert!(selected_entry(&entries, "missing").is_none()); + assert!(selected_entry(&[], "").is_none()); + entry.finished_ms = Some(12); + assert_eq!(entry_elapsed_ms(&entry), 12); + // Opening the inspector can retrieve diagnostics for a completed request. + assert!(needs_diagnostics(&entry)); + entry.snapshot = Some(Snapshot { + request_id: "test".into(), + method: "GET".into(), + path: "test".into(), + state: "running".into(), + elapsed_ms: 0, + status: None, + error: None, + note: None, + phases: vec![], + worker: None, + }); + assert!(needs_diagnostics(&entry)); + for state in ["completed", "failed", "cancelled"] { + entry.snapshot.as_mut().unwrap().state = state.into(); + assert!(!needs_diagnostics(&entry)); + } + entry.snapshot = None; + entry.diagnostic_error = Some("unavailable".into()); + assert!(!needs_diagnostics(&entry)); + } +} diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index bad01ccf8..39d8af5e1 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -208,7 +208,6 @@ fn navigate_app_back(app_history_depth: i32) -> bool { } pub fn App() -> Element { - crate::requests::use_request_polling(); let initial_agent = url_param("agent_id"); let initial_session = url_param("session_id"); let initial_root = url_param("root_session_id"); @@ -248,6 +247,7 @@ pub fn App() -> Element { let runs = use_signal(|| None::); let runs_loading = use_signal(|| true); let runs_generation = use_signal(|| 0u64); + let turn_generation = use_signal(|| 0_u64); let mut last_runs_key = use_signal(|| None::); let mut query = use_signal(|| url_param("q").unwrap_or_default()); let mut dataset_filter = @@ -391,24 +391,36 @@ pub fn App() -> Element { }))); }); + let mut detail_request_key = use_signal(|| None::); use_effect(move || { - if analysis().is_none() - && let Some(run) = selected_run() + let Some(run) = selected_run() else { + detail_request_key.set(None); + return; + }; + let key = run.query(); + if detail_request_key.peek().as_deref() == Some(key.as_str()) + && (analysis().is_some() || *detail_loading.peek() || *detail_failed.peek()) { - load_workspace( - run, - turn_query(), - source(), - analysis, - turns, - turn_search, - compact_record, - detail_loading, - detail_failed, - detail_generation, - error, - ); + return; } + detail_request_key.set(Some(key)); + let debounce = *turn_query_debounce_id.peek() + 1; + turn_query_debounce_id.set(debounce); + load_workspace( + run, + turn_query.peek().clone(), + source.peek().clone(), + analysis, + turns, + turn_search, + turn_loading, + turn_generation, + compact_record, + detail_loading, + detail_failed, + detail_generation, + error, + ); }); use_effect(move || { @@ -682,8 +694,12 @@ pub fn App() -> Element { let selected_path = analysis().map(|value| value.run.path).or_else(|| selected_run().map(|run| run.path)).unwrap_or_default(); rsx! { div { class: "pc2-detail-layout", PathExplorer { paged: !query().trim().is_empty(), runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path, loading: runs_loading(), + page_total: runs().map(|page| page.snapshot.total).unwrap_or_default(), + page_offset: runs().map(|page| page.snapshot.offset).unwrap_or_default(), + page_limit: runs().map(|page| page.snapshot.limit).unwrap_or(50), on_path: move |value| { run_path.set(value); offset.set(0); page.set("runs".into()); }, on_view_mode: move |mode| path_list_mode.set(mode), + on_page: move |value| offset.set(value), on_select: move |run: RunSummary| { turn_query.set(query()); selected_run.set(Some(run)); analysis.set(None); turns.set(Vec::new()); turn_search.set(TurnSearchStatus::default()); selected_turn.set(None); drawer_turn.set(None); drawer_details.set(Vec::new()); drawer_turn_id.set(None); drawer_turn_ids.set(Vec::new()); drawer_title.set(String::new()); drawer_loading.set(false); expanded_turn_id.set(None); }, on_open_chat: move |run: RunSummary| { turn_query.set(query()); @@ -698,10 +714,16 @@ pub fn App() -> Element { copilot_open.set(true); }, } - if let (Some(_run), Some(value)) = (selected_run(), analysis()) { + if let Some(run) = selected_run() { RunDetailWorkspace { - run: value.run.clone(), - analysis: value, + failed: detail_failed(), + on_retry: move |_| { + if let Some(run) = selected_run.peek().clone() { + load_workspace(run,turn_query(),source(),analysis,turns,turn_search,turn_loading,turn_generation,compact_record,detail_loading,detail_failed,detail_generation,error); + } + }, + run, + analysis: analysis(), compact_record: compact_record(), turns: turns(), search: turn_search(), @@ -727,7 +749,12 @@ pub fn App() -> Element { on_view: move |value: String| { trace_mode.set(normalize_trace_view(&value).to_string()); }, - on_source: move |value| source.set(value), + on_source: move |value: String| { + source.set(value.clone()); + if let Some(run) = selected_run() { + load_turns(run, turn_query(), value, turns, turn_search, turn_loading, error, turn_generation); + } + }, on_query: move |value: String| { turn_query.set(value.clone()); let request_id = turn_query_debounce_id() + 1; @@ -745,6 +772,7 @@ pub fn App() -> Element { turn_search, turn_loading, error, + turn_generation, ); } }); @@ -763,6 +791,7 @@ pub fn App() -> Element { turn_search, turn_loading, error, + turn_generation, ); } }, @@ -821,17 +850,6 @@ pub fn App() -> Element { page.set("tools".into()); }, } - } else if detail_failed() && !detail_loading() { - div { class: "pc2-loading", role: "alert", - strong { "Could not load run details" } - p { "Open Requests to inspect the failed stage, or retry this run." } - button { class: "button", onclick: move |_| { - if let Some(run)=selected_run.peek().clone() { - load_workspace(run,turn_query(),source(),analysis,turns,turn_search,compact_record,detail_loading,detail_failed,detail_generation,error); - } - }, "Retry" } - button { class: "button", onclick: move |_| page.set("requests".into()), "View requests" } - } } else { LoadingWorkspace { label: "Loading run details…" } } } } } @@ -839,8 +857,12 @@ pub fn App() -> Element { let path_runs = runs().map(|value| value.path_index).unwrap_or_default(); rsx! { div { class: "pc2-runs-layout", PathExplorer { paged: !query().trim().is_empty(), runs: path_runs, chat_sessions: assistant_index().sessions.clone(), view_mode: path_list_mode(), selected_path: run_path(), loading: runs_loading(), + page_total: runs().map(|value| value.snapshot.total).unwrap_or_default(), + page_offset: runs().map(|value| value.snapshot.offset).unwrap_or_default(), + page_limit: runs().map(|value| value.snapshot.limit).unwrap_or(50), on_path: move |value| { run_path.set(value); offset.set(0); }, on_view_mode: move |mode| path_list_mode.set(mode), + on_page: move |value| offset.set(value), on_select: move |run: RunSummary| { turn_query.set(query()); selected_run.set(Some(run)); analysis.set(None); turns.set(Vec::new()); turn_search.set(TurnSearchStatus::default()); selected_turn.set(None); drawer_turn.set(None); drawer_details.set(Vec::new()); drawer_turn_id.set(None); drawer_turn_ids.set(Vec::new()); drawer_title.set(String::new()); drawer_loading.set(false); expanded_turn_id.set(None); detail_mode.set("trace".into()); page.set("detail".into()); }, on_open_chat: move |run: RunSummary| { turn_query.set(query()); @@ -1257,6 +1279,8 @@ fn load_workspace( mut analysis: Signal>, mut turns: Signal>, mut turn_search: Signal, + mut turn_loading: Signal, + mut turn_generation: Signal, mut compact_record: Signal>, mut loading: Signal, mut failed: Signal, @@ -1267,41 +1291,83 @@ fn load_workspace( let requested = *generation.peek() + 1; generation.set(requested); compact_record.set(None); + analysis.set(None); failed.set(false); loading.set(true); + turn_loading.set(true); + turns.set(Vec::new()); + let step_requested = *turn_generation.peek() + 1; + turn_generation.set(step_requested); spawn(async move { - let work = async { - let (next_analysis, next_turns) = futures_util::try_join!( - api::run_analysis(&run), - api::turns(&run, &query, &source) - )?; - let record = if next_analysis.run.is_compact_jsonl() { - Some(api::compact_record(&next_analysis.run).await?) - } else { - None - }; - Ok::<_, api::ApiFailure>((next_analysis, next_turns, record)) + // Steps first (no embedded analysis), paint the timeline, then load + // statistics. The follow-up /run hits the worker trajectory memo when + // the pool reuses the same child, so aggregates are usually cheap. + let turns_result = match futures_util::future::select( + Box::pin(api::turns(&run, &query, &source, false)), + Box::pin(TimeoutFuture::new(65_000)), + ) + .await + { + futures_util::future::Either::Left((result, _)) => result, + _ => Err(api::ApiFailure::network( + "Loading steps timed out. Open Requests to inspect server progress, then retry.", + )), }; - let result = match futures_util::future::select( - Box::pin(work), + if *generation.peek() != requested { + return; + } + match turns_result { + Ok(page) => { + if *turn_generation.peek() == step_requested { + turns.set(page.records); + turn_search.set(page.search); + turn_loading.set(false); + } + } + Err(failure) => { + if *turn_generation.peek() == step_requested { + turn_loading.set(false); + } + error.set(Some(workspace_notice(&failure))); + loading.set(false); + return; + } + } + + if run.is_compact_jsonl() { + match api::compact_record(&run).await { + Ok(record) if *generation.peek() == requested => { + compact_record.set(Some(record)); + } + Err(failure) if *generation.peek() == requested => { + error.set(Some(workspace_notice(&failure))); + } + _ => {} + } + if *generation.peek() == requested { + loading.set(false); + } + return; + } + + let analysis_result = match futures_util::future::select( + Box::pin(api::run_analysis(&run)), Box::pin(TimeoutFuture::new(65_000)), ) .await { futures_util::future::Either::Left((result, _)) => result, _ => Err(api::ApiFailure::network( - "Run details timed out. Open Requests to inspect server progress, then retry.", + "Run statistics timed out. Steps remain available; open Requests to inspect progress.", )), }; if *generation.peek() != requested { return; } - match result { - Ok((next_analysis, next_turns, record)) => { + match analysis_result { + Ok(next_analysis) => { analysis.set(Some(next_analysis)); - turns.set(next_turns.records); - turn_search.set(next_turns.search); - compact_record.set(record); + failed.set(false); } Err(failure) => { failed.set(true); @@ -1320,10 +1386,17 @@ fn load_turns( mut turn_search: Signal, mut loading: Signal, mut error: Signal>, + mut generation: Signal, ) { + let requested = *generation.peek() + 1; + generation.set(requested); loading.set(true); spawn(async move { - match api::turns(&run, &query, &source).await { + let result = api::turns(&run, &query, &source, false).await; + if *generation.peek() != requested { + return; + } + match result { Ok(value) => { turns.set(value.records); turn_search.set(value.search); @@ -1574,12 +1647,31 @@ fn PathExplorer( selected_path: String, loading: bool, paged: bool, + #[props(default)] page_total: usize, + #[props(default)] page_offset: usize, + #[props(default)] page_limit: usize, on_path: EventHandler, on_view_mode: EventHandler, on_select: EventHandler, on_open_chat: EventHandler, + on_page: EventHandler, ) -> Element { let import_path_tree = build_import_path_tree(&runs); + let limit = page_limit.max(1); + let total = if page_total > 0 { + page_total + } else { + runs.len() + }; + let page_count = total.div_ceil(limit).max(1); + let current_page = (page_offset / limit).min(page_count.saturating_sub(1)); + let page_start = if total == 0 { + 0 + } else { + page_offset.saturating_add(1) + }; + let page_end = page_offset.saturating_add(runs.len()).min(total); + let show_pager = !paged && total > limit; rsx! { aside { class: "pc2-path-explorer", header { div { strong { "Run paths" } span { if paged { "Search results on this page" } else if view_mode == PathListMode::Flat { "All runs in this dataset" } else { "Tree by import path" } } } @@ -1590,7 +1682,7 @@ fn PathExplorer( span { "{runs.len()}" } } div { class: "pc2-path-tree", - button { class: if selected_path.is_empty() { "pc2-path-all active" } else { "pc2-path-all" }, onclick: move |_| on_path.call(String::new()), span { class: "pc2-path-icon root", "⌂" } strong { if paged { "Search results" } else { "All runs" } } code { "{runs.len()}" } } + button { class: if selected_path.is_empty() { "pc2-path-all active" } else { "pc2-path-all" }, onclick: move |_| on_path.call(String::new()), span { class: "pc2-path-icon root", "⌂" } strong { if paged { "Search results" } else { "All runs" } } code { "{total}" } } if loading && runs.is_empty() { div { class: "pc2-path-loading", span { class: "spinner" } "Loading paths…" } } else if runs.is_empty() { div { class: "pc2-path-empty", "No captured run paths." } } else if view_mode == PathListMode::Flat { @@ -1603,7 +1695,38 @@ fn PathExplorer( } } } - footer { if paged { "Showing the current search page." } else if view_mode == PathListMode::Flat { "Showing all runs in this dataset." } else { "Tree follows the imported path." } } + footer { class: if show_pager { "pc2-path-footer paged" } else { "pc2-path-footer" }, + if show_pager { + label { class: "pc2-path-page", + span { "Page" } + select { + value: "{current_page}", + aria_label: "Jump to run path page", + onchange: move |event| { + if let Ok(index) = event.value().parse::() { + on_page.call(index.saturating_mul(limit)); + } + }, + for index in 0..page_count { + option { value: "{index}", selected: index == current_page, + { + let start = index.saturating_mul(limit).saturating_add(1); + let end = index.saturating_mul(limit).saturating_add(limit).min(total); + format!("{}/{} · {start}–{end}", index + 1, page_count) + } + } + } + } + } + span { "{page_start}–{page_end} of {total}" } + } else if paged { + "Showing the current search page." + } else if view_mode == PathListMode::Flat { + "Showing all runs in this dataset." + } else { + "Tree follows the imported path." + } + } } } } @@ -1848,8 +1971,10 @@ fn StatusBadge(value: String) -> Element { #[component] #[allow(clippy::too_many_arguments)] fn RunDetailWorkspace( + failed: bool, + on_retry: EventHandler, run: RunSummary, - analysis: RunAnalysis, + analysis: Option, compact_record: Option, turns: Vec, search: TurnSearchStatus, @@ -1909,6 +2034,11 @@ fn RunDetailWorkspace( }; rsx! { section { class: if compact_header() { "pc2-detail is-condensed" } else { "pc2-detail" }, + if failed { + div { role: "alert", "Run statistics could not be loaded. Steps remain available." + button { class: "button", onclick: on_retry, "Retry" } + } + } header { class: "pc2-detail-head", div { class: "pc2-detail-title", button { class: "pc2-back", onclick: on_back, "← Runs" } div { p { "{run.agent_id}" } h1 { title: "{run.session_id}", "{run.session_id}" } div { StatusBadge { value: run.status.clone() } if let Some(root) = &run.root_session_id { code { "root {short(root, 24)}" } } } } } div { class: "pc2-head-actions", @@ -1934,22 +2064,29 @@ fn RunDetailWorkspace( } } } else { - MetricsStrip { analysis: analysis.clone() } + if let Some(value) = analysis.clone() { + MetricsStrip { analysis: value.clone() } if detail_mode == "trace" { CompactOverviewStrip { - analysis: analysis.clone(), + analysis: value, turns: turns.clone(), on_open_analysis: move |_| on_detail_mode.call("analysis".into()), } } + } else if loading { + div { class: "pc2-inline-loading", role: "status", "Loading run statistics…" } + } nav { class: "pc2-detail-tabs", aria_label: "Run detail view", button { class: if detail_mode == "trace" { "active" } else { "" }, onclick: move |_| on_detail_mode.call("trace".into()), {TIMELINE} } button { class: if detail_mode == "analysis" { "active" } else { "" }, onclick: move |_| on_detail_mode.call("analysis".into()), "Analysis" } - span { "{turns.len()} of {analysis.turn_count} steps loaded for interactive charts" } + if let Some(value) = &analysis { + span { "{turns.len()} of {value.turn_count} steps loaded for interactive charts" } + } else { span { "{turns.len()} steps loaded" } } } if detail_mode == "analysis" { + if let Some(value) = analysis { AnalysisWorkspace { - analysis: analysis.clone(), + analysis: value, turns: turns.clone(), on_turn: move |id| { on_turn.call(id); @@ -1957,6 +2094,7 @@ fn RunDetailWorkspace( }, on_scroll: on_detail_scroll, } + } else { div { role: "status", "Run statistics are not available yet." } } } else { section { class: "pc2-trace-surface pc2-inline-trace", div { class: "pc2-trace-toolbar", @@ -1972,8 +2110,8 @@ fn RunDetailWorkspace( } } div { id: RUN_DETAIL_SCROLL_ID, class: "pc2-turn-list pc2-span-scroll", onscroll: on_detail_scroll, - if loading { div { class: "pc2-inline-loading", span { class: "spinner" } "Refreshing run details…" } } - if turns.is_empty() { div { class: "pc2-empty", strong { "No visible steps" } span { "No loaded steps match this filter." } } } + if turn_loading { div { class: "pc2-inline-loading", role: "status", span { class: "spinner" } "Loading steps…" } } + if turns.is_empty() && !turn_loading { div { class: "pc2-empty", strong { "No visible steps" } span { "No loaded steps match this filter." } } } else { TrajectoryView { turns, expanded_turn_id, detail: selected, loading: turn_loading, view: view_for_list, source: source_for_list, query: query_for_list, on_turn, on_open_drawer } } } } @@ -3295,6 +3433,38 @@ mod tests { ); } + #[test] + fn detail_renders_steps_while_statistics_are_pending() { + let mut dom = VirtualDom::new(|| { + let turn: TurnSummary = serde_json::from_value(serde_json::json!({ + "id": 7, "source": "user", "kind": null, "timestamp": null, + "call_id": null, "preview": "ready-step-before-statistics", "model_name": null, + "latency_ms": null, "ttft_ms": null, "prompt_tokens": null, + "completion_tokens": null, "total_tokens": null, "tool_names": [], + "event_seqs": [], "has_error": false + })) + .unwrap(); + rsx! { RunDetailWorkspace { + failed: false, on_retry: |_| {}, + run: run_at("a/run"), analysis: None, compact_record: None, + turns: vec![turn], search: TurnSearchStatus::default(), + selected: None, drawer: None, drawer_details: vec![], drawer_ids: vec![], + drawer_title: String::new(), drawer_loading: false, expanded_turn_id: None, + loading: true, turn_loading: false, detail_mode: "trace".to_string(), + view: "steps".to_string(), source: "all".to_string(), query: String::new(), + on_back: |_| {}, on_detail_mode: |_| {}, on_view: |_| {}, on_source: |_| {}, + on_query: |_| {}, on_apply_query: |_| {}, on_turn: |_| {}, + on_open_drawer: |_| {}, on_close_drawer: |_| {}, on_open_copilot: |_| {}, + on_analyze: |_| {}, + } } + }); + let mutations = format!("{:?}", dom.rebuild_to_vec()); + assert!( + mutations.contains("ready-step-before-statistics"), + "{mutations}" + ); + } + fn drain_effects(dom: &mut VirtualDom) { use futures_util::FutureExt; for _ in 0..8 { From cef1121c8f53c4479b25a47888ea50ca22675e3c Mon Sep 17 00:00:00 2001 From: Reiase Date: Fri, 18 Sep 2026 16:05:44 +0800 Subject: [PATCH 19/22] Refactor object store and cache management for improved readability - Cleaned up formatting in `object_store_io_gate.rs` and `persistent_cache.rs` to enhance code readability by aligning method calls and reducing line breaks. - Simplified the `lane` assignment in the `Gate` struct for better clarity. - Improved the structure of `upsert` method calls in `PersistentCache` to maintain consistent formatting. This commit aims to enhance code maintainability and readability across the object store and cache management components in the pChronicle application. --- .../src/server/catalog_worker.rs | 3 +-- crates/persisting-pchronicle/src/storage.rs | 5 +++-- .../src/store/blockcache/adapter.rs | 17 +++++++---------- .../src/store/object_store_io_gate.rs | 6 +++++- .../src/store/persistent_cache.rs | 14 +++++++++++--- 5 files changed, 27 insertions(+), 18 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index 07fa1955f..fd7905e55 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -349,8 +349,7 @@ impl Worker { } builder.create(&cache)?; builder.create(&blocks)?; - let mut child = - command(std::env::current_exe()?, home.path(), &cache, &blocks).spawn()?; + let mut child = command(std::env::current_exe()?, home.path(), &cache, &blocks).spawn()?; let input = child.stdin.take().context("worker stdin missing")?; let output = child.stdout.take().context("worker stdout missing")?; let mut worker = Self { diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 4db64032b..0b9d52ebf 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -65,8 +65,9 @@ pub use crate::store::index_build_progress::{ #[cfg(feature = "lance-store")] pub use crate::store::object_store_io_gate::{ IoKind as ObjectStoreIoKind, ObjectStoreGateSnapshot, ObjectStoreThrottleEvent, - ObjectStoreThrottleHookGuard, format_aimd_flow_label as format_object_store_aimd_flow_label, - foreground_object_store_demand, install_throttle_hook as install_object_store_throttle_hook, + ObjectStoreThrottleHookGuard, foreground_object_store_demand, + format_aimd_flow_label as format_object_store_aimd_flow_label, + install_throttle_hook as install_object_store_throttle_hook, snapshot as object_store_gate_snapshot, wait_for_foreground_object_store_idle, with_background_object_store_io, }; diff --git a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs index a844ce8d4..57bfd18af 100644 --- a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs +++ b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs @@ -260,10 +260,7 @@ impl CachedObjectStore { { // Remembered metadata describes bytes the backend no // longer serves; the next read must ask again. - self.heads - .lock() - .expect("head memo") - .forget(path.as_ref()); + self.heads.lock().expect("head memo").forget(path.as_ref()); return Err(object_store::Error::Precondition { path: path.to_string(), source: "object changed while reading cached block".into(), @@ -325,10 +322,11 @@ impl ObjectStore for CachedObjectStore { ) .await?; let head = (head.meta, head.attributes); - self.heads - .lock() - .expect("head memo") - .insert(head_key.clone(), head.0.clone(), head.1.clone()); + self.heads.lock().expect("head memo").insert( + head_key.clone(), + head.0.clone(), + head.1.clone(), + ); head } }; @@ -544,8 +542,7 @@ mod tests { self.inner.put_multipart_opts(p, o).await } async fn get_opts(&self, p: &Path, o: GetOptions) -> ObjectResult { - self.gets - .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + self.gets.fetch_add(1, std::sync::atomic::Ordering::Relaxed); self.inner.get_opts(p, o).await } fn delete_stream( diff --git a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs index 55c9241a7..94b53aec7 100644 --- a/crates/persisting-pchronicle/src/store/object_store_io_gate.rs +++ b/crates/persisting-pchronicle/src/store/object_store_io_gate.rs @@ -331,7 +331,11 @@ fn gate() -> &'static Gate { configured }; Gate { - lane: if background { "background" } else { "foreground" }, + lane: if background { + "background" + } else { + "foreground" + }, concurrency, states: Mutex::new(HashMap::new()), } diff --git a/crates/persisting-pchronicle/src/store/persistent_cache.rs b/crates/persisting-pchronicle/src/store/persistent_cache.rs index 592ad9353..c78dc8466 100644 --- a/crates/persisting-pchronicle/src/store/persistent_cache.rs +++ b/crates/persisting-pchronicle/src/store/persistent_cache.rs @@ -287,13 +287,18 @@ mod tests { assert!(fragments <= MAX_FRAGMENTS, "{fragments} fragments"); // Compaction rewrites fragments, so every entry must survive it. - cache.upsert(&"key-007".into(), &"rewritten".into(), &[]).await?; + cache + .upsert(&"key-007".into(), &"rewritten".into(), &[]) + .await?; cache.upsert(&"gone".into(), &String::new(), &[]).await?; cache .upsert(&"key-000".into(), &"kept".into(), &["gone".into()]) .await?; drop(cache); - let values = PersistentCache::::open(path).await.values().await; + let values = PersistentCache::::open(path) + .await + .values() + .await; assert_eq!(values.len(), 80); assert_eq!(values["key-007"], "rewritten"); assert_eq!(values["key-000"], "kept"); @@ -372,7 +377,10 @@ mod tests { ) .await?; drop(cache); - let values = PersistentCache::::open(path).await.values().await; + let values = PersistentCache::::open(path) + .await + .values() + .await; assert_eq!(values["a"], "last"); // A key this batch writes was just observed; its earlier retirement // inside the same batch must not delete it. From 6a8abd1f4eb7a7ed14b648cebb287e4f90408932 Mon Sep 17 00:00:00 2001 From: Reiase Date: Fri, 18 Sep 2026 16:19:33 +0800 Subject: [PATCH 20/22] Refactor error handling and improve cache management in pChronicle - Simplified error handling in various modules by removing unnecessary conversions, enhancing clarity and reducing boilerplate code. - Improved cache management logic in `opendal_store.rs` and `persistent_cache.rs` to streamline operator insertion and batch processing. - Enhanced readability by restructuring conditional statements and aligning method calls. This commit aims to enhance code maintainability and performance across the pChronicle application. --- .../src/server/catalog_worker.rs | 8 +-- .../src/server/mod.rs | 5 +- .../src/server/request_progress.rs | 50 ++++++++-------- .../src/server/ui_cache.rs | 2 +- .../src/store/blockcache/block.rs | 10 ++-- .../src/store/catalog/discovery.rs | 19 +++--- .../src/store/catalog/manifest_cache.rs | 9 ++- .../src/store/opendal_store.rs | 17 +++--- .../src/store/persistent_cache.rs | 59 ++++++++++--------- pchronicle-web/src/workspace.rs | 1 + 10 files changed, 91 insertions(+), 89 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs index fd7905e55..ff4f96fab 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog_worker.rs @@ -169,10 +169,10 @@ impl PoolState { if guard.idle_since.elapsed() < WORKER_IDLE_TIMEOUT { continue; } - if let Some(worker) = guard.worker.as_mut() { - if worker.child.kill().await.is_err() { - continue; - } + if let Some(worker) = guard.worker.as_mut() + && worker.child.kill().await.is_err() + { + continue; } slots.remove(index); } diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 93d9ecaef..a4891eaf2 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -55,6 +55,9 @@ fn fail(request_id: &RequestId, handler: &'static str, error: anyhow::Error) -> #[cfg(test)] use problem::BoundaryCode; +type CachedTrajectory = (String, Arc); +type TrajectoryCache = Arc>>; + #[derive(Clone)] struct AppState { config: Arc, @@ -62,7 +65,7 @@ struct AppState { /// Serializes global refreshes and stores the next automatic retry time. catalog_refresh: Arc>, catalog_refresh_interval: Duration, - trajectory_cache: Arc)>>>, + trajectory_cache: TrajectoryCache, trajectory_flights: Arc>>>, /// Gateway-backed Warehouses read canonical events from the latest /// manifest for single-trace observation, independent of projection idle. diff --git a/crates/persisting-pchronicle-cli/src/server/request_progress.rs b/crates/persisting-pchronicle-cli/src/server/request_progress.rs index d66a4d550..70892c1ae 100644 --- a/crates/persisting-pchronicle-cli/src/server/request_progress.rs +++ b/crates/persisting-pchronicle-cli/src/server/request_progress.rs @@ -155,14 +155,14 @@ impl Progress { p.elapsed_ms += ms(v.phase_started.elapsed()); } } - if let Some(w) = s.worker.as_mut() { - if w.state == "running" { - let elapsed = ms(v.worker_updated.elapsed()); - w.elapsed_ms += elapsed; - for p in &mut w.phases { - if p.state == "running" { - p.elapsed_ms += elapsed; - } + if let Some(w) = s.worker.as_mut() + && w.state == "running" + { + let elapsed = ms(v.worker_updated.elapsed()); + w.elapsed_ms += elapsed; + for p in &mut w.phases { + if p.state == "running" { + p.elapsed_ms += elapsed; } } } @@ -179,12 +179,10 @@ impl Progress { if v.snapshot.state != "running" { return; } - let state = if status.is_none() { - "cancelled" - } else if status.unwrap() >= 400 || error.is_some() { - "failed" - } else { - "completed" + let state = match status { + None => "cancelled", + Some(code) if code >= 400 || error.is_some() => "failed", + Some(_) => "completed", }; let elapsed = ms(v.phase_started.elapsed()); for p in &mut v.snapshot.phases { @@ -195,13 +193,14 @@ impl Progress { p.state = "skipped".into(); } } - if let Some(w) = v.snapshot.worker.as_mut() { - if w.state == "running" && state != "completed" { - w.state = state.into(); - for p in &mut w.phases { - if p.state == "running" { - p.state = state.into(); - } + if let Some(w) = v.snapshot.worker.as_mut() + && w.state == "running" + && state != "completed" + { + w.state = state.into(); + for p in &mut w.phases { + if p.state == "running" { + p.state = state.into(); } } } @@ -239,8 +238,8 @@ impl Registry { .elapsed() < TTL }); - if entries.len() >= LIMIT { - if let Some(oldest) = entries + if entries.len() >= LIMIT + && let Some(oldest) = entries .iter() .max_by_key(|(_, p)| { p.0.lock() @@ -249,9 +248,8 @@ impl Registry { .elapsed() }) .map(|(k, _)| k.clone()) - { - entries.remove(&oldest); - } + { + entries.remove(&oldest); } let id = progress.snapshot().request_id; entries.insert((owner, id), progress); diff --git a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs index b7a358a2b..10dd0ca43 100644 --- a/crates/persisting-pchronicle-cli/src/server/ui_cache.rs +++ b/crates/persisting-pchronicle-cli/src/server/ui_cache.rs @@ -721,7 +721,7 @@ async fn refresh_tree( } fn finish(pending: &Pending, key: &TreeKey, outcome: std::result::Result<(), String>) { - if let Some(waiters) = lock_recover(&pending).remove(key) { + if let Some(waiters) = lock_recover(pending).remove(key) { for waiter in waiters { let _ = waiter.send(outcome.clone()); } diff --git a/crates/persisting-pchronicle/src/store/blockcache/block.rs b/crates/persisting-pchronicle/src/store/blockcache/block.rs index a13722ed4..05f68c5bb 100644 --- a/crates/persisting-pchronicle/src/store/blockcache/block.rs +++ b/crates/persisting-pchronicle/src/store/blockcache/block.rs @@ -89,11 +89,11 @@ impl BlockCache { }; let flight_guard = flight.lock().await; let result = async { - if let Ok(bytes) = tokio::fs::read(path).await { - if bytes.len() == expected { - self.counters.hits.fetch_add(1, Ordering::Relaxed); - return Ok(Bytes::from(bytes)); - } + if let Ok(bytes) = tokio::fs::read(path).await + && bytes.len() == expected + { + self.counters.hits.fetch_add(1, Ordering::Relaxed); + return Ok(Bytes::from(bytes)); } let bytes = fetch.await?; if bytes.len() != expected { diff --git a/crates/persisting-pchronicle/src/store/catalog/discovery.rs b/crates/persisting-pchronicle/src/store/catalog/discovery.rs index 50f82b10d..5416cc999 100644 --- a/crates/persisting-pchronicle/src/store/catalog/discovery.rs +++ b/crates/persisting-pchronicle/src/store/catalog/discovery.rs @@ -974,15 +974,16 @@ async fn discover_object_candidate_at( let mut parts = file.split('/'); loop { budget.observe_entry()?; - if current == file && is_json_candidate(Path::new(file)) { - if let Some(entry) = store.stat_file(file).await? { - budget.observe_source()?; - return Ok(vec![Candidate::RemoteFile { - file: file.into(), - store: store.clone(), - meta: RemoteObjectMeta::from(entry), - }]); - } + if current == file + && is_json_candidate(Path::new(file)) + && let Some(entry) = store.stat_file(file).await? + { + budget.observe_source()?; + return Ok(vec![Candidate::RemoteFile { + file: file.into(), + store: store.clone(), + meta: RemoteObjectMeta::from(entry), + }]); } match probe_object_prefix(&store, uri, ¤t, root_source_path(¤t)).await? { Some(ObjectProbe::Source(candidate)) => { diff --git a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs index c5a9aa86d..86493d910 100644 --- a/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs +++ b/crates/persisting-pchronicle/src/store/catalog/manifest_cache.rs @@ -469,14 +469,13 @@ impl ManifestCache { root.clone() }) }; - if let Some(root) = root { - if let Err(error) = self + if let Some(root) = root + && let Err(error) = self .disk .upsert(&key_prefix.to_owned(), &serde_json::to_value(root)?, &[]) .await - { - tracing::warn!(target: "pchronicle.serve", %error, "mount completeness persistence failed"); - } + { + tracing::warn!(target: "pchronicle.serve", %error, "mount completeness persistence failed"); } tracing::info!(target: "pchronicle.serve", %key_prefix, refreshed = report.refreshed_directories, partial = report.partial, diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 46394af78..26c50a87a 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -156,15 +156,15 @@ impl OperatorRegistry { } fn insert(&mut self, key: String, operator: Operator, now: Instant) { - if !self.entries.contains_key(&key) && self.entries.len() >= MAX_CACHED_OPERATORS { - if let Some(oldest) = self + if !self.entries.contains_key(&key) + && self.entries.len() >= MAX_CACHED_OPERATORS + && let Some(oldest) = self .entries .iter() .min_by_key(|(_, (_, used))| *used) .map(|(key, _)| key.clone()) - { - self.entries.remove(&oldest); - } + { + self.entries.remove(&oldest); } self.entries.insert(key, (operator, now)); } @@ -325,7 +325,7 @@ impl Store { { return Ok(None); } - Err(error) => return Err(error.into()), + Err(error) => return Err(error), }; let path_owned = path.to_owned(); let bytes = self @@ -392,7 +392,7 @@ impl Store { } }) .await; - result.map_err(|error| { + result.inspect_err(|error| { if let Some(error) = error.downcast_ref::() && is_conflict(error) { @@ -405,7 +405,6 @@ impl Store { "conditional object write conflict (If-Match)" ); } - error.into() }) } @@ -475,7 +474,7 @@ impl Store { { Ok(None) } - Err(error) => Err(error.into()), + Err(error) => Err(error), } } diff --git a/crates/persisting-pchronicle/src/store/persistent_cache.rs b/crates/persisting-pchronicle/src/store/persistent_cache.rs index c78dc8466..9c2721baf 100644 --- a/crates/persisting-pchronicle/src/store/persistent_cache.rs +++ b/crates/persisting-pchronicle/src/store/persistent_cache.rs @@ -265,6 +265,36 @@ where Ok(values) } } + +fn cache_batch(keys: Vec, payloads: Vec) -> Result { + let schema = Arc::new(Schema::new(vec![ + Field::new("key", DataType::Utf8, false), + Field::new("payload", DataType::Utf8, false), + Field::new("schema_version", DataType::Utf8, false), + ])); + let versions = vec![SCHEMA_VERSION; keys.len()]; + Ok(RecordBatch::try_new( + schema, + vec![ + Arc::new(StringArray::from(keys)) as _, + Arc::new(StringArray::from(payloads)) as _, + Arc::new(StringArray::from(versions)) as _, + ], + )?) +} + +fn text_column(batch: &RecordBatch, name: &str) -> Result> { + let array = batch + .column(batch.schema().index_of(name)?) + .as_any() + .downcast_ref::() + .with_context(|| format!("persistent cache column {name} must be Utf8"))?; + anyhow::ensure!(array.null_count() == 0, "null persistent cache field"); + Ok((0..array.len()) + .map(|index| array.value(index).to_owned()) + .collect()) +} + #[cfg(test)] mod tests { use super::*; @@ -388,32 +418,3 @@ mod tests { Ok(()) } } - -fn cache_batch(keys: Vec, payloads: Vec) -> Result { - let schema = Arc::new(Schema::new(vec![ - Field::new("key", DataType::Utf8, false), - Field::new("payload", DataType::Utf8, false), - Field::new("schema_version", DataType::Utf8, false), - ])); - let versions = vec![SCHEMA_VERSION; keys.len()]; - Ok(RecordBatch::try_new( - schema, - vec![ - Arc::new(StringArray::from(keys)) as _, - Arc::new(StringArray::from(payloads)) as _, - Arc::new(StringArray::from(versions)) as _, - ], - )?) -} - -fn text_column(batch: &RecordBatch, name: &str) -> Result> { - let array = batch - .column(batch.schema().index_of(name)?) - .as_any() - .downcast_ref::() - .with_context(|| format!("persistent cache column {name} must be Utf8"))?; - anyhow::ensure!(array.null_count() == 0, "null persistent cache field"); - Ok((0..array.len()) - .map(|index| array.value(index).to_owned()) - .collect()) -} diff --git a/pchronicle-web/src/workspace.rs b/pchronicle-web/src/workspace.rs index 39d8af5e1..2f230594f 100644 --- a/pchronicle-web/src/workspace.rs +++ b/pchronicle-web/src/workspace.rs @@ -1378,6 +1378,7 @@ fn load_workspace( }); } +#[allow(clippy::too_many_arguments)] fn load_turns( run: RunSummary, query: String, From 109e011e73c9f393712ce408eba3839ab38229a2 Mon Sep 17 00:00:00 2001 From: Reiase Date: Fri, 18 Sep 2026 16:29:57 +0800 Subject: [PATCH 21/22] Enhance object store configuration and cache management - Added `RetryPatience` and `set_retry_patience` to `opendal_store` for improved retry behavior in object store operations. - Introduced `PersistentCache` export in `mod.rs` to streamline cache management. - Refactored hash generation in `fingerprint` and `backend_identity_from_env` methods to use `unwrap_or_default()` for safer error handling. This commit aims to improve the configurability and reliability of object store interactions and cache management in the pChronicle application. --- crates/persisting-pchronicle/src/storage.rs | 1 + .../src/store/blockcache/adapter.rs | 50 ++++++++++++------- crates/persisting-pchronicle/src/store/mod.rs | 1 + .../src/store/opendal_store.rs | 2 +- 4 files changed, 35 insertions(+), 19 deletions(-) diff --git a/crates/persisting-pchronicle/src/storage.rs b/crates/persisting-pchronicle/src/storage.rs index 0b9d52ebf..a1c93e03f 100644 --- a/crates/persisting-pchronicle/src/storage.rs +++ b/crates/persisting-pchronicle/src/storage.rs @@ -3,6 +3,7 @@ pub type Result = anyhow::Result; #[cfg(feature = "lance-store")] pub use crate::store::opendal_store::StoreConfig; +#[cfg(feature = "lance-store")] pub use crate::store::opendal_store::{RetryPatience, set_retry_patience}; /// Parse an integer byte size with binary IEC suffixes. diff --git a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs index 57bfd18af..772473084 100644 --- a/crates/persisting-pchronicle/src/store/blockcache/adapter.rs +++ b/crates/persisting-pchronicle/src/store/blockcache/adapter.rs @@ -117,7 +117,7 @@ fn backend_identity_from_env() -> String { key.starts_with("AWS_") || key.starts_with("AZURE_") || key.starts_with("GOOGLE_") }) .collect(); - blake3::hash(serde_json::to_string(&backend).unwrap().as_bytes()) + blake3::hash(&serde_json::to_vec(&backend).unwrap_or_default()) .to_hex() .to_string() } @@ -231,6 +231,12 @@ impl CachedObjectStore { } } + fn heads(&self) -> std::sync::MutexGuard<'_, HeadMemo> { + self.heads + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + } + async fn read_block( &self, path: &Path, @@ -260,7 +266,7 @@ impl CachedObjectStore { { // Remembered metadata describes bytes the backend no // longer serves; the next read must ask again. - self.heads.lock().expect("head memo").forget(path.as_ref()); + self.heads().forget(path.as_ref()); return Err(object_store::Error::Precondition { path: path.to_string(), source: "object changed while reading cached block".into(), @@ -305,7 +311,7 @@ impl ObjectStore for CachedObjectStore { // checked locally below, so remembered metadata reaches the same verdict // a conditional round trip would have returned. let head_key = (p.to_string(), o.version.clone()); - let remembered = self.heads.lock().expect("head memo").get(&head_key); + let remembered = self.heads().get(&head_key); let (head_meta, head_attributes) = match remembered { Some(head) => head, None => { @@ -322,11 +328,8 @@ impl ObjectStore for CachedObjectStore { ) .await?; let head = (head.meta, head.attributes); - self.heads.lock().expect("head memo").insert( - head_key.clone(), - head.0.clone(), - head.1.clone(), - ); + self.heads() + .insert(head_key.clone(), head.0.clone(), head.1.clone()); head } }; @@ -347,15 +350,16 @@ impl ObjectStore for CachedObjectStore { if o.version.is_some() && o.version != head_meta.version { return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; } - let range = o - .range - .as_ref() - .unwrap() - .as_range(head_meta.size) - .map_err(|source| object_store::Error::Generic { - store: "pchronicle-cache", - source: Box::new(source), - })?; + let Some(range) = o.range.as_ref() else { + return remote_request(&self.io_scope, self.inner.get_opts(p, o)).await; + }; + let range = + range + .as_range(head_meta.size) + .map_err(|source| object_store::Error::Generic { + store: "pchronicle-cache", + source: Box::new(source), + })?; // v2 deliberately never reads the previous unnamespaced cache entries. let key = serde_json::to_string(&( "v2", @@ -366,7 +370,17 @@ impl ObjectStore for CachedObjectStore { head_meta.size, self.block_size_bytes, )) - .unwrap(); + .unwrap_or_else(|_| { + format!( + "v2:{}:{}:{:?}:{:?}:{}:{}", + self.store_uri, + p.as_ref(), + version, + etag, + head_meta.size, + self.block_size_bytes + ) + }); let options = GetOptions { version: version.or(o.version), if_match: etag.or(o.if_match), diff --git a/crates/persisting-pchronicle/src/store/mod.rs b/crates/persisting-pchronicle/src/store/mod.rs index 47ccb7e5c..6e7c77d1e 100644 --- a/crates/persisting-pchronicle/src/store/mod.rs +++ b/crates/persisting-pchronicle/src/store/mod.rs @@ -51,6 +51,7 @@ pub(crate) mod opendal_store; pub mod persistent_cache; #[cfg(feature = "lance-store")] mod query_engine; +#[cfg(feature = "lance-store")] pub use persistent_cache::PersistentCache; #[cfg(feature = "lance-store")] mod root_write_lock; diff --git a/crates/persisting-pchronicle/src/store/opendal_store.rs b/crates/persisting-pchronicle/src/store/opendal_store.rs index 26c50a87a..a351521a1 100644 --- a/crates/persisting-pchronicle/src/store/opendal_store.rs +++ b/crates/persisting-pchronicle/src/store/opendal_store.rs @@ -41,7 +41,7 @@ impl StoreConfig { &self.access_key, &self.secret_key, )) - .expect("serialize S3 configuration"), + .unwrap_or_default(), ) .to_hex() .to_string() From 3a8144bd93e5e797bd1886ef30150bd63392f8a2 Mon Sep 17 00:00:00 2001 From: Reiase Date: Fri, 18 Sep 2026 16:48:35 +0800 Subject: [PATCH 22/22] Refactor catalog access control and dataset handling - Updated the `public_mounts` method to utilize `libraries_for_public` and enhance the creation of `DatasetMount` instances with backend configuration. - Refactored dataset listing and access control logic to directly use the `CatalogSnapshot`, improving clarity and reducing reliance on optional states. - Simplified credential retrieval for public datasets, streamlining the request handling process. This commit aims to improve the structure and efficiency of catalog access control and dataset management in the pChronicle application. --- .../src/server/catalog.rs | 38 +- .../src/server/mod.rs | 16 - .../rfcs/0013-pchronicle-warehouse-catalog.md | 406 +++++++++--------- .../rfcs/0013-pchronicle-warehouse-catalog.md | 127 +++--- 4 files changed, 295 insertions(+), 292 deletions(-) diff --git a/crates/persisting-pchronicle-cli/src/server/catalog.rs b/crates/persisting-pchronicle-cli/src/server/catalog.rs index fde84d438..dcf09da3a 100644 --- a/crates/persisting-pchronicle-cli/src/server/catalog.rs +++ b/crates/persisting-pchronicle-cli/src/server/catalog.rs @@ -157,9 +157,20 @@ impl CatalogAcl { } pub(crate) fn public_mounts(&self) -> Vec { - self.public_for_all() + self.libraries_for_public() .into_iter() - .filter_map(|library| DatasetMount::new(&library.name, &library.uri).ok()) + .filter_map(|library| { + DatasetMount::new(&library.name, &library.uri) + .ok() + .map(|mount| { + mount.with_backend(persisting_pchronicle::storage::StoreConfig { + endpoint: library.endpoint.clone(), + region: library.region.clone(), + access_key: library.access_key.clone(), + secret_key: library.secret_key.clone(), + }) + }) + }) .collect() } @@ -854,11 +865,9 @@ pub(super) async fn list_datasets( snapshot: Option>>, headers: axum::http::HeaderMap, ) -> Result>, ApiError> { - let acl = state - .catalog_acl - .as_ref() - .ok_or_else(|| ApiError::not_found("catalog is not enabled"))?; - let libraries = acl.visible_for_headers(&headers)?; + let axum::Extension(snapshot) = + snapshot.ok_or_else(|| ApiError::not_found("catalog is not enabled"))?; + let libraries = snapshot.acl.visible_for_headers(&headers)?; Ok(axum::Json(libraries)) } @@ -913,12 +922,7 @@ pub(super) async fn catalog_data_plane_layer( && !url::form_urlencoded::parse(request.uri().query().unwrap_or("").as_bytes()) .any(|(key, value)| (key == "dataset" || key == "prefix") && !value.is_empty()); if request.method() == axum::http::Method::GET && (ui_tables || root_tree) { - let libraries = match state - .catalog_acl - .as_ref() - .unwrap() - .visible_for_headers(request.headers()) - { + let libraries = match acl.visible_for_headers(request.headers()) { Ok(libraries) => libraries, Err(error) => return error.into_response(), }; @@ -978,11 +982,7 @@ pub(super) async fn catalog_data_plane_layer( .unwrap_or_else(|error| error.into_response()); } if !dataset.is_empty() { - if let Some((access_key, secret_key)) = state - .catalog_acl - .as_ref() - .and_then(|acl| acl.credentials_for_public(dataset)) - { + if let Some((access_key, secret_key)) = acl.credentials_for_public(dataset) { let headers = request.headers_mut(); if let (Ok(access_key), Ok(secret_key)) = (access_key.parse(), secret_key.parse()) { headers.insert(ACCESS_KEY_HEADER, access_key); @@ -1545,7 +1545,7 @@ uri = "{}" .is_err() ); let mut state = super::super::app_state(super::super::ChronicleServerConfig::front_only()); - state.catalog_acl = Some(std::sync::Arc::new(acl)); + state.catalog_acl = Some(std::sync::Arc::new(CatalogState::new(acl, None))); let warehouse = super::super::PreparedWarehouse { state }; let app = warehouse.router(); for path in [ diff --git a/crates/persisting-pchronicle-cli/src/server/mod.rs b/crates/persisting-pchronicle-cli/src/server/mod.rs index 33c8cd627..db27c1db6 100644 --- a/crates/persisting-pchronicle-cli/src/server/mod.rs +++ b/crates/persisting-pchronicle-cli/src/server/mod.rs @@ -295,22 +295,6 @@ impl PreparedWarehouse { mut config: ChronicleServerConfig, catalog_config: Option, ) -> anyhow::Result { - let browse_mounts = acl - .libraries_for_public() - .into_iter() - .filter_map(|library| { - DatasetMount::new(&library.name, &library.uri) - .ok() - .map(|m| { - m.with_backend(persisting_pchronicle::storage::StoreConfig { - endpoint: library.endpoint.clone(), - region: library.region.clone(), - access_key: library.access_key.clone(), - secret_key: library.secret_key.clone(), - }) - }) - }) - .collect::>(); config.datasets.clear(); config.default_dataset = None; let mut state = app_state(config); diff --git a/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md b/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md index 4cde65186..0d2f481cf 100644 --- a/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md +++ b/docs/src/en/rfcs/0013-pchronicle-warehouse-catalog.md @@ -3,334 +3,328 @@ | Field | Value | |---|---| | **Status** | Proposed | -| **Date** | 2026-08-30 | -| **Component** | pChronicle CLI、`pchronicle serve`、pChronicle Web | -| **Related** | [RFC-0003 Ownership](0003-pchronicle-ownership.md) · [Warehouse 指南](../pchronicle/guides/serve.md) · [CLI 参考](../pchronicle/reference/cli.md) · [架构](../pchronicle/design/architecture.md) | +| **Date** | 2026-09-18 | +| **Component** | pChronicle CLI, `pchronicle serve`, pChronicle Web | +| **Related** | [RFC-0003 Ownership](0003-pchronicle-ownership.md) · [Warehouse guide](../pchronicle/guides/serve.md) · [CLI reference](../pchronicle/reference/cli.md) · [Architecture](../pchronicle/design/architecture.md) | --- -## 摘要 +## Summary -本 RFC 定义 pChronicle 平台部署时打开 **path** 的一种方式:**Directory**(名字 → path + ACL + 换票)。 +This RFC defines one way to open a **path** in a pChronicle platform deployment: a **Directory** (name → path + ACL + ticket exchange). -Dataset 身份始终是 path(本机路径或 `s3://` / `az://` / `gs://` URI)。Directory 不是第三种 Dataset,也不替代 Snapshot。它只决定调用方可以解析到哪些 path;换票后的 `uri` 才是引擎打开的 Dataset。 +Dataset identity is always a path (local path or `s3://` / `az://` / `gs://` URI). Directory is not a third Dataset kind and does not replace Snapshot. It only decides which paths a caller may resolve; the ticket `uri` is what the engine opens. -CLI 标志、配置文件和 HTTP 路径为兼容性仍使用 `catalog` 一词(`--catalog-config`、`catalog.toml`、`catalog://`、`/api/v1/catalog/datasets`)。产品与 RFC 口径称 Directory。 +CLI flags, config files, and HTTP paths keep the word `catalog` for compatibility (`--catalog-config`, `catalog.toml`, `catalog://`, `/api/v1/catalog/datasets`). Product and RFC language use Directory. The config is TOML (extension may be `.toml` / `.yml`, etc.; content is still parsed as TOML). -规范实现挂在现有 `pchronicle serve --catalog-config` 上,不引入独立 `catalog serve` 进程。 -Listener 默认可为 loopback;也允许绑定非环回地址,但部署方 MUST 自行保证网络边界。 +The normative implementation hangs off `pchronicle serve --catalog-config`. There is no separate `catalog serve` process. +The listener defaults to loopback and MAY bind non-loopback addresses; deployers MUST own the network boundary. -- **Serve 挂载**:`pchronicle serve --catalog-config FILE` MUST 把 `catalog.toml` 中的 **全部** - `[datasets.*]` 挂进 Warehouse(与位置参数挂载等价)。本机 Web / 无用户钥的数据面请求在 - **父进程内**打开这些 mount,不再 front-only。 -- **CLI 配置**:`pchronicle serve catalog dataset add|remove|list` 改写 libraries; - `issue|grant|revoke` 改写用户与授权。 -- **Directory 换票**:`@team` 解析为 `catalog://…`;`@team/prod` 换票后客户端打开票里的 path。 - `/api/v1/catalog/datasets` 仍按用户钥过滤可见 library。 +- **Serve**: the parent only authenticates, serves the directory/tickets, and spawns workers (front-only). It MUST NOT open `[datasets.*]` in-process. Authorized data-plane requests open mounts in a one-shot `--catalog-query-worker`. +- **Public browse**: datasets granted with `user = "*"` are visible anonymously; the parent MAY cache browse projections for them without backend keys. +- **CLI config**: `pchronicle serve catalog dataset add|remove|list` rewrites datasets; `issue|grant|revoke` rewrites users and grants. Users/grants hot-reload; Dataset URI and backend credential changes REQUIRE a restart. +- **Directory tickets**: `@team` resolves as `catalog://…`; `@team/prod` opens the ticket path after exchange. `/api/v1/catalog/datasets` filters by user keys (no headers → public libraries only). ```text pchronicle serve catalog dataset add --catalog-config catalog.toml prod --uri s3://bucket/prod \ --access-key BACKEND_AK --secret-key BACKEND_SK pchronicle serve catalog issue --catalog-config catalog.toml alice -pchronicle serve catalog grant --catalog-config catalog.toml alice prod evals +pchronicle serve catalog grant --catalog-config catalog.toml alice prod pchronicle serve --catalog-config catalog.toml --listen 127.0.0.1:8081 pchronicle dataset pin team catalog://127.0.0.1:8081 --ak USER_AK --sk USER_SK pchronicle query @team/prod 'SELECT 1' ``` -## 动机 +## Motivation -本机路径和静态 Warehouse mount 假设操作者已经能看见全部 Dataset。把对象存储上的多个评测库交给一组人使用时,出现三个缺口: +Local paths and static Warehouse mounts assume the operator can already see every Dataset. Sharing several object-store evaluation libraries with a group leaves three gaps: -1. **发现与授权混在一起**。用户需要一份目录,列出自己可以打开的 library 名,而不是把所有 bucket URI 写进每人的 `config.toml`。 -2. **后端密钥不能进用户配置**。对象存储 ak/sk 属于存储账户;用户钥只用于 Directory 鉴权。把后端钥写入本机 dataset pin 会扩散到每台笔记本,也无法按人裁剪可见库。 -3. **Web 与 CLI 的数据面不同**。CLI 可以在换票后自己打开 `s3://`。Web 的查询跑在 serve 进程里;若父进程加载全部 library 的后端密钥并执行 SQL,一次鉴权绕过就会看到未授权库。 +1. **Discovery and authorization are mixed**. Users need a directory of library names they may open, not every bucket URI in each laptop `config.toml`. +2. **Backend keys must not live in user config**. Object-store ak/sk belong to the storage account; user keys only authenticate to Directory. Writing backend keys into local dataset pins spreads them and cannot trim visibility per person. +3. **Web and CLI data planes differ**. After a ticket exchange, the CLI can open `s3://` itself. Web queries run inside serve; if the parent loads every library's backend keys and runs SQL, one auth bypass sees unauthorized libraries. -本 RFC 把 Directory 定义为 **目录 + ACL + 换票**,把存储访问留给已有 `open(path)`,并把 Web 数据面隔离到一次性 worker。 +This RFC defines Directory as **directory + ACL + ticket exchange**, leaves storage access to existing `open(path)`, and isolates the Web data plane in a one-shot worker. -## 目标与非目标 +## Goals and non-goals -### 目标 +### Goals -- 用一份 `catalog.toml` 同时描述 libraries 和 users。 -- 用 CLI 签发用户钥并改写 ACL:`pchronicle serve catalog issue|grant|revoke` 不启动 HTTP。 -- 让 `@name/library` 解析为一条 path(换票后的 `uri`);引擎随后只打开该 path。 -- 换票后 CLI 自己访问存储;后端密钥只出现在票和 worker stdin 中,不写入用户 `config.toml`。 -- Web 用用户钥换授权范围,查询只看到该用户的 mounts。 -- 允许 Warehouse 绑定任意 listen 地址;默认示例仍用 loopback。Catalog 头不是公网认证边界,不可信网络上的暴露由部署方负责。 +- Describe `meta`, `users`, `datasets`, and `[[grants]]` in one Directory config. +- Issue user keys and rewrite ACL via CLI: `pchronicle serve catalog issue|grant|revoke` without starting HTTP. +- Resolve `@name/library` to one path (the ticket `uri`); the engine then opens only that path. +- After exchange, the CLI talks to storage itself; backend keys appear only in tickets and worker stdin, never in the user's `config.toml`. +- Web exchanges user keys for an authorized mount set; datasets with `user = "*"` MAY be listed/browsed anonymously. +- Hot-reload users and grants about every 3 seconds; reject hot-reload of Dataset definitions and backend credentials (restart required). +- Allow Warehouse to bind any listen address; examples stay on loopback. Catalog headers are not a public auth boundary; exposure on untrusted networks is the deployer's responsibility. -### 非目标 +### Non-goals -- STS、临时凭证轮换、或把用户钥映射成短时 AWS session。 -- 热加载 `catalog.toml`;改配置 MUST 重启 serve。 -- 在运行中的 Warehouse 上提供 HTTP 签发接口。 -- 提供独立 `catalog serve` 二进制。 -- 在已运行的 Tokio runtime 上 `fork(2)`(未定义行为)。 -- 把后端对象存储密钥写入本机 dataset pin 配置。 -- 改变 Snapshot 协议、SQL schema 或 Gateway/Control 协议。 +- STS, short-lived credential rotation, or mapping user keys to AWS sessions. +- Hot-reloading Dataset URI / endpoint / region / backend ak/sk (restart serve). +- HTTP mint APIs on a running Warehouse. +- A separate `catalog serve` binary. +- `fork(2)` of a running Tokio runtime (undefined behavior). +- Writing backend object-store keys into local dataset pin config. +- Enforcing fine-grained `permissions` in v1 (the field is writable; semantics remain library membership). +- Changing Snapshot protocol, SQL schema, or Gateway/Control protocols. -本 RFC 的 Directory 与打开 path 之后的 **Snapshot**(见 [Snapshot 设计](../pchronicle/design/catalog.md))不是同一对象。Directory 列出授权 path;Snapshot 钉住一条已打开 path 上的 Source 成员与版本。 +Directory in this RFC is not the same object as **Snapshot** after a path is opened (see [Snapshot design](../pchronicle/design/catalog.md)). Directory lists authorized paths; Snapshot pins Source membership and versions on an opened path. -## 角色与信任边界 +## Roles and trust boundary -| 角色 | 持有 | 用途 | +| Role | Holds | Use | |---|---|---| -| 存储账户 | 后端 `access_key` / `secret_key`,以及可选 endpoint、region | 打开 `s3://` library | -| Directory 用户 | 用户 `access_key` / `secret_key` | 列出/领取被授权 library 的票 | -| 本机 CLI | 用户钥(存在 dataset pin 配置) | 换票后把后端钥注入进程环境并打开票中的 path | -| 浏览器 | 用户钥(`localStorage`) | 作为请求头发给 loopback serve | -| serve 父进程 | 完整 `catalog.toml` | 鉴权、返回票、spawn worker;不把后端钥写入 AWS 环境 | -| query worker | 该用户被授权 library 的票 | 一次性执行 Warehouse 数据面请求 | +| Storage account | Backend `access_key` / `secret_key`, optional endpoint, region | Open `s3://` libraries | +| Directory user | User `access_key` / `secret_key` | List/fetch tickets for granted libraries | +| Local CLI | User keys (in dataset pin config) | After exchange, inject backend keys into process env and open the ticket path | +| Browser | User keys (`localStorage`) | Send as request headers to loopback serve | +| serve parent | Full Directory config | Authenticate, return tickets, spawn worker; do not write backend keys into AWS env | +| query worker | Tickets for that user's libraries | One-shot Warehouse data-plane request | -ACL 是 **发现与授权** 边界,不是对象存储的强制隔离。持有后端密钥或能猜测 URI 的调用方,仍可能绕过 Directory 直接访问存储。Directory 不替代 bucket policy。 +ACL is a **discovery and authorization** boundary, not mandatory object-store isolation. Callers who hold backend keys or can guess URIs may still bypass Directory. Directory does not replace bucket policy. -## 进程模型 +## Process model -Directory 挂在现有 Warehouse listener 上。未传 `--catalog-config` 时,`pchronicle serve` 行为不变:静态 mount、无用户鉴权。 +Directory hangs on the existing Warehouse listener. Without `--catalog-config`, `pchronicle serve` is unchanged: static mounts, no user auth. ```text -浏览器 / CLI +browser / CLI → Warehouse listener ├─ GET /health - ├─ GET /api/v1/catalog/datasets[/{name}] 父进程:鉴权 + 目录/票 - ├─ 静态 UI - └─ 其余 /api/* 父进程内挂载 / 或 spawn worker + ├─ GET /api/v1/catalog/datasets[/{name}] parent: auth + directory/ticket + ├─ static UI + └─ other /api/* parent mounts / or spawn worker → pchronicle serve --catalog-query-worker - stdin: mounts + HTTP 请求 + stdin: mounts + HTTP request stdout: status / content-type / body - 退出 + exit ``` -约束: +Constraints: -1. Listener MAY 绑定非 loopback 地址。本 RFC 不把 catalog 头当作公网认证边界;部署方 MUST 在不可信网络上自行加边界。 -2. 父进程 MUST NOT 打开 `catalog.toml` 中的 libraries。父进程使用空 mount 的 front-only Warehouse。 -3. Worker MUST 由 `Command` 启动新进程,MUST NOT `fork(2)` 已运行的 Tokio runtime。 -4. Worker MUST NOT 监听端口、MUST NOT 读取 `catalog.toml`、MUST NOT 读取用户钥。它只消费 stdin 中过滤后的 mounts 和原始请求。 -5. Worker 继承父进程环境(证书、`PATH` 等),但父进程 MUST NOT 预先把 catalog 后端密钥写入 `AWS_*`。Worker 在打开存储前为自己设置该用户票中的后端环境。 -6. 同一 `catalog.toml` 内所有 `s3://` library MUST 共用同一组 endpoint、region 和后端 ak/sk。进程级 AWS 环境一次只能持有一套凭据。 -7. 隐藏 flag `--catalog-query-worker` MUST NOT 出现在用户可见的 `serve --help` 中。 +1. The listener MAY bind non-loopback. This RFC does not treat catalog headers as a public auth boundary; deployers MUST add a boundary on untrusted networks. +2. The parent MUST NOT open datasets from the config. It uses a front-only Warehouse with empty mounts; public browse caches hold paths only and MUST NOT write backend keys into parent `AWS_*`. +3. Workers MUST be started with `Command`, MUST NOT `fork(2)` a running Tokio runtime. +4. Workers MUST NOT listen, MUST NOT read the Directory config, MUST NOT read user keys. They only consume filtered mounts and the raw request from stdin. +5. Workers inherit parent env (certs, `PATH`, …), but the parent MUST NOT pre-write catalog backend keys into `AWS_*`. The worker sets backend env from the user's tickets before opening storage. +6. Each `[datasets.*]` MAY carry its own endpoint, region, and backend ak/sk. One worker process can hold only one process-global AWS env; if a user is granted incompatible `s3://` backends, the request MUST include `dataset=` to select one, or MUST fail. +7. The hidden flag `--catalog-query-worker` MUST NOT appear in user-facing `serve --help`. -Worker 超时后父进程 MUST 返回 `unavailable`,不得把 stdin 中的密钥写进日志。 +On worker timeout the parent MUST return `unavailable` and MUST NOT log keys from stdin. -## 配置 +## Configuration -`catalog.toml` 是唯一配置面: +The Directory config only manages users, datasets, and grants. It is the single source of truth; runtime serve options still come from `pchronicle serve` flags. When the file is missing, catalog management commands create an empty catalog (with `[meta]`). + +Authoritative schema (matches current implementation / deployment samples): ```toml -[libraries.prod] -uri = "s3://bucket/prod" -endpoint = "http://127.0.0.1:9000" -region = "us-west-2" -access_key = "BACKEND_AK" -secret_key = "BACKEND_SK" - -[libraries.evals] -uri = "s3://bucket/evals" -endpoint = "http://127.0.0.1:9000" -region = "us-west-2" -access_key = "BACKEND_AK" -secret_key = "BACKEND_SK" +[meta] +version = 1 +revision = 1 +name = "default" [users.alice] -access_key = "USER_AK" -secret_key = "USER_SK" -datasets = ["prod", "evals"] - -[users.bob] -access_key = "BOB_AK" -secret_key = "BOB_SK" -datasets = ["evals"] +access_key = "pcak_…" +secret_key = "…" + +[datasets.default] +uri = "/data/warehouse" + +[datasets.prod] +uri = "s3://prod/" +endpoint = "http://s3-a.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_A" +secret_key = "BACKEND_SK_A" + +[datasets.prod2] +uri = "s3://prod" +endpoint = "http://s3-b.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_B" +secret_key = "BACKEND_SK_B" + +[[grants]] +user = "*" +dataset = "prod" + +[[grants]] +user = "*" +dataset = "prod2" + +[[grants]] +user = "alice" +dataset = "default" +# permissions optional; v1 ignores fine-grained semantics (membership only) +# permissions = ["read", "query", "analyze"] ``` -规则: - -- 启动 `pchronicle serve --catalog-config` MUST 至少有一个 library 和一个 user。 -- `pchronicle serve catalog issue` MAY 在只有 `[libraries.*]`、尚无 `[users]` 的文件上签发第一个用户。 -- library 名与用户段名 MUST 是合法 Dataset mount 名(小写 `[A-Za-z_][A-Za-z0-9_]*`)。 -- `s3://` library MUST 同时设置后端 `access_key` 和 `secret_key`。 -- 非 `s3://` library MUST NOT 设置后端密钥。 -- 所有 `s3://` library 的 endpoint、region、后端密钥 MUST 完全一致。 -- `users.*.datasets` 引用的名字 MUST 存在于 `libraries`。 -- 用户 `access_key` MUST 全局唯一。 -- 配置文件 MUST 是普通文件,大小有上界;解析失败则 serve 拒绝启动。 +Rules: -本地路径 library 允许不设后端密钥,便于同机目录通过 catalog 做授权发现。客户端换票后仍按票中的 URI 打开。 +- `meta.version` MUST be a supported config version; successful CLI writes SHOULD maintain `meta.revision` / `meta.name` (optional). +- User and Dataset names MUST be lowercase `[A-Za-z_][A-Za-z0-9_]*`. +- `[users.*]` contains only `access_key` / `secret_key`; `access_key` MUST be globally unique; v1 allows plaintext `secret_key`. +- `[datasets.*]` MUST include `uri`; local paths MUST NOT set backend keys; `s3://` MUST set both `access_key` and `secret_key`, and MAY set `endpoint` / `region`. +- Different datasets MAY use different endpoint / region / backend keys (see process model item 6). +- `[[grants]]` MUST include `user` and `dataset`; `permissions` is optional and not enforced in v1. +- `grants.user = "*"` marks the dataset public (anonymous list/browse) and expands to **all current** users at parse time; newly issued users inherit it after hot-reload. +- Named `grants.user` MUST reference an existing user; `grants.dataset` MUST reference an existing dataset. +- Duplicate grants for the same user and dataset MUST be rejected (including after `*` expansion). +- Config size MUST be bounded; parse/validation failure refuses serve start; hot-reload failure MUST keep the last valid ACL. +- TOML is authoritative; future SQLite/Postgres may only be indexes or derived projections. -## CLI 签发与授权 +## CLI management -签发和改授权是 **写 `catalog.toml` 的 CLI**,不是运行中 Warehouse 的 HTTP API。出现 `catalog` 子命令时 MUST NOT 启动 listener。运行中的 serve 每 3 秒检查配置,用户和授权无需重启即可生效。 +Catalog management commands only edit the config file and do not start an HTTP listener. Missing files create the parent directory and an empty config. ```text -pchronicle serve catalog dataset add --catalog-config FILE NAME --uri URI [--endpoint URL] [--region REGION] [--access-key KEY] [--secret-key KEY] +pchronicle serve catalog dataset add --catalog-config FILE NAME --uri URI [OPTIONS] pchronicle serve catalog dataset remove --catalog-config FILE NAME... pchronicle serve catalog dataset list --catalog-config FILE + pchronicle serve catalog issue --catalog-config FILE NAME pchronicle serve catalog grant --catalog-config FILE NAME DATASET... pchronicle serve catalog revoke --catalog-config FILE NAME DATASET... -pchronicle serve --catalog-config FILE --listen 127.0.0.1:8081 ``` -`catalog` 是 `serve` 的保留子命令。要挂载名为 `catalog` 的路径,使用 `./catalog` 或 `NAME=./catalog`。 -`--catalog-config` MUST NOT 与位置参数 Dataset 同时使用。 - -### `dataset add` / `remove` / `list` - -- MUST NOT 启动 Warehouse。只改 `FILE` 后退出。 -- `add` 写入 `[datasets.NAME]`。已存在的名字 MUST 拒绝。`s3://` MUST 设置后端钥,且 MUST 与文件中已有 s3 library 的 endpoint/region/ak/sk 完全一致;非 `s3://` MUST NOT 设置后端钥。 -- `remove` 删除列出的 library。若仍有 grant 引用该 library,MUST 失败且 MUST NOT 改文件。 -- `list` 打印 `name` / `uri`(及可选 endpoint/region),MUST NOT 打印后端密钥。 - -### `issue` - -- MUST NOT 启动 Warehouse。只改 `FILE` 后退出。 -- 已存在的用户名 MUST 拒绝,MUST NOT 覆盖或轮换密钥。本 RFC 不引入 `issue --rotate`。 -- 生成的用户钥: - - `access_key`:`pcak_` 前缀 + 24 字节小写 hex(48 个 hex 字符) - - `secret_key`:32 字节小写 hex(无前缀) -- 写入 `[users.NAME]`:`access_key`、`secret_key`、`datasets = []`。签发 MUST NOT 授予任何 library。 -- stdout 打印该用户的 `name` / `access_key` / `secret_key`(表或 JSON)。secret MUST 只在这次 stdout 出现;stderr 只报 `config= updated=true`,MUST NOT 打印 sk。`dataset list` 等其它命令 MUST NOT 回显 catalog 用户 sk。 -- `access_key` 碰撞时 MUST 重试生成,MUST NOT 写入半截配置。 - -### `grant` / `revoke` - -- `grant` 是累加:已授权的 library 保持不变,新名字追加。未知用户或未知 library MUST 失败,且 MUST NOT 改文件。 -- `revoke` 从该用户的 `datasets` 里去掉列出的名字。未知用户、或该用户当前并未持有的 library 名 MUST 失败。 -- 两个命令的 stdout 只报 `name` 与更新后的 `datasets`,MUST NOT 打印密钥。 - -改写配置可以整表重写,不要求保留注释。新用户通常在 3 秒内生效。 +`dataset add` only registers a Dataset; it does not create or delete backend data. When appending `s3://` via CLI, if other `s3://` entries already exist, the new endpoint / region / backend keys MUST match them exactly (hand-written multi-backend configs remain valid, but workers must select per item 6). `issue` generates user AK/SK (secret printed once on stdout) and MUST NOT write any grant. `grant` / `revoke` edit `[[grants]]`; NAME `*` writes/removes **named** grants for every **current** user (it does not write a `user = "*"` public row). All writes MUST replace the file atomically and keep the previous file on failure. ## HTTP -Directory 路由与 Warehouse 共用 `/api` 与 `/api/v1` 前缀。鉴权头: +Directory routes share Warehouse `/api` and `/api/v1` prefixes. Auth headers: -| Header | 含义 | +| Header | Meaning | |---|---| -| `x-pchronicle-access-key` | 用户 access key | -| `x-pchronicle-secret-key` | 用户 secret key | +| `x-pchronicle-access-key` | User access key | +| `x-pchronicle-secret-key` | User secret key | -缺失、空白或密钥不匹配 MUST 返回 `401`,且 MUST NOT 区分“用户不存在”与“密钥错误”。 +Missing, blank, or mismatched keys MUST return `401`, and MUST NOT distinguish “unknown user” from “bad secret”. +With no catalog headers at all, list MAY return only datasets granted to `user = "*"`; partial headers still MUST `401`. -未授权的 library 名与不存在的 library 名 MUST 都返回 `404`。 +Unauthorized and unknown library names MUST both return `404`. -| 路由 | 父进程 | 响应 | +| Route | Parent | Response | |---|---|---| -| `GET /api/v1/catalog/datasets` | 是 | 该用户可见 library 的 `name`、`uri`、可选 `endpoint`/`region`;**不含**后端密钥 | -| `GET /api/v1/catalog/datasets/{name}` | 是 | 授权时返回完整票,含后端 `access_key` / `secret_key` | -| `GET /api/health` | 是 | 无鉴权 | -| 静态 UI | 是 | 无鉴权 | -| 其余 `/api/*`(含 `GET /api/catalog`,返回当前 Snapshot) | 否,转发 worker | 先鉴权,再按用户 mounts 执行 | +| `GET /api/v1/catalog/datasets` | yes | Libraries visible to the auth user, or public libraries when anonymous: `name`, `uri`, optional `endpoint`/`region`; **no** backend keys | +| `GET /api/v1/catalog/datasets/{name}` | yes | Full ticket with backend `access_key` / `secret_key` when authenticated and authorized; anonymous MUST `401` (public libraries allow key-free list/browse only) | +| `GET /api/health` | yes | No auth | +| Static UI | yes | No auth | +| Other `/api/*` (including `GET /api/catalog` Snapshot) | no, forward to worker | Authenticate, then run with the user's mounts; multi-backend requires `dataset=` | -错误 JSON 沿用 Warehouse 的 `code`、`message`、`request_id`。日志可以包含用户段名、library 名和 `request_id`,MUST NOT 打印用户钥或后端钥。 +Error JSON keeps Warehouse `code`, `message`, `request_id`. Logs MAY include user segment, library name, and `request_id`; MUST NOT print user or backend keys. -`GET /api/v1/catalog/datasets/{name}` 是 CLI 换票接口。拿到票的客户端随后直接打开 `uri`(Dataset path),不再把查询代理回 Directory。 +`GET /api/v1/catalog/datasets/{name}` is the CLI ticket exchange. Clients then open `uri` (the Dataset path) directly and do not proxy queries back through Directory. ## CLI dataset pin -`catalog://` 是 pin **类型**,不是 DatasetLocation 可解析的存储 URI。换票成功后 Dataset 身份是票里的 path,不是 `catalog://…` 本身。 +`catalog://` is a pin **type**, not a storage URI `DatasetLocation` can open. After a successful exchange, Dataset identity is the ticket path, not `catalog://…` itself. ```bash pchronicle dataset pin team catalog://127.0.0.1:8081 --ak USER_AK --sk USER_SK ``` -规范化规则: +Normalization: -- scheme MUST 为 `catalog`; -- host MUST 是环回 IP(如 `127.0.0.1`),MUST 带端口; -- MUST NOT 包含 userinfo、path、query 或 fragment; -- MUST NOT 接受 `--endpoint` / `--region`(那是对象存储参数,来自票而不是 pin)。 +- scheme MUST be `catalog`; +- host MUST be a loopback IP (e.g. `127.0.0.1`) with a port; +- MUST NOT include userinfo, path, query, or fragment; +- MUST NOT accept `--endpoint` / `--region` (those come from the ticket, not the pin). -解析按 pin **类型** 分派,而不是把所有 `@name/suffix` 都做路径拼接: +Resolution dispatches on pin **type**, not path join for every `@name/suffix`: -| 引用 | catalog pin | 普通 URI pin | +| Reference | catalog pin | ordinary URI pin | |---|---|---| -| `@team` / `@team/` | `ls` 列出该用户可访问的 Datasets | 解析为 pin 根 URI | -| `@team/prod` | 向 Directory 领取 library `prod` 的票,打开票中 path | 根 URI 再拼接路径 `prod` | -| `@team/prod/more` | 先领 `prod`,再把 `more` 拼到票的 path 上 | 根 URI 拼接 `prod/more` | +| `@team` / `@team/` | `ls` libraries the user may access | resolve to pin root URI | +| `@team/prod` | fetch ticket for library `prod`, open ticket path | join `prod` onto root URI | +| `@team/prod/more` | fetch `prod`, then join `more` onto ticket path | join `prod/more` onto root URI | -用户 `--ak/--sk` 存入本机 dataset pin 凭据表,与 S3 pin 相同的隔离方式:不出现在 `dataset list` / `dataset show` 的 URI 里。后端密钥 MUST NOT 写入该文件。 +User `--ak/--sk` live in the local dataset pin credential table, isolated like S3 pins: not shown in `dataset list` / `dataset show` URIs. Backend keys MUST NOT be written there. -换到的票缓存在 CLI 进程内(`thread_local`),按 catalog URL、用户 access key 和 library 名索引。长生命周期的 `serve` 进程不使用这份 CLI 缓存;Web 每次请求重新鉴权。进程退出即丢弃缓存。 +Tickets cache in the CLI process (`thread_local`), keyed by catalog URL, user access key, and library name. Long-lived `serve` does not use this CLI cache; Web re-authenticates each request. Cache is dropped on process exit. ## Web -Settings(左侧 **Keys**)保存 catalog 用户钥到 `localStorage`: +Settings (left **Keys**) store catalog user keys in `localStorage`: - `pchronicle.catalog.access_key` - `pchronicle.catalog.secret_key` -浏览器把这两项作为上述 HTTP 头附加到 **发往当前 pChronicle serve 的** `/api/` 请求。这与 Assistant 的 Browser BYOK 相反:Assistant 钥只发给模型端点,catalog 钥必须到达 serve 才能鉴权。 +The browser attaches these as the HTTP headers above on `/api/` requests **to the current pChronicle serve**. This is the opposite of Assistant Browser BYOK: Assistant keys go only to model endpoints; catalog keys must reach serve for auth. -未配置用户钥时,Web MUST NOT 假装本地 Warehouse 已授权;catalog 模式下无头请求在数据面得到 `401`。无 `--catalog-config` 的普通 serve 不要求这些头。 +Without user keys, Web MAY still browse `user = "*"` public libraries; authenticated data-plane requests MUST `401`. Ordinary serve without `--catalog-config` does not require these headers. -查询在 worker 中执行。浏览器不直接持有后端对象存储密钥。 +Queries run in the worker. The browser does not hold backend object-store keys. -## 数据面隔离 +## Data-plane isolation -父进程在数据面中间件中: +In the data-plane middleware the parent: -1. 校验用户钥; -2. 过滤该用户的 library 票; -3. 把 HTTP method、path、query、body 和 mounts 写成 JSON job; -4. spawn 同源二进制 `serve --catalog-query-worker`; -5. 把 stdout 信封还原为 HTTP 响应。 +1. Validates user keys; +2. Filters that user's library tickets; +3. Writes HTTP method, path, query, body, and mounts as a JSON job; +4. Spawns the same binary as `serve --catalog-query-worker`; +5. Reconstructs the HTTP response from the stdout envelope. -Worker 用票构造 `ChronicleServerConfig` mounts,执行与普通 Warehouse 相同的只读路由,然后退出。 +The worker builds `ChronicleServerConfig` mounts from tickets, runs the same read-only Warehouse routes, then exits. -不得把未授权 library 的票放进 job。空授权集合 MUST 表现为 `404`,而不是启动一个空 Warehouse。 +Unauthorized library tickets MUST NOT enter the job. An empty grant set MUST surface as `404`, not an empty Warehouse. -## 被拒绝的方案 +## Rejected alternatives -### 把签发做成 Warehouse HTTP mint +### Warehouse HTTP mint -拒绝。Catalog 头不是公网认证边界;loopback 上无认证的 mint 会把用户钥发给任何能打到端口的本机进程。签发入口是改写 `catalog.toml` 的 CLI。 +Rejected. Catalog headers are not a public auth boundary; unauthenticated mint on loopback would hand user keys to any local process that can hit the port. Issuance is CLI rewriting of the Directory config. -### 独立 `catalog serve` 进程 +### Separate `catalog serve` process -拒绝。第二套 listener、端口和生命周期会与 Warehouse 文档分叉。Catalog 目录流量很小,适合挂在现有 `pchronicle serve` 上。 +Rejected. A second listener, port, and lifecycle would fork Warehouse docs. Directory traffic is small and belongs on existing `pchronicle serve`. -### 父进程打开全部 libraries 再按用户过滤 SQL +### Parent opens every dataset then filters SQL per user -拒绝。DataFusion 与对象存储客户端一旦持有全量后端密钥和 mount,过滤错误就会越权。Web 查询必须在只含授权 mounts 的进程里执行。 +Rejected. Once DataFusion and object-store clients hold all backend keys and mounts, a filter bug is a privilege escalation. Web queries MUST run in a process that only has authorized mounts. -### `fork(2)` 已运行的 Tokio 以“降权” +### `fork(2)` a running Tokio “to drop privilege” -拒绝。在多线程 runtime 上 fork 是未定义行为。使用 `Command` 新进程。 +Rejected. Fork on a multi-threaded runtime is undefined behavior. Use a new `Command` process. -### STS / 短时会话券 +### STS / short-lived session tickets -拒绝。当前目标是本机协作目录,不是云上身份联邦。透传后端密钥给已授权客户端,配置更简单,也与现有 S3 pin 注入 `AWS_*` 的方式一致。 +Rejected. The target is a local collaboration directory, not cloud identity federation. Passing backend keys to authorized clients is simpler and matches existing S3 pin injection into `AWS_*`. -### 把 catalog 做成普通路径拼接 pin +### Treat catalog as ordinary path-join pins -拒绝。`@prod/evals` 对 `s3://bucket` 是路径拼接;对 Directory locator 则是“名字 + library 名”,换票后打开票中 path。混用会让 `@team/prod` 被拼成非法 URI `catalog://127.0.0.1:8081/prod`。 +Rejected. `@prod/evals` on `s3://bucket` is path join; on a Directory locator it is “name + library name”, then open the ticket path. Mixing would produce the illegal URI `catalog://127.0.0.1:8081/prod`. -### 非环回 bind + 把 catalog 头当公网认证 +### Non-loopback bind + catalog headers as public auth -拒绝。Warehouse 仍是本机检查面。打开 `0.0.0.0` 需要独立的认证、TLS 与多租户威胁模型,超出本 RFC。 +Rejected. Warehouse remains a local inspection surface. Binding `0.0.0.0` needs separate auth, TLS, and multi-tenant threat modeling beyond this RFC. -## 兼容性与演进 +## Compatibility and evolution -- 无 `--catalog-config` 时,现有 Dataset 引用、普通 pin 的 `@name/suffix` 路径拼接、以及无鉴权 loopback Warehouse MUST 保持不变。 -- `catalog://` MUST NOT 成为 `DatasetLocation` 可打开的存储 scheme;只有 dataset pin 解析器认识它。 -- 新增 library 字段、鉴权头或 worker 协议属于破坏性变更,需要修订本 RFC。 -- 未来的 STS 或热加载可以作为后续 RFC,不得 silently 改变“透传后端密钥 / 重启生效”的语义。 +- Without `--catalog-config`, existing Dataset references, ordinary pin `@name/suffix` path joins, and unauthenticated loopback Warehouse MUST stay unchanged. +- `catalog://` MUST NOT become an openable `DatasetLocation` storage scheme; only the dataset pin resolver understands it. +- Authoritative config keys are `meta` / `users` / `datasets` / `grants`; legacy `[libraries.*]` or grants embedded in `users.*.datasets` MUST NOT remain normative. +- New Dataset fields, auth headers, or worker protocols are breaking and require revising this RFC. +- Future STS or Dataset hot-reload may be follow-on RFCs and MUST NOT silently change “pass through backend keys / Dataset changes require restart”. -本 RFC 修正架构文档中“loopback Warehouse 完全没有 authentication”的表述:在 `--catalog-config` 下,数据面和 Directory 路由使用用户钥请求头;它仍不是公网多租户服务。 +This RFC corrects architecture language that said loopback Warehouse had no authentication at all: with `--catalog-config`, the data plane and Directory routes use user-key headers, except for public libraries. It is still not a public multi-tenant service. -## 实施状态 +## Implementation status -当前实现覆盖本 RFC 的核心范围: +Current implementation covers the core of this RFC: -- `catalog.toml` 解析与启动期校验; -- `pchronicle serve catalog issue|grant|revoke` 改写 ACL(签发不授权,sk 只打一次 stdout); -- `GET /api/v1/catalog/datasets` 与 `/{name}`; -- `--catalog-config` front-only 父进程与 `--catalog-query-worker`; -- `catalog://` pin、`@team/prod` 换票与进程内票缓存; -- Web `localStorage` 用户钥与数据面请求头。 +- TOML Directory parse and startup validation (`meta` / `users` / `datasets` / `[[grants]]`, including `user = "*"`); +- `pchronicle serve catalog issue|grant|revoke|dataset …` config editors (issue grants nothing; sk printed once on stdout); +- ~3s hot-reload of users/grants; Dataset / backend credential changes rejected with the previous ACL kept; +- `GET /api/v1/catalog/datasets` and `/{name}` (including anonymous public list); +- `--catalog-config` front-only parent and `--catalog-query-worker`; multi-backend narrowed by `dataset=`; +- `catalog://` pins, `@team/prod` ticket exchange, in-process ticket cache; +- Web `localStorage` user keys and data-plane headers. -后续工作: +Follow-ups: -1. 覆盖真实 worker 子进程的集成测试(环境中不得出现未授权 library 的密钥); -2. 评估是否为本地路径 library 提供与 S3 相同的显式审计日志字段; -3. `issue --rotate`:轮换已有用户密钥(当前重名签发直接拒绝)。 +1. Integration tests that cover real worker subprocesses (unauthorized library keys must not appear in the environment); +2. Evaluate explicit audit fields for local-path libraries comparable to S3; +3. `issue --rotate` for existing user keys (duplicate-name issue currently rejects); +4. Whether CLI `dataset add` should formally write multiple S3 backends (hand-written multi-backend is valid today; CLI append still requires matching existing s3 backend identity). diff --git a/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md b/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md index f7807c4ad..1d7ec1c25 100644 --- a/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md +++ b/docs/src/zh/rfcs/0013-pchronicle-warehouse-catalog.md @@ -3,7 +3,7 @@ | Field | Value | |---|---| | **Status** | Proposed | -| **Date** | 2026-08-30 | +| **Date** | 2026-09-18 | | **Component** | pChronicle CLI、`pchronicle serve`、pChronicle Web | | **Related** | [RFC-0003 Ownership](0003-pchronicle-ownership.md) · [Warehouse 指南](../pchronicle/guides/serve.md) · [CLI 参考](../pchronicle/reference/cli.md) · [架构](../pchronicle/design/architecture.md) | @@ -15,24 +15,25 @@ Dataset 身份始终是 path(本机路径或 `s3://` / `az://` / `gs://` URI)。Directory 不是第三种 Dataset,也不替代 Snapshot。它只决定调用方可以解析到哪些 path;换票后的 `uri` 才是引擎打开的 Dataset。 -CLI 标志、配置文件和 HTTP 路径为兼容性仍使用 `catalog` 一词(`--catalog-config`、`catalog.toml`、`catalog://`、`/api/v1/catalog/datasets`)。产品与 RFC 口径称 Directory。 +CLI 标志、配置文件和 HTTP 路径为兼容性仍使用 `catalog` 一词(`--catalog-config`、`catalog.toml`、`catalog://`、`/api/v1/catalog/datasets`)。产品与 RFC 口径称 Directory。配置文件是 TOML(扩展名可为 `.toml` / `.yml` 等,内容仍按 TOML 解析)。 规范实现挂在现有 `pchronicle serve --catalog-config` 上,不引入独立 `catalog serve` 进程。 Listener 默认可为 loopback;也允许绑定非环回地址,但部署方 MUST 自行保证网络边界。 -- **Serve 挂载**:`pchronicle serve --catalog-config FILE` MUST 把 `catalog.toml` 中的 **全部** - `[datasets.*]` 挂进 Warehouse(与位置参数挂载等价)。本机 Web / 无用户钥的数据面请求在 - **父进程内**打开这些 mount,不再 front-only。 -- **CLI 配置**:`pchronicle serve catalog dataset add|remove|list` 改写 libraries; - `issue|grant|revoke` 改写用户与授权。 +- **Serve**:父进程只做鉴权、目录/换票与 spawn worker(front-only),MUST NOT 在父进程打开 + `[datasets.*]`。授权范围内的数据面请求由一次性 `--catalog-query-worker` 打开对应 mount。 +- **公开浏览**:`[[grants]]` 中 `user = "*"` 的 Dataset 对匿名调用可见;父进程可用其做浏览缓存, + 不含后端密钥。 +- **CLI 配置**:`pchronicle serve catalog dataset add|remove|list` 改写 datasets; + `issue|grant|revoke` 改写用户与授权。用户/授权热加载;Dataset URI 与后端凭证变更 MUST 重启。 - **Directory 换票**:`@team` 解析为 `catalog://…`;`@team/prod` 换票后客户端打开票里的 path。 - `/api/v1/catalog/datasets` 仍按用户钥过滤可见 library。 + `/api/v1/catalog/datasets` 按用户钥过滤可见 library(无头时仅返回公开库)。 ```text pchronicle serve catalog dataset add --catalog-config catalog.toml prod --uri s3://bucket/prod \ --access-key BACKEND_AK --secret-key BACKEND_SK pchronicle serve catalog issue --catalog-config catalog.toml alice -pchronicle serve catalog grant --catalog-config catalog.toml alice prod evals +pchronicle serve catalog grant --catalog-config catalog.toml alice prod pchronicle serve --catalog-config catalog.toml --listen 127.0.0.1:8081 pchronicle dataset pin team catalog://127.0.0.1:8081 --ak USER_AK --sk USER_SK pchronicle query @team/prod 'SELECT 1' @@ -52,21 +53,23 @@ pchronicle query @team/prod 'SELECT 1' ### 目标 -- 用一份 `catalog.toml` 描述 users、datasets 和 grants。 +- 用一份 Directory 配置描述 `meta`、`users`、`datasets` 和 `[[grants]]`。 - 用 CLI 签发用户钥并改写 ACL:`pchronicle serve catalog issue|grant|revoke` 不启动 HTTP。 - 让 `@name/library` 解析为一条 path(换票后的 `uri`);引擎随后只打开该 path。 - 换票后 CLI 自己访问存储;后端密钥只出现在票和 worker stdin 中,不写入用户 `config.toml`。 -- Web 用用户钥换授权范围,查询只看到该用户的 mounts。 +- Web 用用户钥换授权范围,查询只看到该用户的 mounts;`user = "*"` 的公开库可匿名列出/浏览。 +- 运行中每约 3 秒热加载用户与授权;Dataset 定义与后端凭证变更拒绝热加载,须重启。 - 允许 Warehouse 绑定任意 listen 地址;默认示例仍用 loopback。Catalog 头不是公网认证边界,不可信网络上的暴露由部署方负责。 ### 非目标 - STS、临时凭证轮换、或把用户钥映射成短时 AWS session。 -- 热加载 Dataset 定义和 S3 后端凭证;这类变更仍需重启 serve。 +- 热加载 Dataset URI / endpoint / region / 后端 ak/sk(须重启 serve)。 - 在运行中的 Warehouse 上提供 HTTP 签发接口。 - 提供独立 `catalog serve` 二进制。 - 在已运行的 Tokio runtime 上 `fork(2)`(未定义行为)。 - 把后端对象存储密钥写入本机 dataset pin 配置。 +- v1 细粒度 `permissions` 强制执行(字段可写,语义仍是库成员关系)。 - 改变 Snapshot 协议、SQL schema 或 Gateway/Control 协议。 本 RFC 的 Directory 与打开 path 之后的 **Snapshot**(见 [Snapshot 设计](../pchronicle/design/catalog.md))不是同一对象。Directory 列出授权 path;Snapshot 钉住一条已打开 path 上的 Source 成员与版本。 @@ -104,57 +107,75 @@ Directory 挂在现有 Warehouse listener 上。未传 `--catalog-config` 时, 约束: 1. Listener MAY 绑定非 loopback 地址。本 RFC 不把 catalog 头当作公网认证边界;部署方 MUST 在不可信网络上自行加边界。 -2. 父进程 MUST NOT 打开 `catalog.toml` 中的 datasets。父进程使用空 mount 的 front-only Warehouse。 +2. 父进程 MUST NOT 打开配置中的 datasets。父进程使用空 mount 的 front-only Warehouse;公开库的浏览缓存只持有 path,不把后端钥写入父进程 `AWS_*`。 3. Worker MUST 由 `Command` 启动新进程,MUST NOT `fork(2)` 已运行的 Tokio runtime。 -4. Worker MUST NOT 监听端口、MUST NOT 读取 `catalog.toml`、MUST NOT 读取用户钥。它只消费 stdin 中过滤后的 mounts 和原始请求。 +4. Worker MUST NOT 监听端口、MUST NOT 读取 Directory 配置、MUST NOT 读取用户钥。它只消费 stdin 中过滤后的 mounts 和原始请求。 5. Worker 继承父进程环境(证书、`PATH` 等),但父进程 MUST NOT 预先把 catalog 后端密钥写入 `AWS_*`。Worker 在打开存储前为自己设置该用户票中的后端环境。 -6. 每个 Dataset 可以使用自己的 endpoint、region 和后端 ak/sk;worker 必须按 Dataset ticket 设置对应存储环境。 +6. 每个 `[datasets.*]` MAY 自带 endpoint、region 和后端 ak/sk。同一 worker 进程的进程级 AWS 环境一次只能持有一套凭据;若用户同时授权了不兼容的多个 `s3://` 后端,请求 MUST 带 `dataset=` 显式选中其中一个,否则 MUST 失败。 7. 隐藏 flag `--catalog-query-worker` MUST NOT 出现在用户可见的 `serve --help` 中。 Worker 超时后父进程 MUST 返回 `unavailable`,不得把 stdin 中的密钥写进日志。 ## 配置 -`catalog.toml` 只管理用户、Dataset 和授权关系。它是唯一事实来源;运行时服务配置仍由 `pchronicle serve` 参数提供。配置文件不存在时,Catalog 管理命令会创建一个空 Catalog。 +Directory 配置只管理用户、Dataset 和授权关系。它是唯一事实来源;运行时服务配置仍由 `pchronicle serve` 参数提供。配置文件不存在时,Catalog 管理命令会创建一个空 Catalog(带 `[meta]`)。 + +权威 schema(与当前实现 / 部署样例一致): ```toml [meta] version = 1 revision = 1 -name = "team-catalog" +name = "default" [users.alice] -display_name = "Alice" -status = "active" -access_key = "USER_AK" -secret_key = "USER_SK" +access_key = "pcak_…" +secret_key = "…" + +[datasets.default] +uri = "/data/warehouse" [datasets.prod] -display_name = "Production trajectories" -description = "Production agent trajectories" -status = "active" -uri = "s3://bucket/prod" -endpoint = "http://127.0.0.1:9000" -region = "us-west-2" -access_key = "BACKEND_AK" -secret_key = "BACKEND_SK" +uri = "s3://prod/" +endpoint = "http://s3-a.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_A" +secret_key = "BACKEND_SK_A" + +[datasets.prod2] +uri = "s3://prod" +endpoint = "http://s3-b.example:8060" +region = "us-east-1" +access_key = "BACKEND_AK_B" +secret_key = "BACKEND_SK_B" [[grants]] -user = "alice" +user = "*" dataset = "prod" -permissions = ["read", "query", "analyze"] + +[[grants]] +user = "*" +dataset = "prod2" + +[[grants]] +user = "alice" +dataset = "default" +# permissions 可选;v1 忽略细粒度语义,仅表示库成员关系 +# permissions = ["read", "query", "analyze"] ``` 规则: -- `meta.version` 必须为支持的配置版本;每次成功写入 MUST 递增 `meta.revision`。 +- `meta.version` 必须为支持的配置版本;CLI 成功写入后 SHOULD 维护 `meta.revision` / `meta.name`(可缺省)。 - 用户名和 Dataset 名必须是小写 `[A-Za-z_][A-Za-z0-9_]*`。 -- `users.*.access_key` 必须全局唯一;第一版允许明文 `secret_key`。 -- Dataset 的 `uri` 必须是有效的本地、`s3://`、`az://`、`gs://` 或测试存储 URI。 -- 对象存储 Dataset 可以设置 `endpoint`、`region`、`access_key` 和 `secret_key`;本地 Dataset 不需要这些字段。 -- `grants.user` 和 `grants.dataset` 必须分别引用已存在的用户和 Dataset。 -- 同一用户和 Dataset 的 grant 不得重复;权限只能来自 `read`、`query`、`analyze`、`write`、`admin`。 -- 配置文件大小必须有上界;解析或校验失败时服务拒绝启动。 +- `[users.*]` 只含 `access_key` / `secret_key`;`access_key` 必须全局唯一;第一版允许明文 `secret_key`。 +- `[datasets.*]` 必含 `uri`;本地 path 不得设置后端密钥;`s3://` MUST 同时设置 `access_key` 与 `secret_key`,并可设 `endpoint` / `region`。 +- 不同 Dataset MAY 使用不同的 endpoint / region / 后端密钥(见进程模型第 6 条)。 +- `[[grants]]` 必含 `user` 与 `dataset`;`permissions` 可选,v1 不强制执行。 +- `grants.user = "*"`:将该 Dataset 标为公开(匿名可列出/浏览),并在解析时展开给配置中**当前全部**用户;新签发用户在热加载后自动继承。 +- `grants.user` 为具名用户时必须引用已存在用户;`grants.dataset` 必须引用已存在 Dataset。 +- 同一用户与 Dataset 的 grant 不得重复(含 `*` 展开后的冲突)。 +- 配置文件大小必须有上界;解析或校验失败时服务拒绝启动;热加载失败 MUST 保留上一份有效 ACL。 - TOML 是权威配置,后续 SQLite/Postgres 只能作为索引和派生投影。 ## CLI 管理 @@ -171,7 +192,7 @@ pchronicle serve catalog grant --catalog-config FILE NAME DATASET... pchronicle serve catalog revoke --catalog-config FILE NAME DATASET... ``` -`dataset add` 只登记 Dataset,不创建或删除后端数据。`issue` 生成用户 AK/SK(secret 只在本次 stdout 输出)。`grant` / `revoke` 修改 `[[grants]]`。所有写操作 MUST 原子替换文件,失败时保留原文件。`serve --catalog-config` MUST 挂载文件中全部 datasets。 +`dataset add` 只登记 Dataset,不创建或删除后端数据。通过 CLI 追加 `s3://` 时,若文件中已有其它 `s3://`,新条目的 endpoint / region / 后端密钥 MUST 与之完全一致(手写多后端配置仍合法,但 worker 须按第 6 条选库)。`issue` 生成用户 AK/SK(secret 只在本次 stdout 输出),MUST NOT 写入任何 grant。`grant` / `revoke` 修改 `[[grants]]`;NAME 为 `*` 时对**当前全部用户**逐个写入/删除具名 grant(不是写入 `user = "*"` 公开行)。所有写操作 MUST 原子替换文件,失败时保留原文件。 ## HTTP @@ -183,16 +204,17 @@ Directory 路由与 Warehouse 共用 `/api` 与 `/api/v1` 前缀。鉴权头: | `x-pchronicle-secret-key` | 用户 secret key | 缺失、空白或密钥不匹配 MUST 返回 `401`,且 MUST NOT 区分“用户不存在”与“密钥错误”。 +完全无 catalog 头时,列表接口 MAY 只返回 `user = "*"` 的公开 Dataset;带残缺头仍 MUST `401`。 未授权的 library 名与不存在的 library 名 MUST 都返回 `404`。 | 路由 | 父进程 | 响应 | |---|---|---| -| `GET /api/v1/catalog/datasets` | 是 | 该用户可见 library 的 `name`、`uri`、可选 `endpoint`/`region`;**不含**后端密钥 | -| `GET /api/v1/catalog/datasets/{name}` | 是 | 授权时返回完整票,含后端 `access_key` / `secret_key` | +| `GET /api/v1/catalog/datasets` | 是 | 鉴权用户可见 library,或匿名时的公开 library:`name`、`uri`、可选 `endpoint`/`region`;**不含**后端密钥 | +| `GET /api/v1/catalog/datasets/{name}` | 是 | 鉴权且授权时返回完整票,含后端 `access_key` / `secret_key`;匿名 MUST `401`(公开库只允许无密钥列表/浏览) | | `GET /api/health` | 是 | 无鉴权 | | 静态 UI | 是 | 无鉴权 | -| 其余 `/api/*`(含 `GET /api/catalog`,返回当前 Snapshot) | 否,转发 worker | 先鉴权,再按用户 mounts 执行 | +| 其余 `/api/*`(含 `GET /api/catalog`,返回当前 Snapshot) | 否,转发 worker | 先鉴权,再按用户 mounts 执行;多后端时须 `dataset=` | 错误 JSON 沿用 Warehouse 的 `code`、`message`、`request_id`。日志可以包含用户段名、library 名和 `request_id`,MUST NOT 打印用户钥或后端钥。 @@ -234,7 +256,7 @@ Settings(左侧 **Keys**)保存 catalog 用户钥到 `localStorage`: 浏览器把这两项作为上述 HTTP 头附加到 **发往当前 pChronicle serve 的** `/api/` 请求。这与 Assistant 的 Browser BYOK 相反:Assistant 钥只发给模型端点,catalog 钥必须到达 serve 才能鉴权。 -未配置用户钥时,Web MUST NOT 假装本地 Warehouse 已授权;catalog 模式下无头请求在数据面得到 `401`。无 `--catalog-config` 的普通 serve 不要求这些头。 +未配置用户钥时,Web 仍可浏览 `user = "*"` 公开库;需鉴权的数据面请求 MUST `401`。无 `--catalog-config` 的普通 serve 不要求这些头。 查询在 worker 中执行。浏览器不直接持有后端对象存储密钥。 @@ -286,19 +308,21 @@ Worker 用票构造 `ChronicleServerConfig` mounts,执行与普通 Warehouse - 无 `--catalog-config` 时,现有 Dataset 引用、普通 pin 的 `@name/suffix` 路径拼接、以及无鉴权 loopback Warehouse MUST 保持不变。 - `catalog://` MUST NOT 成为 `DatasetLocation` 可打开的存储 scheme;只有 dataset pin 解析器认识它。 -- 新增 library 字段、鉴权头或 worker 协议属于破坏性变更,需要修订本 RFC。 -- 未来的 STS 或热加载可以作为后续 RFC,不得 silently 改变“透传后端密钥 / 重启生效”的语义。 +- 权威配置键为 `meta` / `users` / `datasets` / `grants`;旧式 `[libraries.*]` 或把授权嵌在 `users.*.datasets` 的写法 MUST NOT 再作为规范。 +- 新增 Dataset 字段、鉴权头或 worker 协议属于破坏性变更,需要修订本 RFC。 +- 未来的 STS 或 Dataset 热加载可以作为后续 RFC,不得 silently 改变“透传后端密钥 / Dataset 变更须重启”的语义。 -本 RFC 修正架构文档中“loopback Warehouse 完全没有 authentication”的表述:在 `--catalog-config` 下,数据面和 Directory 路由使用用户钥请求头;它仍不是公网多租户服务。 +本 RFC 修正架构文档中“loopback Warehouse 完全没有 authentication”的表述:在 `--catalog-config` 下,数据面和 Directory 路由使用用户钥请求头;公开库除外。它仍不是公网多租户服务。 ## 实施状态 当前实现覆盖本 RFC 的核心范围: -- `catalog.toml` 解析与启动期校验; -- `pchronicle serve catalog issue|grant|revoke` 改写 ACL(签发不授权,sk 只打一次 stdout); -- `GET /api/v1/catalog/datasets` 与 `/{name}`; -- `--catalog-config` front-only 父进程与 `--catalog-query-worker`; +- TOML Directory 配置解析与启动期校验(`meta` / `users` / `datasets` / `[[grants]]`,含 `user = "*"`); +- `pchronicle serve catalog issue|grant|revoke|dataset …` 改写配置(签发不授权,sk 只打一次 stdout); +- 用户与授权约 3 秒热加载;Dataset / 后端凭证变更拒绝热加载并保留旧 ACL; +- `GET /api/v1/catalog/datasets` 与 `/{name}`(含匿名公开列表); +- `--catalog-config` front-only 父进程与 `--catalog-query-worker`;多后端时按 `dataset=` 收窄 mount; - `catalog://` pin、`@team/prod` 换票与进程内票缓存; - Web `localStorage` 用户钥与数据面请求头。 @@ -306,4 +330,5 @@ Worker 用票构造 `ChronicleServerConfig` mounts,执行与普通 Warehouse 1. 覆盖真实 worker 子进程的集成测试(环境中不得出现未授权 library 的密钥); 2. 评估是否为本地路径 library 提供与 S3 相同的显式审计日志字段; -3. `issue --rotate`:轮换已有用户密钥(当前重名签发直接拒绝)。 +3. `issue --rotate`:轮换已有用户密钥(当前重名签发直接拒绝); +4. 是否让 CLI `dataset add` 正式支持写入多套 S3 后端(当前手写合法,CLI 追加仍要求与已有 s3 后端一致)。