From d54b62366927e2119ef3ac2d31bd39fd8b3b92c7 Mon Sep 17 00:00:00 2001 From: tison Date: Sun, 20 Sep 2026 23:38:16 +0800 Subject: [PATCH 01/12] refactor: remove the region lifecycle backend trait --- cache2/src/cache.rs | 3 +- cache2/src/region/file_backend/mod.rs | 111 +++---- cache2/src/region/file_backend/tests.rs | 126 +++++++- cache2/src/region/store.rs | 370 ++---------------------- 4 files changed, 183 insertions(+), 427 deletions(-) diff --git a/cache2/src/cache.rs b/cache2/src/cache.rs index f138406..f178274 100644 --- a/cache2/src/cache.rs +++ b/cache2/src/cache.rs @@ -46,7 +46,6 @@ use crate::error::ErrorOperation; use crate::error::from_io; use crate::region::file_backend::FileRegionBackend; use crate::region::file_backend::RegionFiles; -use crate::region::file_backend::SystemRegionFileSystem; use crate::region::recovery::DataSuperblock; use crate::region::recovery::PersistentId; use crate::region::recovery::RECOVERY_IMAGE_INDEX_OFFSET; @@ -120,7 +119,7 @@ pub struct Cache { read_recording: bool, mutation_recording: bool, data_plane: RegionDataPlane, - owner: Arc>>>, + owner: Arc>, startup: StartupMode, path: PathBuf, logical_disk_peak_bytes: u64, diff --git a/cache2/src/region/file_backend/mod.rs b/cache2/src/region/file_backend/mod.rs index faac2de..9a65ff3 100644 --- a/cache2/src/region/file_backend/mod.rs +++ b/cache2/src/region/file_backend/mod.rs @@ -87,15 +87,7 @@ use crate::region::recovery::prepare_next_state; use crate::region::recovery::prepare_running_barrier; use crate::region::recovery::recovery_image_index_len; use crate::region::region_metadata_io_error; -#[cfg(test)] -use crate::region::runtime::HybridValueRead; use crate::region::runtime::RegionDataPlane; -use crate::region::store::RegionBackend; -use crate::region::store::RegionStore; -#[cfg(test)] -use crate::snapshot::CacheSnapshot; -#[cfg(test)] -use crate::snapshot::DetailedCacheSnapshot; /// Shared shard count for compact concrete-backend fixtures. #[cfg(test)] @@ -245,48 +237,6 @@ impl FileRegionRuntime { } } -impl RegionStore> { - pub fn data_plane_handle(&self) -> io::Result { - Ok(self.runtime()?.data_plane()?.clone()) - } - - #[cfg(test)] - fn put_value(&self, key: &[u8], value: &[u8]) -> io::Result { - self.runtime()?.data_plane()?.put(key, value) - } - - #[cfg(test)] - fn get_value(&self, key: &[u8]) -> io::Result> { - self.runtime()?.data_plane()?.get(key) - } - - #[cfg(test)] - async fn get_value_async( - &self, - key: &[u8], - tokio_handle: &tokio::runtime::Handle, - ) -> io::Result> { - self.runtime()? - .data_plane()? - .get_async(key, tokio_handle, None) - .await - } - - #[cfg(test)] - pub fn drain(&self) -> io::Result<()> { - self.runtime()?.data_plane()?.drain() - } - - #[cfg(test)] - pub fn snapshot(&self) -> io::Result { - self.runtime()?.data_plane()?.snapshot() - } - - #[cfg(test)] - pub fn detailed_snapshot(&self) -> io::Result { - self.runtime()?.data_plane()?.detailed_snapshot() - } -} pub trait RegionFileSystem { type File: ControlIoBackend; @@ -505,18 +455,9 @@ where self.log_cold_recovery(reason); Ok(None) } -} -impl RegionBackend for FileRegionBackend -where - F: RegionFileSystem, -{ - type Runtime = FileRegionRuntime; - type CleanImage = CleanFileRegionImage; - type FrozenView = FrozenFileRegionView; - type PreparedClean = PreparedFileRegionClean; - - fn acquire_exclusive(&mut self) -> io::Result<()> { + /// Acquire exclusive ownership of all files before inspection. + pub fn acquire_exclusive(&mut self) -> io::Result<()> { if self.locked { return Err(io::Error::new( io::ErrorKind::AlreadyExists, @@ -585,7 +526,12 @@ where Ok(()) } - fn inspect_recovery(&mut self, index_slots: usize) -> io::Result> { + /// Return an eligible clean image, or `None` to select a cold start. + /// This must not allocate or scan the full index or Region data extents. + pub fn inspect_recovery( + &mut self, + index_slots: usize, + ) -> io::Result> { self.file_system .remove_file(&recovery_temporary_path(&self.files.image))?; let format_data = self.format_data; @@ -743,7 +689,8 @@ where })) } - fn anonymous_runtime(&mut self, index_slots: usize) -> io::Result { + /// Construct a provisional empty runtime without starting workers. + pub fn anonymous_runtime(&mut self, index_slots: usize) -> io::Result { self.file_system.remove_file(&self.files.image)?; self.file_system .remove_file(&recovery_temporary_path(&self.files.image))?; @@ -785,11 +732,12 @@ where Ok(runtime) } - fn map_clean_runtime( + /// `Ok(None)` rejects the complete image and selects a cold start. + pub fn map_clean_runtime( &mut self, - mut clean: Self::CleanImage, + mut clean: CleanFileRegionImage, index_slots: usize, - ) -> io::Result> { + ) -> io::Result> { let data = self.data_superblock()?; let expected_slots = u64::try_from(index_slots).map_err(|_| { io::Error::new( @@ -859,7 +807,9 @@ where Ok(Some(runtime)) } - fn publish_running(&mut self) -> io::Result<()> { + /// Replace both state slots with durable `RUNNING` generations so a torn + /// page cannot revive a previous `CLEAN` generation after Region reuse. + pub fn publish_running(&mut self) -> io::Result<()> { let binding = StateBinding::from_data(self.data_superblock()?, None); let barrier = prepare_running_barrier(self.current_state, binding) .map_err(|_| io::Error::other("RUNNING generation cannot advance"))?; @@ -877,7 +827,11 @@ where Ok(()) } - fn start_runtime(&mut self, mut runtime: Self::Runtime) -> io::Result { + /// Start workers only after `RUNNING`; an error must tear them down. + pub fn start_runtime( + &mut self, + mut runtime: FileRegionRuntime, + ) -> io::Result { let data = self.data_superblock()?; let data_file = self .data_file @@ -889,7 +843,8 @@ where Ok(runtime) } - fn stop_fast(&mut self, mut runtime: Self::Runtime) -> io::Result<()> { + /// Quiesce all mutation sources without constructing recovery metadata. + pub fn stop_fast(&mut self, mut runtime: FileRegionRuntime) -> io::Result<()> { match runtime.shutdown_data_plane() { Ok(false) => Ok(()), Ok(true) => { @@ -905,7 +860,11 @@ where } } - fn freeze_warm(&mut self, mut runtime: Self::Runtime) -> io::Result { + /// Quiesce the runtime and return its immutable recovery authority. + pub fn freeze_warm( + &mut self, + mut runtime: FileRegionRuntime, + ) -> io::Result { match runtime.shutdown_data_plane() { Ok(false) => {} Ok(true) => { @@ -941,7 +900,11 @@ where Ok(FrozenFileRegionView { core, metadata }) } - fn persist_frozen(&mut self, view: &Self::FrozenView) -> io::Result { + /// Make completed data and one complete image durable. + pub fn persist_frozen( + &mut self, + view: &FrozenFileRegionView, + ) -> io::Result { let health = &view.core.health; health.require_healthy()?; let source_metadata = &view.metadata; @@ -1067,7 +1030,8 @@ where }) } - fn publish_clean(&mut self, prepared: Self::PreparedClean) -> io::Result<()> { + /// Publish `CLEAN` durably using the token returned after persistence. + pub fn publish_clean(&mut self, prepared: PreparedFileRegionClean) -> io::Result<()> { prepared.health.require_healthy()?; if self.prepared_clean.take() != Some((prepared.state.slot, prepared.state.record)) { return Err(io::Error::new( @@ -1102,7 +1066,8 @@ where Ok(()) } - fn release_exclusive(&mut self) -> io::Result<()> { + /// Release ownership, including during error unwinding after acquisition. + pub fn release_exclusive(&mut self) -> io::Result<()> { if !self.locked { return Ok(()); } diff --git a/cache2/src/region/file_backend/tests.rs b/cache2/src/region/file_backend/tests.rs index 8d9d4b5..6160c64 100644 --- a/cache2/src/region/file_backend/tests.rs +++ b/cache2/src/region/file_backend/tests.rs @@ -63,10 +63,47 @@ use crate::region::record::codec::required_record_bytes; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; use crate::region::recovery::PersistentId; +use crate::region::runtime::HybridValueRead; use crate::region::staging::RegionStaging; use crate::region::staging::StagedRecord; +use crate::region::store::RegionStore; +use crate::snapshot::CacheSnapshot; +use crate::snapshot::DetailedCacheSnapshot; use crate::snapshot::StartupMode; +impl RegionStore { + fn put_value(&self, key: &[u8], value: &[u8]) -> io::Result { + self.runtime()?.data_plane()?.put(key, value) + } + + fn get_value(&self, key: &[u8]) -> io::Result> { + self.runtime()?.data_plane()?.get(key) + } + + async fn get_value_async( + &self, + key: &[u8], + tokio_handle: &tokio::runtime::Handle, + ) -> io::Result> { + self.runtime()? + .data_plane()? + .get_async(key, tokio_handle, None) + .await + } + + pub fn drain(&self) -> io::Result<()> { + self.runtime()?.data_plane()?.drain() + } + + pub fn snapshot(&self) -> io::Result { + self.runtime()?.data_plane()?.snapshot() + } + + pub fn detailed_snapshot(&self) -> io::Result { + self.runtime()?.data_plane()?.detailed_snapshot() + } +} + static NEXT_TEST_DIRECTORY: AtomicU64 = AtomicU64::new(0); fn eventually_admitted(mut put: impl FnMut() -> io::Result) -> T { @@ -142,6 +179,7 @@ fn state_page_reads_stop_after_the_interrupted_retry_budget() { #[derive(Clone, Copy, Debug, Eq, PartialEq)] enum FileSystemFault { + CloneRuntimeFiles, Rename, SyncParent, } @@ -199,6 +237,11 @@ impl RegionFileSystem for FaultRegionFileSystem { } } + fn try_clone_runtime_files(&self, _file: &Self::File) -> io::Result> { + self.file_system.check(FileSystemFault::CloneRuntimeFiles)?; + Ok(None) + } + fn create_new(&self, path: &Path) -> io::Result { FaultBackend::create_new_buffered_with_handle(path, self.io.clone()) } @@ -1463,6 +1506,47 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { cold.close_fast().unwrap(); } +#[test] +fn invalid_capacity_is_rejected_before_creating_files() { + for index_slots in [0, 1, 7, usize::MAX] { + let directory = TestDirectory::new(); + let opened = RegionStore::open( + index_slots, + FileRegionBackend::for_test(directory.files.clone(), test_data_superblock(), 8), + ); + assert_eq!(opened.err().unwrap().kind(), io::ErrorKind::InvalidInput); + assert_eq!(fs::read_dir(&directory.root).unwrap().count(), 0); + } +} + +#[test] +fn dropping_a_warm_runtime_releases_ownership_and_reopens_cold() { + let directory = TestDirectory::new(); + let data = test_data_superblock(); + let mut initial = RegionStore::open( + 8, + FileRegionBackend::for_test(directory.files.clone(), data, 8), + ) + .unwrap(); + initial.close_warm().unwrap(); + + let recovered = RegionStore::open( + 8, + FileRegionBackend::for_test(directory.files.clone(), data, 8), + ) + .unwrap(); + assert_eq!(recovered.startup(), StartupMode::Warm); + drop(recovered); + + let mut cold = RegionStore::open( + 8, + FileRegionBackend::for_test(directory.files.clone(), data, 8), + ) + .unwrap(); + assert_eq!(cold.startup(), StartupMode::Cold); + cold.close_fast().unwrap(); +} + #[test] fn complete_warm_image_maps_without_rebuilding_index_slots() { let directory = TestDirectory::new(); @@ -1490,6 +1574,13 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { .unwrap(); first.close_warm().unwrap(); assert!(directory.files.image.exists()); + first.close_warm().unwrap(); + first.close_fast().unwrap(); + assert_eq!( + first.runtime().err().unwrap().kind(), + io::ErrorKind::BrokenPipe + ); + drop(first); let mut recovered = RegionStore::open( config, @@ -1717,8 +1808,9 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let (file_system, faults, _) = FaultRegionFileSystem::new(); + let (file_system, faults, file_system_faults) = FaultRegionFileSystem::new(); faults.arm(event, occurrence, action); + file_system_faults.arm(FileSystemFault::CloneRuntimeFiles); let opened = RegionStore::open( config, FileRegionBackend::for_test_with_file_system( @@ -1729,6 +1821,11 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { ), ); assert!(opened.is_err(), "RUNNING barrier case {case}"); + assert_eq!( + *file_system_faults.armed.lock().unwrap(), + Some(FileSystemFault::CloneRuntimeFiles), + "runtime files must not be requested before RUNNING is durable" + ); let mut cold = RegionStore::open( config, @@ -1740,6 +1837,33 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { } } +#[test] +fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { + let directory = TestDirectory::new(); + let data = test_data_superblock(); + let (file_system, faults, file_system_faults) = FaultRegionFileSystem::new(); + file_system_faults.arm(FileSystemFault::CloneRuntimeFiles); + let opened = RegionStore::open( + 8, + FileRegionBackend::for_test_with_file_system(directory.files.clone(), data, 8, file_system), + ); + assert_eq!(opened.err().unwrap().raw_os_error(), Some(5)); + assert_eq!(*file_system_faults.armed.lock().unwrap(), None); + assert!( + faults + .events() + .contains(&FaultEvent::Sync(SyncPoint::RunningState)) + ); + + let mut cold = RegionStore::open( + 8, + FileRegionBackend::for_test(directory.files.clone(), data, 8), + ) + .unwrap(); + assert_eq!(cold.startup(), StartupMode::Cold); + cold.close_fast().unwrap(); +} + #[test] fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { let directory = TestDirectory::new(); diff --git a/cache2/src/region/store.rs b/cache2/src/region/store.rs index 868ceec..895c79b 100644 --- a/cache2/src/region/store.rs +++ b/cache2/src/region/store.rs @@ -15,7 +15,7 @@ //! Recovery and shutdown state machine for the Region-backed cache. //! //! The coordinator owns no file-format or data-plane logic. A -//! [`RegionBackend`] supplies those operations, while this module enforces the +//! [`FileRegionBackend`] supplies those operations, while this module enforces the //! order that makes warm recovery safe: //! //! - inspect recovery before constructing an index; @@ -25,67 +25,24 @@ use std::io; +use crate::region::file_backend::FileRegionBackend; +use crate::region::file_backend::FileRegionRuntime; +use crate::region::file_backend::RegionFileSystem; +use crate::region::file_backend::SystemRegionFileSystem; use crate::region::index::storage::validated_index_partition_ranges; +use crate::region::runtime::RegionDataPlane; use crate::snapshot::StartupMode; -/// Physical lifecycle operations required by [`RegionStore`]. -pub trait RegionBackend { - type Runtime; - type CleanImage; - type FrozenView; - type PreparedClean; - - /// Acquire exclusive ownership of all files before inspection. - fn acquire_exclusive(&mut self) -> io::Result<()>; - - /// Return an eligible clean image, or `None` to select a cold start. - /// This must not allocate or scan the full index or Region data extents. - fn inspect_recovery(&mut self, index_slots: usize) -> io::Result>; - - /// Construct a provisional empty runtime without starting workers. - fn anonymous_runtime(&mut self, index_slots: usize) -> io::Result; - - /// `Ok(None)` rejects the complete image and selects a cold start. - fn map_clean_runtime( - &mut self, - clean: Self::CleanImage, - index_slots: usize, - ) -> io::Result>; - - /// Publish `RUNNING` durably before the runtime can be observed. The - /// concrete backend replaces both state slots so a torn page cannot revive - /// a previous `CLEAN` generation after Region reuse starts. - fn publish_running(&mut self) -> io::Result<()>; - - /// Start workers only after `RUNNING`; an error must tear them down. - fn start_runtime(&mut self, runtime: Self::Runtime) -> io::Result; - - /// Quiesce all mutation sources without constructing recovery metadata. - fn stop_fast(&mut self, runtime: Self::Runtime) -> io::Result<()>; - - /// Quiesce the runtime and return its immutable recovery authority. - fn freeze_warm(&mut self, runtime: Self::Runtime) -> io::Result; - - /// Make completed data and one complete image durable. - fn persist_frozen(&mut self, view: &Self::FrozenView) -> io::Result; - - /// Publish `CLEAN` durably using the token returned after persistence. - fn publish_clean(&mut self, prepared: Self::PreparedClean) -> io::Result<()>; - - /// Release ownership, including during error unwinding after acquisition. - fn release_exclusive(&mut self) -> io::Result<()>; -} - -/// Owns the exclusive lifecycle of one backend runtime. -pub struct RegionStore { - backend: B, - runtime: Option, +/// Owns the files and runtime for one Region-backed cache. +pub struct RegionStore { + backend: FileRegionBackend, + runtime: Option, startup: StartupMode, closed: bool, } -impl RegionStore { - pub fn open(index_slots: usize, mut backend: B) -> io::Result { +impl RegionStore { + pub fn open(index_slots: usize, mut backend: FileRegionBackend) -> io::Result { validate_index_slots(index_slots)?; backend.acquire_exclusive()?; @@ -122,7 +79,11 @@ impl RegionStore { self.startup } - pub fn runtime(&self) -> io::Result<&B::Runtime> { + pub fn data_plane_handle(&self) -> io::Result { + Ok(self.runtime()?.data_plane()?.clone()) + } + + pub fn runtime(&self) -> io::Result<&FileRegionRuntime> { if self.closed { return Err(closed_error()); } @@ -130,7 +91,7 @@ impl RegionStore { } #[cfg(test)] - pub fn runtime_mut(&mut self) -> io::Result<&mut B::Runtime> { + pub fn runtime_mut(&mut self) -> io::Result<&mut FileRegionRuntime> { if self.closed { return Err(closed_error()); } @@ -168,7 +129,7 @@ impl RegionStore { } } -impl Drop for RegionStore { +impl Drop for RegionStore { fn drop(&mut self) { if !self.closed { let _ = self.close_fast(); @@ -191,296 +152,3 @@ fn validate_index_slots(index_slots: usize) -> io::Result<()> { .map(|_| ()) .map_err(|error| io::Error::new(io::ErrorKind::InvalidInput, error)) } - -#[cfg(test)] -mod tests { - use std::cell::RefCell; - use std::rc::Rc; - - use super::*; - - #[derive(Clone, Copy, Debug, Eq, PartialEq)] - enum Event { - Lock, - Inspect, - Anonymous, - Map, - Running, - Start, - StopFast, - Freeze, - Persist, - Clean, - Unlock, - } - - #[derive(Clone, Copy)] - enum RecoveryScenario { - Cold, - Clean, - RejectedClean, - } - - struct Backend { - scenario: RecoveryScenario, - events: Rc>>, - fail_at: Option, - } - - impl Backend { - fn record(&self, event: Event) -> io::Result<()> { - self.events.borrow_mut().push(event); - if self.fail_at == Some(event) { - Err(io::Error::other(format!("{event:?}"))) - } else { - Ok(()) - } - } - } - - impl RegionBackend for Backend { - type Runtime = usize; - type CleanImage = bool; - type FrozenView = usize; - type PreparedClean = (); - - fn acquire_exclusive(&mut self) -> io::Result<()> { - self.record(Event::Lock) - } - - fn inspect_recovery( - &mut self, - _index_slots: usize, - ) -> io::Result> { - self.record(Event::Inspect)?; - Ok(match self.scenario { - RecoveryScenario::Cold => None, - RecoveryScenario::Clean => Some(true), - RecoveryScenario::RejectedClean => Some(false), - }) - } - - fn anonymous_runtime(&mut self, index_slots: usize) -> io::Result { - self.record(Event::Anonymous)?; - Ok(index_slots) - } - - fn map_clean_runtime( - &mut self, - clean: Self::CleanImage, - index_slots: usize, - ) -> io::Result> { - self.record(Event::Map)?; - Ok(clean.then_some(index_slots)) - } - - fn publish_running(&mut self) -> io::Result<()> { - self.record(Event::Running) - } - - fn start_runtime(&mut self, runtime: Self::Runtime) -> io::Result { - self.record(Event::Start)?; - Ok(runtime) - } - - fn stop_fast(&mut self, _runtime: Self::Runtime) -> io::Result<()> { - self.record(Event::StopFast) - } - - fn freeze_warm(&mut self, runtime: Self::Runtime) -> io::Result { - self.record(Event::Freeze)?; - Ok(runtime) - } - - fn persist_frozen(&mut self, _view: &Self::FrozenView) -> io::Result { - self.record(Event::Persist) - } - - fn publish_clean(&mut self, _prepared: Self::PreparedClean) -> io::Result<()> { - self.record(Event::Clean) - } - - fn release_exclusive(&mut self) -> io::Result<()> { - self.record(Event::Unlock) - } - } - - fn backend( - scenario: RecoveryScenario, - fail_at: Option, - ) -> (Backend, Rc>>) { - let events = Rc::new(RefCell::new(Vec::new())); - ( - Backend { - scenario, - events: Rc::clone(&events), - fail_at, - }, - events, - ) - } - - #[test] - fn invalid_capacity_is_rejected_before_ownership_or_allocation() { - for index_slots in [0, 1, 7, usize::MAX] { - let (backend, events) = backend(RecoveryScenario::Cold, None); - assert!(RegionStore::open(index_slots, backend).is_err()); - assert!(events.borrow().is_empty()); - } - } - - #[test] - fn recovery_inspection_selects_one_runtime_before_the_running_barrier() { - for (scenario, startup, expected) in [ - ( - RecoveryScenario::Cold, - StartupMode::Cold, - vec![ - Event::Lock, - Event::Inspect, - Event::Anonymous, - Event::Running, - Event::Start, - ], - ), - ( - RecoveryScenario::Clean, - StartupMode::Warm, - vec![ - Event::Lock, - Event::Inspect, - Event::Map, - Event::Running, - Event::Start, - ], - ), - ( - RecoveryScenario::RejectedClean, - StartupMode::Cold, - vec![ - Event::Lock, - Event::Inspect, - Event::Map, - Event::Anonymous, - Event::Running, - Event::Start, - ], - ), - ] { - let (backend, events) = backend(scenario, None); - let mut store = RegionStore::open(8, backend).unwrap(); - assert_eq!(store.startup(), startup); - assert_eq!(*events.borrow(), expected); - store.close_fast().unwrap(); - } - } - - #[test] - fn shutdown_modes_are_disjoint_and_release_ownership() { - for (warm, expected) in [ - (false, vec![Event::StopFast, Event::Unlock]), - ( - true, - vec![Event::Freeze, Event::Persist, Event::Clean, Event::Unlock], - ), - ] { - let (backend, events) = backend(RecoveryScenario::Cold, None); - let mut store = RegionStore::open(8, backend).unwrap(); - events.borrow_mut().clear(); - if warm { - store.close_warm().unwrap(); - } else { - store.close_fast().unwrap(); - } - assert_eq!(*events.borrow(), expected); - } - } - - #[test] - fn drop_uses_the_non_recoverable_shutdown_path() { - let (backend, events) = backend(RecoveryScenario::Cold, None); - let store = RegionStore::open(8, backend).unwrap(); - events.borrow_mut().clear(); - - drop(store); - - assert_eq!(*events.borrow(), vec![Event::StopFast, Event::Unlock]); - } - - #[test] - fn open_failure_stops_at_the_failed_stage_and_releases_ownership() { - for (scenario, failed, expected) in [ - ( - RecoveryScenario::Cold, - Event::Inspect, - vec![Event::Lock, Event::Inspect, Event::Unlock], - ), - ( - RecoveryScenario::Cold, - Event::Anonymous, - vec![Event::Lock, Event::Inspect, Event::Anonymous, Event::Unlock], - ), - ( - RecoveryScenario::Clean, - Event::Map, - vec![Event::Lock, Event::Inspect, Event::Map, Event::Unlock], - ), - ( - RecoveryScenario::Cold, - Event::Running, - vec![ - Event::Lock, - Event::Inspect, - Event::Anonymous, - Event::Running, - Event::Unlock, - ], - ), - ( - RecoveryScenario::Cold, - Event::Start, - vec![ - Event::Lock, - Event::Inspect, - Event::Anonymous, - Event::Running, - Event::Start, - Event::Unlock, - ], - ), - ] { - let (backend, events) = backend(scenario, Some(failed)); - assert!(RegionStore::open(8, backend).is_err()); - assert_eq!(*events.borrow(), expected, "failed at {failed:?}"); - } - } - - #[test] - fn shutdown_failure_does_not_cross_a_publication_boundary() { - for (warm, failed, expected) in [ - (false, Event::StopFast, vec![Event::StopFast, Event::Unlock]), - (true, Event::Freeze, vec![Event::Freeze, Event::Unlock]), - ( - true, - Event::Persist, - vec![Event::Freeze, Event::Persist, Event::Unlock], - ), - ( - true, - Event::Clean, - vec![Event::Freeze, Event::Persist, Event::Clean, Event::Unlock], - ), - ] { - let (backend, events) = backend(RecoveryScenario::Cold, Some(failed)); - let mut store = RegionStore::open(8, backend).unwrap(); - events.borrow_mut().clear(); - let result = if warm { - store.close_warm() - } else { - store.close_fast() - }; - assert!(result.is_err(), "failed at {failed:?}"); - assert_eq!(*events.borrow(), expected, "failed at {failed:?}"); - } - } -} From fd40d60f7edac84e47e11dee854f7dbf3fa1eda6 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:01:50 +0800 Subject: [PATCH 02/12] refactor: separate positioned I/O from storage file ownership Narrow PositionedIo to byte-offset reads and writes, and move extent management, durability, locking, identity, and mapping access into StorageFile. Rename the concrete file resources and their counters to describe the owned descriptors and file I/O paths. Implement positioned I/O directly on DataFileHandles and remove RuntimeFileBackend, including its unsupported lock methods. Update driver state, variables, fixtures, and test doubles to the same vocabulary without changing dispatch or buffer ownership. Validation: workspace all-target checks, 276 library tests, repository lint, and Linux io-uring cross-target check. --- cache2/src/fixtures.rs | 10 +- cache2/src/io/engine/mod.rs | 136 +++--- cache2/src/io/engine/posix.rs | 94 ++-- cache2/src/io/engine/tests.rs | 296 +++++------- cache2/src/io/engine/uring.rs | 102 ++-- cache2/src/io/{backend.rs => file.rs} | 501 ++++++++------------ cache2/src/io/mod.rs | 2 +- cache2/src/region/appender.rs | 53 +-- cache2/src/region/file_backend/mod.rs | 68 +-- cache2/src/region/file_backend/tests.rs | 30 +- cache2/src/region/manager.rs | 2 +- cache2/src/region/mod.rs | 2 +- cache2/src/region/reader.rs | 52 +- cache2/src/region/record/codec.rs | 2 +- cache2/src/region/runtime/mod.rs | 32 +- cache2/src/region/runtime/shutdown_tests.rs | 38 +- cache2/src/region/staging.rs | 2 +- 17 files changed, 603 insertions(+), 819 deletions(-) rename cache2/src/io/{backend.rs => file.rs} (76%) diff --git a/cache2/src/fixtures.rs b/cache2/src/fixtures.rs index 702c4de..c2e1427 100644 --- a/cache2/src/fixtures.rs +++ b/cache2/src/fixtures.rs @@ -33,8 +33,8 @@ use std::sync::Arc; use std::sync::atomic::AtomicU64; use std::sync::atomic::Ordering; -use crate::io::backend::FileBackend; -use crate::io::backend::IoBackend; +use crate::io::file::CacheFile; +use crate::io::file::PositionedIo; /// Checks every byte, including zero padding, and returns the committed bytes /// for decoder compatibility checks. @@ -124,9 +124,9 @@ impl TestFile { .unwrap() } - /// Opens the file as a buffered backend. - pub fn backend(&self) -> Arc { - Arc::new(FileBackend::open(&self.path).unwrap()) + /// Opens the file for buffered positioned I/O. + pub fn io(&self) -> Arc { + Arc::new(CacheFile::open(&self.path).unwrap()) } } diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index 6eff683..6dc77d4 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -49,7 +49,7 @@ use asyncband::semaphore::Semaphore; #[cfg(unix)] use crate::config::runtime::IoEngineConfig; #[cfg(unix)] -use crate::io::backend::RuntimeFileSet; +use crate::io::file::DataFileHandles; #[cfg(all( feature = "io-uring", target_os = "linux", @@ -61,7 +61,7 @@ use crate::io::backend::RuntimeFileSet; target_arch = "powerpc64" ) ))] -use crate::io::backend::RuntimeIoDirection; +use crate::io::file::FileIoDirection; #[cfg(all( feature = "io-uring", target_os = "linux", @@ -73,10 +73,10 @@ use crate::io::backend::RuntimeIoDirection; target_arch = "powerpc64" ) ))] -use crate::io::backend::RuntimeIoPath; -use crate::io::backend::RuntimeIoStats; -use crate::io::backend::RuntimeIoStatsHandle; -use crate::io::backend::WritePoint; +use crate::io::file::FileIoPath; +use crate::io::file::FileIoStats; +use crate::io::file::FileIoStatsHandle; +use crate::io::file::WritePoint; use crate::managed_memory::BufferLease; use crate::snapshot::CacheIoDirectionSnapshot; @@ -102,14 +102,14 @@ mod uring; /// POSIX workers and io_uring drivers share this command and completion protocol. /// Callers share the engine through `Arc`; the final owner shuts down its workers. pub struct IoEngine { - shared: Arc, + state: Arc, commands: SyncSender, submit_state: Arc>, next_request_id: AtomicU64, wake: Option>, workers: Mutex>>>, shutdown: ShutdownState, - io_stats: RuntimeIoStatsHandle, + io_stats: FileIoStatsHandle, } const IO_BUFFER_ALIGNMENT: usize = 4096; @@ -138,7 +138,7 @@ const CACHE_IO_CANCEL_GRACE: Duration = Duration::from_millis(100); #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] pub struct EngineIoSnapshot { pub requests: CacheIoDirectionSnapshot, - pub runtime: RuntimeIoStats, + pub file_io: FileIoStats, } /// A logical range of an engine-budgeted aligned buffer lease. @@ -351,10 +351,10 @@ impl OperationKind { target_arch = "powerpc64" ) ))] - const fn io_direction(self) -> RuntimeIoDirection { + const fn io_direction(self) -> FileIoDirection { match self { - Self::Read => RuntimeIoDirection::Read, - Self::Write => RuntimeIoDirection::Write, + Self::Read => FileIoDirection::Read, + Self::Write => FileIoDirection::Write, } } } @@ -440,11 +440,11 @@ impl IoOperation { target_arch = "powerpc64" ) ))] - fn runtime_io_path( + fn file_io_path( &self, - files: &RuntimeFileSet, + handles: &DataFileHandles, transferred: usize, - ) -> io::Result { + ) -> io::Result { match self { Self::Read { buffer, offset } => { let remaining = buffer.len().checked_sub(transferred).ok_or_else(|| { @@ -453,7 +453,7 @@ impl IoOperation { let offset = offset.checked_add(transferred as u64).ok_or_else(|| { io::Error::new(io::ErrorKind::InvalidInput, "read offset overflow") })?; - Ok(files.select_path( + Ok(handles.select_path( buffer.read_target()?.wrapping_add(transferred), remaining, offset, @@ -471,7 +471,7 @@ impl IoOperation { let offset = offset.checked_add(transferred as u64).ok_or_else(|| { io::Error::new(io::ErrorKind::InvalidInput, "write offset overflow") })?; - Ok(files.select_path( + Ok(handles.select_path( buffer.as_ptr()?.wrapping_add(transferred), remaining, offset, @@ -1016,7 +1016,7 @@ fn submit_cache_io_until( } struct IoSlot { - shared: Arc, + state: Arc, write: bool, // This permit drops only after `IoSlot::drop` publishes physical capacity. read_permit: Option, @@ -1031,7 +1031,7 @@ pub struct ReadSlot { /// An async reservation handle backed by the engine's physical slot state. pub struct ReadSlotWaiter { - shared: Arc, + state: Arc, } // The physical completion path must not allocate. Asyncband keeps waiter nodes @@ -1142,25 +1142,25 @@ impl ReadSlotWaiter { tokio_handle: &tokio::runtime::Handle, ) -> io::Result { let admission = self - .shared + .state .read_slot_admission .as_ref() .ok_or_else(|| io::Error::other("async read admission is disabled"))?; let _waiter = admission.register_waiter(); - self.shared.ensure_accepting()?; + self.state.ensure_accepting()?; let permit = admission.acquire_until(deadline, tokio_handle).await?; - self.shared.try_reserve_read_slot(Some(permit)) + self.state.try_reserve_read_slot(Some(permit)) } } impl Drop for IoSlot { fn drop(&mut self) { { - let _slot = lock_unpoisoned(&self.shared.slot_lock); - self.shared + let _slot = lock_unpoisoned(&self.state.slot_lock); + self.state .slot_state .fetch_sub(slot_delta(self.write), Ordering::AcqRel); - self.shared.slot_available.notify_one(); + self.state.slot_available.notify_one(); } drop(self.read_permit.take()); } @@ -1182,7 +1182,7 @@ const fn active_write_slots(state: u64) -> usize { ((state >> WRITE_SLOT_SHIFT) & SLOT_COUNT_MASK) as usize } -struct RuntimeShared { +struct EngineState { latency: std::sync::OnceLock, max_in_flight: usize, activity_counters_enabled: bool, @@ -1212,7 +1212,7 @@ enum SlotWaitError { TimedOut, } -impl RuntimeShared { +impl EngineState { fn new(max_in_flight: usize, activity_counters_enabled: bool, read_wait_enabled: bool) -> Self { Self { latency: std::sync::OnceLock::new(), @@ -1276,7 +1276,7 @@ impl RuntimeShared { update_peak(&self.in_flight_peak, total + 1); } return Some(IoSlot { - shared: Arc::clone(self), + state: Arc::clone(self), write, read_permit: None, }); @@ -1616,38 +1616,38 @@ impl IoEngine { /// Installed once during construction, before any requests are admitted. pub fn set_latency_recorder(&self, recorder: crate::stats::recording::IoTiming) { assert!( - self.shared.latency.set(recorder).is_ok(), + self.state.latency.set(recorder).is_ok(), "I/O recorder installed twice" ); } pub fn wake_slot_waiters(&self) { - self.shared.wake_slot_waiters(); + self.state.wake_slot_waiters(); } pub fn in_flight(&self) -> usize { - self.shared.total_in_flight() + self.state.total_in_flight() } pub fn writes_in_flight(&self) -> usize { - self.shared.writes_in_flight() + self.state.writes_in_flight() } /// True means a failed driver could not fence an issued write. /// The cache must retain its exclusive file lock for process lifetime. pub fn has_unfenced_writes(&self) -> bool { - self.shared.has_unfenced_writes() + self.state.has_unfenced_writes() } #[cfg(test)] pub fn mark_unfenced_writes_for_test(&self) { - self.shared.mark_unfenced_writes(); + self.state.mark_unfenced_writes(); } pub fn stats(&self) -> EngineIoSnapshot { EngineIoSnapshot { - requests: self.shared.snapshot(), - runtime: self.io_stats.snapshot(), + requests: self.state.snapshot(), + file_io: self.io_stats.snapshot(), } } @@ -1670,7 +1670,7 @@ impl IoEngine { max_in_flight: usize, activity_counters_enabled: bool, read_wait_enabled: bool, - io_stats: RuntimeIoStatsHandle, + io_stats: FileIoStatsHandle, ) -> io::Result<(Self, Receiver)> { if !(1..=MAX_IO_REQUESTS_PER_ENGINE).contains(&max_in_flight) { return Err(io::Error::new( @@ -1686,7 +1686,7 @@ impl IoEngine { io_stats.set_activity_counters_enabled(activity_counters_enabled); Ok(( Self { - shared: Arc::new(RuntimeShared::new( + state: Arc::new(EngineState::new( max_in_flight, activity_counters_enabled, read_wait_enabled, @@ -1728,17 +1728,17 @@ impl IoEngine { pub fn try_reserve_read(&self) -> io::Result { let permit = self - .shared + .state .read_slot_admission .as_ref() .map(ReadSlotAdmission::try_acquire) .transpose()?; - self.shared.try_reserve_read_slot(permit) + self.state.try_reserve_read_slot(permit) } pub fn read_slot_waiter(&self) -> ReadSlotWaiter { ReadSlotWaiter { - shared: Arc::clone(&self.shared), + state: Arc::clone(&self.state), } } @@ -1747,8 +1747,7 @@ impl IoEngine { slot: ReadSlot, operation: IoOperation, ) -> Result { - if operation.kind() != OperationKind::Read || !Arc::ptr_eq(&slot.slot.shared, &self.shared) - { + if operation.kind() != OperationKind::Read || !Arc::ptr_eq(&slot.slot.state, &self.state) { return Err(SubmitError { error: io::Error::new( io::ErrorKind::InvalidInput, @@ -1801,26 +1800,26 @@ impl IoEngine { return Err(SubmitError { error, operation }); } let write = operation.kind().uses_write_slot(); - let slot_wait_started = self.shared.activity_counters_enabled.then(Instant::now); + let slot_wait_started = self.state.activity_counters_enabled.then(Instant::now); let slot = match slot_mode { #[cfg(test)] - SlotMode::Try if !self.shared.accepting.load(Ordering::Acquire) => Err(io::Error::new( + SlotMode::Try if !self.state.accepting.load(Ordering::Acquire) => Err(io::Error::new( io::ErrorKind::BrokenPipe, "I/O engine is shut down", )), #[cfg(test)] - SlotMode::Try => self.shared.try_reserve_slot(write).ok_or_else(|| { + SlotMode::Try => self.state.try_reserve_slot(write).ok_or_else(|| { io::Error::new(io::ErrorKind::WouldBlock, "no I/O slot is available") }), #[cfg(test)] - SlotMode::Wait => self.shared.reserve_slot_wait(write).ok_or_else(|| { + SlotMode::Wait => self.state.reserve_slot_wait(write).ok_or_else(|| { io::Error::new(io::ErrorKind::BrokenPipe, "I/O engine is shut down") }), SlotMode::Controlled { cancelled, deadline, } => self - .shared + .state .reserve_slot_controlled(write, cancelled, deadline) .map_err(|error| match error { SlotWaitError::Shutdown => { @@ -1841,11 +1840,11 @@ impl IoEngine { } }; if let Some(slot_wait_started) = slot_wait_started { - add_duration_ns(&self.shared.slot_wait_ns, slot_wait_started.elapsed()); + add_duration_ns(&self.state.slot_wait_ns, slot_wait_started.elapsed()); } - let request_started = (self.shared.activity_counters_enabled - || self.shared.latency.get().is_some()) + let request_started = (self.state.activity_counters_enabled + || self.state.latency.get().is_some()) .then(Instant::now); self.submit_with_slot(operation, slot, request_started, nonblocking) } @@ -1893,8 +1892,8 @@ impl IoEngine { slot, submitted_at: request_started, }; - if self.shared.activity_counters_enabled { - self.shared + if self.state.activity_counters_enabled { + self.state .requests_submitted .fetch_add(1, Ordering::Release); } @@ -1911,8 +1910,8 @@ impl IoEngine { }) } Err(TrySendError::Full(DriverCommand::Submit(task))) => { - if self.shared.activity_counters_enabled { - self.shared + if self.state.activity_counters_enabled { + self.state .requests_submitted .fetch_sub(1, Ordering::Relaxed); } @@ -1926,8 +1925,8 @@ impl IoEngine { }) } Err(TrySendError::Disconnected(DriverCommand::Submit(task))) => { - if self.shared.activity_counters_enabled { - self.shared + if self.state.activity_counters_enabled { + self.state .requests_submitted .fetch_sub(1, Ordering::Relaxed); } @@ -1952,9 +1951,7 @@ impl IoEngine { match self.commands.try_send(DriverCommand::Cancel(request_id)) { Ok(()) => {} Err(TrySendError::Full(_)) => { - self.shared - .cancel_scan_needed - .store(true, Ordering::Release); + self.state.cancel_scan_needed.store(true, Ordering::Release); } Err(TrySendError::Disconnected(_)) => { return Err(io::Error::new( @@ -1977,7 +1974,7 @@ impl IoEngine { .write() .unwrap_or_else(|poisoned| poisoned.into_inner()); submit_state.accepting = false; - self.shared.stop_accepting_slots(); + self.state.stop_accepting_slots(); drop(submit_state); if let Some(wake) = &self.wake { wake.wake(); @@ -2015,7 +2012,7 @@ impl IoEngine { .write() .unwrap_or_else(|poisoned| poisoned.into_inner()); submit_state.accepting = false; - self.shared.stop_accepting_slots(); + self.state.stop_accepting_slots(); let mut error = None; for _ in 0..worker_count { if self.commands.send(DriverCommand::Shutdown).is_err() { @@ -2057,14 +2054,14 @@ impl Drop for IoEngine { #[cfg(unix)] pub fn build_file_engine( - files: RuntimeFileSet, + handles: DataFileHandles, config: IoEngineConfig, activity_counters_enabled: bool, read_wait_enabled: bool, ) -> io::Result> { match config { IoEngineConfig::Posix { workers } => posix::start( - files, + handles, workers, workers, activity_counters_enabled, @@ -2084,8 +2081,13 @@ pub fn build_file_engine( ) ))] { - uring::start(files, config, activity_counters_enabled, read_wait_enabled) - .map(Arc::new) + uring::start( + handles, + config, + activity_counters_enabled, + read_wait_enabled, + ) + .map(Arc::new) } #[cfg(not(all( feature = "io-uring", @@ -2099,7 +2101,7 @@ pub fn build_file_engine( ) )))] { - let _ = files; + let _ = handles; let _ = config; Err(io::Error::new( io::ErrorKind::Unsupported, diff --git a/cache2/src/io/engine/posix.rs b/cache2/src/io/engine/posix.rs index fe2b19a..afa109e 100644 --- a/cache2/src/io/engine/posix.rs +++ b/cache2/src/io/engine/posix.rs @@ -20,34 +20,33 @@ use std::sync::Mutex; use std::sync::atomic::Ordering; use std::sync::mpsc::Receiver; -use crate::io::backend::IoBackend; -#[cfg(unix)] -use crate::io::backend::RuntimeFileBackend; -#[cfg(unix)] -use crate::io::backend::RuntimeFileSet; -use crate::io::backend::RuntimeIoStatsHandle; -use crate::io::backend::read_exact_at_uninit_with_progress; -use crate::io::backend::write_all_at_with_progress; use crate::io::engine::CompletionStatus; use crate::io::engine::DriverCommand; +use crate::io::engine::EngineState; use crate::io::engine::IoEngine; use crate::io::engine::IoOperation; -use crate::io::engine::RuntimeShared; use crate::io::engine::lock_unpoisoned; +#[cfg(unix)] +#[cfg(unix)] +use crate::io::file::DataFileHandles; +use crate::io::file::FileIoStatsHandle; +use crate::io::file::PositionedIo; +use crate::io::file::read_exact_at_uninit_with_progress; +use crate::io::file::write_all_at_with_progress; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; #[cfg(unix)] pub fn start( - files: RuntimeFileSet, + handles: DataFileHandles, max_in_flight: usize, worker_count: usize, activity_counters_enabled: bool, read_wait_enabled: bool, ) -> io::Result { - let io_stats = files.stats_handle(); - let backend = Arc::new(RuntimeFileBackend::new(files)); - start_backend( - backend, + let io_stats = handles.stats_handle(); + let io = Arc::new(handles); + start_workers( + io, io_stats, max_in_flight, worker_count, @@ -58,27 +57,27 @@ pub fn start( #[cfg(test)] impl IoEngine { - pub fn for_test(backend: Arc, max_in_flight: usize) -> io::Result { - Self::for_test_with_options(backend, max_in_flight, max_in_flight.min(4), true, false) + pub fn for_test(io: Arc, max_in_flight: usize) -> io::Result { + Self::for_test_with_options(io, max_in_flight, max_in_flight.min(4), true, false) } pub fn for_test_with_read_wait( - backend: Arc, + io: Arc, max_in_flight: usize, ) -> io::Result { - Self::for_test_with_options(backend, max_in_flight, max_in_flight.min(4), true, true) + Self::for_test_with_options(io, max_in_flight, max_in_flight.min(4), true, true) } pub fn for_test_with_options( - backend: Arc, + io: Arc, max_in_flight: usize, worker_count: usize, activity_counters_enabled: bool, read_wait_enabled: bool, ) -> io::Result { - start_backend( - backend, - RuntimeIoStatsHandle::new(false), + start_workers( + io, + FileIoStatsHandle::new(false), max_in_flight, worker_count, activity_counters_enabled, @@ -87,9 +86,9 @@ impl IoEngine { } } -fn start_backend( - backend: Arc, - io_stats: RuntimeIoStatsHandle, +fn start_workers( + io: Arc, + io_stats: FileIoStatsHandle, max_in_flight: usize, worker_count: usize, activity_counters_enabled: bool, @@ -114,22 +113,22 @@ fn start_backend( .unwrap() .reserve_exact(worker_count); for worker_index in 0..worker_count { - let worker_backend = Arc::clone(&backend); - let worker_shared = Arc::clone(&engine.shared); + let worker_io = Arc::clone(&io); + let worker_state = Arc::clone(&engine.state); let worker_receiver = Arc::clone(&receiver); let worker = std::thread::Builder::new() .name(format!("cache2-sync-io-{worker_index}")) .stack_size(CACHE_THREAD_STACK_BYTES) - .spawn(move || backend_driver(worker_backend, worker_shared, worker_receiver))?; + .spawn(move || posix_worker(worker_io, worker_state, worker_receiver))?; // Retain each worker immediately so engine Drop joins it if a later spawn fails. engine.workers.get_mut().unwrap().push(worker); } Ok(engine) } -fn backend_driver( - backend: Arc, - shared: Arc, +fn posix_worker( + io: Arc, + state: Arc, receiver: Arc>>, ) -> io::Result<()> { loop { @@ -140,19 +139,19 @@ fn backend_driver( match command { DriverCommand::Submit(mut task) => { if task.completion.cancel_requested.load(Ordering::Acquire) { - shared.finish(task, CompletionStatus::Cancelled, 0); + state.finish(task, CompletionStatus::Cancelled, 0); continue; } let (status, transferred) = panic::catch_unwind(AssertUnwindSafe(|| { - execute_backend(backend.as_ref(), &mut task.operation) + execute_operation(io.as_ref(), &mut task.operation) })) .unwrap_or_else(|_| { ( - CompletionStatus::Failed(io::Error::other("I/O backend panicked")), + CompletionStatus::Failed(io::Error::other("positioned I/O panicked")), 0, ) }); - shared.finish(task, status, transferred); + state.finish(task, status, transferred); } DriverCommand::Cancel(request_id) => { // The cancel flag is visible directly through CompletionState. @@ -165,22 +164,18 @@ fn backend_driver( Ok(()) } -fn execute_backend( - backend: &dyn IoBackend, +fn execute_operation( + io: &dyn PositionedIo, operation: &mut IoOperation, ) -> (CompletionStatus, usize) { match operation { IoOperation::Read { buffer, offset } => match buffer.read_target() { Ok(buffer_pointer) => { - let (result, transferred) = read_exact_at_uninit_with_progress( - backend, - buffer_pointer, - buffer.len(), - *offset, - ); - backend_result(result, transferred) + let (result, transferred) = + read_exact_at_uninit_with_progress(io, buffer_pointer, buffer.len(), *offset); + completion_result(result, transferred) } - Err(error) => backend_result(Err(error), 0), + Err(error) => completion_result(Err(error), 0), }, IoOperation::Write { point, @@ -188,16 +183,15 @@ fn execute_backend( offset, } => match buffer.as_slice() { Ok(buffer) => { - let (result, transferred) = - write_all_at_with_progress(backend, *point, buffer, *offset); - backend_result(result, transferred) + let (result, transferred) = write_all_at_with_progress(io, *point, buffer, *offset); + completion_result(result, transferred) } - Err(error) => backend_result(Err(error), 0), + Err(error) => completion_result(Err(error), 0), }, } } -fn backend_result(result: io::Result<()>, transferred: usize) -> (CompletionStatus, usize) { +fn completion_result(result: io::Result<()>, transferred: usize) -> (CompletionStatus, usize) { match result { Ok(()) => (CompletionStatus::Completed, transferred), Err(error) => (CompletionStatus::Failed(error), transferred), diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index 69dc461..05d37b2 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -21,9 +21,7 @@ use crate::IoOutcome; use crate::IoRole; use crate::StatsOptions; use crate::fixtures::TestFile; -use crate::io::backend::IoBackend; -use crate::io::backend::SyncMode; -use crate::io::backend::SyncPoint; +use crate::io::file::PositionedIo; use crate::managed_memory::ManagedMemory; use crate::managed_memory::ManagedMemoryLimits; use crate::managed_memory::aligned_buffer_capacity; @@ -32,7 +30,7 @@ use crate::stats::recording::Recorder; async fn wait_for_registered_read_waiters(engine: &IoEngine, expected: usize) { for _ in 0..100 { let actual = engine - .shared + .state .read_slot_admission .as_ref() .map_or(0, |admission| admission.waiters.load(Ordering::Acquire)); @@ -75,12 +73,12 @@ struct BlockingState { } #[derive(Default)] -struct BlockingBackend { +struct BlockingIo { state: Mutex, changed: Condvar, } -impl BlockingBackend { +impl BlockingIo { fn wait_for_entered(&self, expected: usize) -> bool { let state = lock_unpoisoned(&self.state); let (state, _) = self @@ -118,15 +116,7 @@ impl BlockingBackend { } } -impl IoBackend for BlockingBackend { - fn len(&self) -> io::Result { - Ok(1024 * 1024) - } - - fn set_len(&self, _len: u64) -> io::Result<()> { - Ok(()) - } - +impl PositionedIo for BlockingIo { fn read_at(&self, buffer: &mut [u8], _offset: u64) -> io::Result { self.enter_and_wait(); buffer.fill(0); @@ -137,31 +127,19 @@ impl IoBackend for BlockingBackend { self.enter_and_wait(); Ok(buffer.len()) } - - fn sync(&self, _point: SyncPoint, _mode: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } -struct PanicOnceBackend { +struct PanicOnceIo { panic_next_read: AtomicBool, } #[derive(Default)] -struct ShortThenErrorBackend { +struct ShortThenErrorIo { read_calls: AtomicUsize, write_calls: AtomicUsize, } -impl PanicOnceBackend { +impl PanicOnceIo { fn new() -> Self { Self { panic_next_read: AtomicBool::new(true), @@ -169,18 +147,10 @@ impl PanicOnceBackend { } } -impl IoBackend for PanicOnceBackend { - fn len(&self) -> io::Result { - Ok(1024 * 1024) - } - - fn set_len(&self, _len: u64) -> io::Result<()> { - Ok(()) - } - +impl PositionedIo for PanicOnceIo { fn read_at(&self, buffer: &mut [u8], _offset: u64) -> io::Result { if self.panic_next_read.swap(false, Ordering::AcqRel) { - panic!("injected backend panic"); + panic!("injected io panic"); } buffer.fill(0); Ok(buffer.len()) @@ -189,29 +159,9 @@ impl IoBackend for PanicOnceBackend { fn write_at(&self, _point: WritePoint, buffer: &[u8], _offset: u64) -> io::Result { Ok(buffer.len()) } - - fn sync(&self, _point: SyncPoint, _mode: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } -impl IoBackend for ShortThenErrorBackend { - fn len(&self) -> io::Result { - Ok(1024 * 1024) - } - - fn set_len(&self, _len: u64) -> io::Result<()> { - Ok(()) - } - +impl PositionedIo for ShortThenErrorIo { fn read_at(&self, buffer: &mut [u8], _offset: u64) -> io::Result { if self.read_calls.fetch_add(1, Ordering::Relaxed) == 0 { let transferred = 3.min(buffer.len()); @@ -229,18 +179,6 @@ impl IoBackend for ShortThenErrorBackend { Err(io::Error::from_raw_os_error(5)) } } - - fn sync(&self, _point: SyncPoint, _mode: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } fn managed_memory() -> Arc { @@ -278,7 +216,7 @@ fn aligned_buffer_has_stable_alignment() { #[test] fn posix_engine_round_trips_owned_buffers_and_drains() { let file = TestFile::new("io-engine"); - let engine = IoEngine::for_test(file.backend(), 4).unwrap(); + let engine = IoEngine::for_test(file.io(), 4).unwrap(); let managed_memory = managed_memory(); let input = b"owned async positioned I/O"; let write = engine @@ -317,7 +255,7 @@ fn posix_engine_round_trips_owned_buffers_and_drains() { #[test] fn posix_engine_reports_progress_before_a_terminal_short_io_error() { - let engine = IoEngine::for_test(Arc::new(ShortThenErrorBackend::default()), 2).unwrap(); + let engine = IoEngine::for_test(Arc::new(ShortThenErrorIo::default()), 2).unwrap(); let managed_memory = managed_memory(); let read = engine @@ -344,7 +282,7 @@ fn posix_engine_reports_progress_before_a_terminal_short_io_error() { async fn async_request_is_woken_by_driver_completion() { let file = TestFile::new("io-engine"); file.open().set_len(4096).unwrap(); - let engine = Arc::new(IoEngine::for_test(file.backend(), 2).unwrap()); + let engine = Arc::new(IoEngine::for_test(file.io(), 2).unwrap()); let managed_memory = managed_memory(); let request = submit_cache_io( engine.as_ref(), @@ -364,8 +302,8 @@ async fn async_request_is_woken_by_driver_completion() { #[tokio::test] async fn dropping_async_wait_requests_bounded_cancellation() { - let backend = Arc::new(BlockingBackend::default()); - let engine = Arc::new(IoEngine::for_test(backend.clone(), 1).unwrap()); + let io = Arc::new(BlockingIo::default()); + let engine = Arc::new(IoEngine::for_test(io.clone(), 1).unwrap()); let recorder = Arc::new( Recorder::new(StatsOptions { io_latency: true, @@ -387,7 +325,7 @@ async fn dropping_async_wait_requests_bounded_cancellation() { .await }); tokio::task::yield_now().await; - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); waiter.abort(); assert!(waiter.await.unwrap_err().is_cancelled()); @@ -397,7 +335,7 @@ async fn dropping_async_wait_requests_bounded_cancellation() { .iter() .all(|row| row.latency.count == 0) ); - backend.release(); + io.release(); engine.shutdown().unwrap(); assert_eq!(engine.in_flight(), 0); assert_eq!( @@ -416,7 +354,7 @@ async fn reserved_read_latency_includes_time_before_submission() { let file = TestFile::new("io-engine"); file.open().set_len(4096).unwrap(); let engine = - IoEngine::for_test_with_options(file.backend(), 1, 1, activity_counters_enabled, true) + IoEngine::for_test_with_options(file.io(), 1, 1, activity_counters_enabled, true) .unwrap(); let recorder = Arc::new( Recorder::new(StatsOptions { @@ -464,8 +402,8 @@ async fn reserved_read_latency_includes_time_before_submission() { #[tokio::test] async fn read_slot_waits_for_cancelled_request_to_release_physical_capacity() { - let backend = Arc::new(BlockingBackend::default()); - let engine = Arc::new(IoEngine::for_test_with_read_wait(backend.clone(), 1).unwrap()); + let io = Arc::new(BlockingIo::default()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(io.clone(), 1).unwrap()); let managed_memory = managed_memory(); let slot = engine.try_reserve_read().unwrap(); let request = submit_cache_read( @@ -481,7 +419,7 @@ async fn read_slot_waits_for_cancelled_request_to_release_physical_capacity() { .await }); tokio::task::yield_now().await; - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); request_waiter.abort(); assert!(request_waiter.await.unwrap_err().is_cancelled()); @@ -498,7 +436,7 @@ async fn read_slot_waits_for_cancelled_request_to_release_physical_capacity() { "caller cancellation must not publish physical capacity" ); - backend.release(); + io.release(); let slot = reservation.await.unwrap(); drop(slot); engine.shutdown().unwrap(); @@ -508,7 +446,7 @@ async fn read_slot_waits_for_cancelled_request_to_release_physical_capacity() { #[tokio::test] async fn read_slot_wait_is_woken_by_engine_shutdown() { let file = TestFile::new("io-engine"); - let engine = Arc::new(IoEngine::for_test_with_read_wait(file.backend(), 1).unwrap()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(file.io(), 1).unwrap()); let slot = engine.try_reserve_read().unwrap(); let mut waiters = Vec::new(); for expected in 1..=3 { @@ -531,7 +469,7 @@ async fn read_slot_wait_is_woken_by_engine_shutdown() { #[tokio::test(flavor = "current_thread")] async fn queued_read_reservation_precedes_new_immediate_read() { let file = TestFile::new("io-engine"); - let engine = Arc::new(IoEngine::for_test_with_read_wait(file.backend(), 1).unwrap()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(file.io(), 1).unwrap()); let held = engine.try_reserve_read().unwrap(); let queued = spawn_registered_read_slot_waiter(&engine, Duration::from_secs(1), 1).await; @@ -550,7 +488,7 @@ async fn queued_read_reservation_precedes_new_immediate_read() { #[tokio::test(flavor = "current_thread")] async fn queued_read_reservations_are_fifo() { let file = TestFile::new("io-engine"); - let engine = Arc::new(IoEngine::for_test_with_read_wait(file.backend(), 1).unwrap()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(file.io(), 1).unwrap()); let held = engine.try_reserve_read().unwrap(); let first = spawn_registered_read_slot_waiter(&engine, Duration::from_secs(1), 1).await; @@ -573,7 +511,7 @@ async fn queued_read_reservations_are_fifo() { #[tokio::test(flavor = "current_thread")] async fn queued_reads_use_every_released_engine_slot() { let file = TestFile::new("io-engine"); - let engine = Arc::new(IoEngine::for_test_with_read_wait(file.backend(), 2).unwrap()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(file.io(), 2).unwrap()); let held: Vec<_> = (0..2).map(|_| engine.try_reserve_read().unwrap()).collect(); let first = spawn_registered_read_slot_waiter(&engine, Duration::from_secs(1), 1).await; let second = spawn_registered_read_slot_waiter(&engine, Duration::from_secs(1), 2).await; @@ -592,7 +530,7 @@ async fn queued_reads_use_every_released_engine_slot() { #[tokio::test(flavor = "current_thread")] async fn timed_out_queue_head_passes_priority_to_next_read() { let file = TestFile::new("io-engine"); - let engine = Arc::new(IoEngine::for_test_with_read_wait(file.backend(), 1).unwrap()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(file.io(), 1).unwrap()); let held = engine.try_reserve_read().unwrap(); let first = spawn_registered_read_slot_waiter(&engine, Duration::from_millis(20), 1).await; @@ -610,7 +548,7 @@ async fn timed_out_queue_head_passes_priority_to_next_read() { #[tokio::test(flavor = "current_thread")] async fn cancelled_queue_head_passes_priority_to_next_read() { let file = TestFile::new("io-engine"); - let engine = Arc::new(IoEngine::for_test_with_read_wait(file.backend(), 1).unwrap()); + let engine = Arc::new(IoEngine::for_test_with_read_wait(file.io(), 1).unwrap()); let held = engine.try_reserve_read().unwrap(); let first = spawn_registered_read_slot_waiter(&engine, Duration::from_secs(1), 1).await; @@ -631,8 +569,8 @@ async fn cancelled_queue_head_passes_priority_to_next_read() { #[tokio::test] async fn async_read_deadline_keeps_other_slots_available() { - let backend = Arc::new(BlockingBackend::default()); - let engine = Arc::new(IoEngine::for_test(backend.clone(), 2).unwrap()); + let io = Arc::new(BlockingIo::default()); + let engine = Arc::new(IoEngine::for_test(io.clone(), 2).unwrap()); let managed_memory = managed_memory(); let request = submit_cache_io_until( engine.as_ref(), @@ -641,7 +579,7 @@ async fn async_read_deadline_keeps_other_slots_available() { Duration::from_millis(10), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let timeout = request .wait_async(Arc::clone(&engine), &tokio::runtime::Handle::current()) @@ -653,7 +591,7 @@ async fn async_read_deadline_keeps_other_slots_available() { assert_eq!(engine.in_flight(), 1); drop(engine.try_reserve_read().unwrap()); - backend.release(); + io.release(); engine.shutdown().unwrap(); } @@ -667,7 +605,7 @@ fn posix_engine_routes_only_aligned_record_io_to_direct() { buffered_file.set_len(8192).unwrap(); direct_file.set_len(8192).unwrap(); let engine = posix::start( - RuntimeFileSet::new(buffered_file, Some(direct_file)), + DataFileHandles::new(buffered_file, Some(direct_file)), 2, 2, true, @@ -702,19 +640,19 @@ fn posix_engine_routes_only_aligned_record_io_to_direct() { CompletionStatus::Completed )); - assert!(engine.stats().runtime.direct_active); + assert!(engine.stats().file_io.direct_active); let stats = engine.stats(); - assert_eq!(stats.runtime.write.direct.operations, 1); - assert_eq!(stats.runtime.write.direct.bytes, 4096); - assert_eq!(stats.runtime.write.buffered.operations, 1); - assert_eq!(stats.runtime.write.buffered.bytes, 32); + assert_eq!(stats.file_io.write.direct.operations, 1); + assert_eq!(stats.file_io.write.direct.bytes, 4096); + assert_eq!(stats.file_io.write.buffered.operations, 1); + assert_eq!(stats.file_io.write.buffered.bytes, 32); engine.shutdown().unwrap(); } #[test] fn unfenced_write_state_remains_unsafe_after_shutdown() { let file = TestFile::new("io-engine"); - let engine = IoEngine::for_test(file.backend(), 1).unwrap(); + let engine = IoEngine::for_test(file.io(), 1).unwrap(); assert!(!engine.has_unfenced_writes()); engine.mark_unfenced_writes_for_test(); @@ -726,8 +664,8 @@ fn unfenced_write_state_remains_unsafe_after_shutdown() { #[test] fn read_completion_deadline_retains_only_its_bounded_slot() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let deadline = Instant::now() + Duration::from_millis(20); let request = submit_cache_io_until( @@ -737,7 +675,7 @@ fn read_completion_deadline_retains_only_its_bounded_slot() { Duration::from_millis(10), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let timeout = request.wait(&engine).unwrap_err(); let pending = engine.in_flight(); @@ -753,15 +691,15 @@ fn read_completion_deadline_retains_only_its_bounded_slot() { .unwrap_err(); assert_eq!(rejected.error.kind(), io::ErrorKind::WouldBlock); - backend.release(); + io.release(); engine.shutdown().unwrap(); assert_eq!(engine.in_flight(), 0); } #[test] fn completion_deadline_keeps_an_issued_write_counted_until_target_completion() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let request = submit_cache_io_until( &engine, @@ -774,7 +712,7 @@ fn completion_deadline_keeps_an_issued_write_counted_until_target_completion() { Duration::from_millis(10), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let timeout = request.wait(&engine).unwrap_err(); let pending = engine.in_flight(); @@ -787,7 +725,7 @@ fn completion_deadline_keeps_an_issued_write_counted_until_target_completion() { )) .unwrap_err(); assert_eq!(rejected.error.kind(), io::ErrorKind::BrokenPipe); - backend.release(); + io.release(); let (error, buffer) = timeout.into_buffer(); assert_eq!(error.kind(), io::ErrorKind::TimedOut); assert!(buffer.is_none()); @@ -803,7 +741,7 @@ fn completion_deadline_keeps_an_issued_write_counted_until_target_completion() { fn engine_request_capacity_is_hard_bounded() { let file = TestFile::new("io-engine"); assert!(matches!( - IoEngine::for_test(file.backend(), MAX_IO_REQUESTS_PER_ENGINE + 1), + IoEngine::for_test(file.io(), MAX_IO_REQUESTS_PER_ENGINE + 1), Err(error) if error.kind() == io::ErrorKind::InvalidInput )); } @@ -812,9 +750,9 @@ fn engine_request_capacity_is_hard_bounded() { #[test] fn configured_posix_engine_shares_its_worker_capacity() { let file = TestFile::new("io-engine"); - let files = RuntimeFileSet::new(file.open(), None); + let handles = DataFileHandles::new(file.open(), None); let engine = - build_file_engine(files, IoEngineConfig::Posix { workers: 4 }, false, false).unwrap(); + build_file_engine(handles, IoEngineConfig::Posix { workers: 4 }, false, false).unwrap(); let reserved: Vec<_> = (0..4).map(|_| engine.try_reserve_read().unwrap()).collect(); assert_eq!( @@ -829,7 +767,7 @@ fn configured_posix_engine_shares_its_worker_capacity() { #[test] fn disabled_io_statistics_skip_cumulative_engine_counters() { let file = TestFile::new("io-engine"); - let engine = IoEngine::for_test_with_options(file.backend(), 1, 1, false, false).unwrap(); + let engine = IoEngine::for_test_with_options(file.io(), 1, 1, false, false).unwrap(); let managed_memory = managed_memory(); let completion = engine .write_all_at( @@ -846,21 +784,21 @@ fn disabled_io_statistics_skip_cumulative_engine_counters() { #[test] fn slot_state_tracks_full_write_capacity() { - let shared = Arc::new(RuntimeShared::new(2, true, false)); - let first = shared.try_reserve_slot(true).unwrap(); - let second = shared.try_reserve_slot(true).unwrap(); - assert!(shared.try_reserve_slot(true).is_none()); - assert_eq!(shared.total_in_flight(), 2); - assert_eq!(shared.writes_in_flight(), 2); + let state = Arc::new(EngineState::new(2, true, false)); + let first = state.try_reserve_slot(true).unwrap(); + let second = state.try_reserve_slot(true).unwrap(); + assert!(state.try_reserve_slot(true).is_none()); + assert_eq!(state.total_in_flight(), 2); + assert_eq!(state.writes_in_flight(), 2); drop((first, second)); - assert_eq!(shared.total_in_flight(), 0); - assert_eq!(shared.writes_in_flight(), 0); + assert_eq!(state.total_in_flight(), 0); + assert_eq!(state.writes_in_flight(), 0); } #[test] fn unused_read_reservation_releases_its_engine_slot() { let file = TestFile::new("io-engine"); - let engine = IoEngine::for_test(file.backend(), 1).unwrap(); + let engine = IoEngine::for_test(file.io(), 1).unwrap(); let slot = engine.try_reserve_read().unwrap(); assert_eq!(engine.in_flight(), 1); assert_eq!( @@ -876,7 +814,7 @@ fn unused_read_reservation_releases_its_engine_slot() { #[test] fn nowait_submission_does_not_wait_for_the_shutdown_fence() { let file = TestFile::new("io-engine"); - let engine = IoEngine::for_test(file.backend(), 1).unwrap(); + let engine = IoEngine::for_test(file.io(), 1).unwrap(); let managed_memory = managed_memory(); let fence = engine .submit_state @@ -894,9 +832,9 @@ fn nowait_submission_does_not_wait_for_the_shutdown_fence() { } #[test] -fn backend_workers_execute_independent_reads_concurrently() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 2).unwrap(); +fn posix_workers_execute_independent_reads_concurrently() { + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 2).unwrap(); let managed_memory = managed_memory(); let first = engine .read_exact_at(read_buffer(&managed_memory, 1), 0) @@ -905,27 +843,27 @@ fn backend_workers_execute_independent_reads_concurrently() { .read_exact_at(read_buffer(&managed_memory, 1), 1) .unwrap(); - let both_entered = backend.wait_for_entered(2); - backend.release(); + let both_entered = io.wait_for_entered(2); + io.release(); assert!(matches!(first.wait().status, CompletionStatus::Completed)); assert!(matches!(second.wait().status, CompletionStatus::Completed)); assert!( both_entered, "both workers must enter before either is released" ); - assert_eq!(backend.maximum_active(), 2); + assert_eq!(io.maximum_active(), 2); engine.shutdown().unwrap(); } #[test] fn submit_wait_blocks_at_engine_capacity_and_resumes() { - let backend = Arc::new(BlockingBackend::default()); - let engine = Arc::new(IoEngine::for_test(backend.clone(), 1).unwrap()); + let io = Arc::new(BlockingIo::default()); + let engine = Arc::new(IoEngine::for_test(io.clone(), 1).unwrap()); let managed_memory = managed_memory(); let first = engine .read_exact_at(read_buffer(&managed_memory, 1), 0) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let waiting_engine = engine.clone(); let waiting_buffer = read_buffer(&managed_memory, 1); @@ -946,7 +884,7 @@ fn submit_wait_blocks_at_engine_capacity_and_resumes() { let early = receiver.recv_timeout(Duration::from_millis(30)); let was_blocked = matches!(&early, Err(mpsc::RecvTimeoutError::Timeout)); - backend.release(); + io.release(); assert!(matches!(first.wait().status, CompletionStatus::Completed)); let second = match early { Ok(result) => result, @@ -968,13 +906,13 @@ fn submit_wait_blocks_at_engine_capacity_and_resumes() { #[test] fn controlled_slot_wait_observes_cancel_wake_and_absolute_deadline() { - let backend = Arc::new(BlockingBackend::default()); - let engine = Arc::new(IoEngine::for_test(backend.clone(), 1).unwrap()); + let io = Arc::new(BlockingIo::default()); + let engine = Arc::new(IoEngine::for_test(io.clone(), 1).unwrap()); let managed_memory = managed_memory(); let first = engine .read_exact_at(read_buffer(&managed_memory, 1), 0) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let cancelled = Arc::new(AtomicBool::new(false)); let waiting_engine = engine.clone(); @@ -1018,14 +956,14 @@ fn controlled_slot_wait_observes_cancel_wake_and_absolute_deadline() { assert_eq!(timed_out.error.kind(), io::ErrorKind::TimedOut); drop(timed_out); - backend.release(); + io.release(); assert!(matches!(first.wait().status, CompletionStatus::Completed)); engine.shutdown().unwrap(); } #[test] -fn backend_panic_completes_the_request_and_worker_survives() { - let engine = IoEngine::for_test(Arc::new(PanicOnceBackend::new()), 1).unwrap(); +fn positioned_io_panic_completes_the_request_and_worker_survives() { + let engine = IoEngine::for_test(Arc::new(PanicOnceIo::new()), 1).unwrap(); let managed_memory = managed_memory(); let failed = engine .read_exact_at(read_buffer(&managed_memory, 1), 0) @@ -1057,11 +995,11 @@ fn backend_panic_completes_the_request_and_worker_survives() { #[test] fn quarantined_completion_does_not_return_a_potentially_live_buffer() { - let shared = Arc::new(RuntimeShared::new(1, true, false)); - let slot = shared.try_reserve_slot(false).unwrap(); + let state = Arc::new(EngineState::new(1, true, false)); + let slot = state.try_reserve_slot(false).unwrap(); let request_id = RequestId(1); let completion = Arc::new(CompletionState::new()); - shared.requests_submitted.fetch_add(1, Ordering::Relaxed); + state.requests_submitted.fetch_add(1, Ordering::Relaxed); let managed_memory = managed_memory(); let task = Task { @@ -1071,7 +1009,7 @@ fn quarantined_completion_does_not_return_a_potentially_live_buffer() { slot, submitted_at: Some(Instant::now()), }; - shared.finish_quarantined( + state.finish_quarantined( task, CompletionStatus::Failed(io::Error::other("uncertain kernel lifetime")), 0, @@ -1080,7 +1018,7 @@ fn quarantined_completion_does_not_return_a_potentially_live_buffer() { let completed = completion.wait(); assert!(matches!(completed.status, CompletionStatus::Failed(_))); assert!(completed.buffer.is_none()); - assert_eq!(shared.snapshot().requests_in_flight, 0); + assert_eq!(state.snapshot().requests_in_flight, 0); // The uncertain buffer remains charged instead of being reused while // the kernel may still own its address. @@ -1092,14 +1030,14 @@ fn quarantined_completion_does_not_return_a_potentially_live_buffer() { drop(completed); drop(completion); - drop(shared); + drop(state); assert_eq!(managed_memory.snapshot().current_bytes, 0); } #[test] fn io_histograms_include_failures_when_activity_counters_are_disabled() { let file = TestFile::new("io-engine"); - let engine = IoEngine::for_test_with_options(file.backend(), 1, 1, false, false).unwrap(); + let engine = IoEngine::for_test_with_options(file.io(), 1, 1, false, false).unwrap(); let recorder = Arc::new( Recorder::new(StatsOptions { io_latency: true, @@ -1127,8 +1065,8 @@ fn io_histograms_include_failures_when_activity_counters_are_disabled() { #[test] fn configured_background_read_deadline_expires_and_retains_owned_buffer() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 2).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 2).unwrap(); let managed_memory = managed_memory(); let request = submit_cache_io_with_timeout( &engine, @@ -1136,9 +1074,9 @@ fn configured_background_read_deadline_expires_and_retains_owned_buffer() { Duration::from_millis(20), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let (error, buffer) = request.wait(&engine).unwrap_err().into_buffer(); - backend.release(); + io.release(); engine.shutdown().unwrap(); assert_eq!(error.kind(), io::ErrorKind::TimedOut); assert!(buffer.is_none()); @@ -1146,8 +1084,8 @@ fn configured_background_read_deadline_expires_and_retains_owned_buffer() { #[test] fn background_read_can_complete_after_default_deadline() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let request = submit_cache_io_with_timeout( &engine, @@ -1155,10 +1093,10 @@ fn background_read_can_complete_after_default_deadline() { Duration::from_secs(30), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); let release = std::thread::spawn(move || { std::thread::sleep(CACHE_IO_COMPLETION_TIMEOUT + Duration::from_millis(200)); - backend.release(); + io.release(); }); let completion = request.wait(&engine); release.join().unwrap(); @@ -1169,8 +1107,8 @@ fn background_read_can_complete_after_default_deadline() { #[test] fn background_recovery_keeps_the_original_request_and_accepts_late_completion() { for write in [false, true] { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let operation = if write { IoOperation::write( @@ -1182,8 +1120,8 @@ fn background_recovery_keeps_the_original_request_and_accepts_late_completion() IoOperation::read(read_buffer(&managed_memory, 4096), 0) }; let mut request = submit_cache_io(&engine, operation).unwrap(); - assert!(backend.wait_for_entered(1)); - // Force the normal deadline to expire while the backend still owns I/O. + assert!(io.wait_for_entered(1)); + // Force the normal deadline to expire while the io still owns I/O. request.deadline = Instant::now(); let id = request.id(); std::thread::scope(|scope| { @@ -1199,7 +1137,7 @@ fn background_recovery_keeps_the_original_request_and_accepts_late_completion() let early = rx.recv_timeout(Duration::from_millis(30)); let in_flight = engine.in_flight(); let charged = managed_memory.snapshot().current_bytes; - backend.release(); + io.release(); assert!(matches!(early, Err(mpsc::RecvTimeoutError::Timeout))); assert_eq!(in_flight, 1); assert!(charged >= 4096); @@ -1208,7 +1146,7 @@ fn background_recovery_keeps_the_original_request_and_accepts_late_completion() assert_eq!(completion.bytes_transferred, 4096); assert!(completion.into_io_result().0.is_ok()); }); - assert_eq!(lock_unpoisoned(&backend.state).entered, 1); + assert_eq!(lock_unpoisoned(&io.state).entered, 1); // Recovery does not poison admission: the next request also completes. let next = engine .read_exact_at(read_buffer(&managed_memory, 4096), 0) @@ -1222,8 +1160,8 @@ fn background_recovery_keeps_the_original_request_and_accepts_late_completion() #[test] fn exhausted_background_recovery_still_fences_unfinished_writes() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let mut request = submit_cache_io( &engine, @@ -1234,7 +1172,7 @@ fn exhausted_background_recovery_still_fences_unfinished_writes() { ), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); request.deadline = Instant::now(); request.cancel_grace = Duration::from_millis(10); let result = request.wait_with_recovery( @@ -1247,7 +1185,7 @@ fn exhausted_background_recovery_still_fences_unfinished_writes() { write_buffer(&managed_memory, &[8; 4096]), 4096, )); - backend.release(); + io.release(); engine.shutdown().unwrap(); let (error, buffer) = result.unwrap_err().into_buffer(); assert_eq!(error.kind(), io::ErrorKind::TimedOut); @@ -1261,13 +1199,13 @@ fn exhausted_background_recovery_still_fences_unfinished_writes() { #[test] fn background_admission_recovers_without_duplicate_submission() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let first = engine .read_exact_at(read_buffer(&managed_memory, 4096), 0) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); std::thread::scope(|scope| { let (tx, rx) = mpsc::channel(); let engine = &engine; @@ -1285,7 +1223,7 @@ fn background_admission_recovers_without_duplicate_submission() { tx.send(result).unwrap(); }); let early = rx.recv_timeout(Duration::from_millis(50)); - backend.release(); + io.release(); assert!(matches!(early, Err(mpsc::RecvTimeoutError::Timeout))); assert!( rx.recv_timeout(Duration::from_secs(2)) @@ -1299,13 +1237,13 @@ fn background_admission_recovers_without_duplicate_submission() { }); assert!(first.wait().into_io_result().0.is_ok()); engine.shutdown().unwrap(); - assert_eq!(lock_unpoisoned(&backend.state).entered, 2); + assert_eq!(lock_unpoisoned(&io.state).entered, 2); } #[test] fn unlimited_recovery_keeps_admission_paused_until_validation() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let recovery = BackgroundRecovery::new(None); let mut request = submit_cache_io( @@ -1317,7 +1255,7 @@ fn unlimited_recovery_keeps_admission_paused_until_validation() { ), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); request.deadline = Instant::now(); std::thread::scope(|scope| { let (completed_tx, completed_rx) = mpsc::channel(); @@ -1334,7 +1272,7 @@ fn unlimited_recovery_keeps_admission_paused_until_validation() { // Cross more than one polling interval while preserving the same I/O. let early = completed_rx.recv_timeout(Duration::from_millis(1100)); let paused = recovery.is_recovering(); - backend.release(); + io.release(); assert!(matches!(early, Err(mpsc::RecvTimeoutError::Timeout))); assert!(paused); let completion = completed_rx.recv_timeout(Duration::from_secs(2)).unwrap(); @@ -1346,14 +1284,14 @@ fn unlimited_recovery_keeps_admission_paused_until_validation() { validate_tx.send(()).unwrap(); }); assert!(!recovery.is_recovering()); - assert_eq!(lock_unpoisoned(&backend.state).entered, 1); + assert_eq!(lock_unpoisoned(&io.state).entered, 1); engine.shutdown().unwrap(); } #[test] fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { - let backend = Arc::new(BlockingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(BlockingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); let recovery = BackgroundRecovery::new(None); let mut request = submit_cache_io( @@ -1365,7 +1303,7 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { ), ) .unwrap(); - assert!(backend.wait_for_entered(1)); + assert!(io.wait_for_entered(1)); request.deadline = Instant::now(); std::thread::scope(|scope| { let (tx, rx) = mpsc::channel(); @@ -1381,7 +1319,7 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { let stopped = rx.recv_timeout(Duration::from_secs(2)); let pending = engine.writes_in_flight(); let charged = managed_memory.snapshot().current_bytes; - backend.release(); + io.release(); assert!(matches!(early, Err(mpsc::RecvTimeoutError::Timeout))); let (error, buffer) = stopped.unwrap().unwrap_err().into_buffer(); assert_eq!(error.kind(), io::ErrorKind::TimedOut); diff --git a/cache2/src/io/engine/uring.rs b/cache2/src/io/engine/uring.rs index eaebcc0..8364356 100644 --- a/cache2/src/io/engine/uring.rs +++ b/cache2/src/io/engine/uring.rs @@ -39,14 +39,12 @@ use io_uring::squeue; use io_uring::types; use crate::config::runtime::IoUringEngineConfig; -use crate::io::backend::RuntimeFileSet; -use crate::io::backend::RuntimeIoPath; -use crate::io::backend::retry_interrupted; #[cfg(test)] use crate::io::engine::CompletionState; use crate::io::engine::CompletionStatus; use crate::io::engine::DriverCommand; use crate::io::engine::DriverWake; +use crate::io::engine::EngineState; #[cfg(test)] use crate::io::engine::IO_QUEUE_ENTRY_RESERVATION_BYTES; #[cfg(test)] @@ -57,11 +55,13 @@ use crate::io::engine::IoOperation; use crate::io::engine::MAX_IO_REQUESTS_PER_ENGINE; use crate::io::engine::OperationKind; use crate::io::engine::RequestId; -use crate::io::engine::RuntimeShared; use crate::io::engine::SubmitState; use crate::io::engine::Task; #[cfg(test)] use crate::io::engine::io_uring_extra_memory_bytes; +use crate::io::file::DataFileHandles; +use crate::io::file::FileIoPath; +use crate::io::file::retry_interrupted; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; const CANCEL_CQE_BIT: u64 = 1_u64 << 63; @@ -108,7 +108,7 @@ impl DriverWake for SocketWake { /// Starts a driver that owns the ring and all submitted buffer pointers. pub fn start( - files: RuntimeFileSet, + handles: DataFileHandles, config: IoUringEngineConfig, activity_counters_enabled: bool, read_wait_enabled: bool, @@ -118,7 +118,7 @@ pub fn start( max_in_flight, activity_counters_enabled, read_wait_enabled, - files.stats_handle(), + handles.stats_handle(), )?; let ring_entries = max_in_flight .checked_add(2) @@ -153,7 +153,7 @@ pub fn start( if config.sq_poll.is_some() && !ring.params().is_feature_sqpoll_nonfixed() { return Err(io::Error::new( io::ErrorKind::Unsupported, - "kernel io_uring SQPOLL requires registered files", + "kernel io_uring SQPOLL requires registered handles", )); } let mut probe = Probe::new(); @@ -180,18 +180,18 @@ pub fn start( pending: Arc::clone(&wake_pending), }); engine.wake = Some(wake); - let worker_shared = Arc::clone(&engine.shared); + let worker_state = Arc::clone(&engine.state); let worker_submit_state = Arc::clone(&engine.submit_state); let worker = std::thread::Builder::new() .name("cache2-uring-io".into()) .stack_size(CACHE_THREAD_STACK_BYTES) .spawn(move || { uring_driver( - files, + handles, ring, wake_receiver, wake_pending, - worker_shared, + worker_state, worker_submit_state, receiver, ) @@ -204,7 +204,7 @@ struct Flight { task: Task, transferred: usize, active: bool, - active_path: Option, + active_path: Option, cancel_submitted: bool, } @@ -217,14 +217,14 @@ enum PendingEntry { } struct UringDriver { - files: Option, + handles: Option, ring: Option, wake_receiver: UnixStream, wake_pending: Arc, wake_active: bool, wake_cancel_submitted: bool, wake_cancel_completed: bool, - shared: Arc, + state: Arc, submit_state: Arc>, receiver: Receiver, flights: HashMap, @@ -239,27 +239,27 @@ struct UringDriver { } fn uring_driver( - files: RuntimeFileSet, + handles: DataFileHandles, mut ring: IoUring, wake_receiver: UnixStream, wake_pending: Arc, - shared: Arc, + state: Arc, submit_state: Arc>, receiver: Receiver, ) -> io::Result<()> { - let max_in_flight = shared.max_in_flight; + let max_in_flight = state.max_in_flight; let submission_capacity = ring.submission().capacity(); let completion_capacity = ring.completion().capacity(); let io_poll = ring.params().is_setup_iopoll(); let mut driver = UringDriver { - files: Some(files), + handles: Some(handles), ring: Some(ring), wake_receiver, wake_pending, wake_active: false, wake_cancel_submitted: false, wake_cancel_completed: false, - shared, + state, submit_state, receiver, flights: HashMap::with_capacity_and_hasher(max_in_flight, Xxh3_64Builder::default()), @@ -333,9 +333,9 @@ impl UringDriver { match self.receiver.try_recv() { Ok(DriverCommand::Submit(task)) => { if task.completion.cancel_requested.load(Ordering::Acquire) { - self.shared.finish(task, CompletionStatus::Cancelled, 0); + self.state.finish(task, CompletionStatus::Cancelled, 0); } else if task.operation_is_empty() { - self.shared.finish(task, CompletionStatus::Completed, 0); + self.state.finish(task, CompletionStatus::Completed, 0); } else { let request_id = task.request_id; self.flights.insert( @@ -366,7 +366,7 @@ impl UringDriver { } fn queue_requested_cancels(&mut self) { - if !self.shared.cancel_scan_needed.swap(false, Ordering::AcqRel) { + if !self.state.cancel_scan_needed.swap(false, Ordering::AcqRel) { return; } self.requested_cancels.clear(); @@ -443,7 +443,7 @@ impl UringDriver { .remove(&request_id) .expect("checked flight exists"); let status = cancelled_before_resubmit_status(&flight); - self.shared.finish(flight.task, status, flight.transferred); + self.state.finish(flight.task, status, flight.transferred); continue; } self.pending_entries.push(PendingEntry::Target(request_id)); @@ -483,23 +483,23 @@ impl UringDriver { .flights .get(&request_id) .expect("pending target has a flight"); - flight.task.operation.runtime_io_path( - self.files + flight.task.operation.file_io_path( + self.handles .as_ref() - .expect("runtime files exist while driver runs"), + .expect("runtime handles exist while driver runs"), flight.transferred, )? }; - if self.io_poll && path != RuntimeIoPath::Direct { + if self.io_poll && path != FileIoPath::Direct { return Err(io::Error::new( io::ErrorKind::InvalidInput, "io_uring IOPOLL operation is not direct-I/O aligned", )); } let file_fd = self - .files + .handles .as_ref() - .expect("runtime files exist while driver runs") + .expect("runtime handles exist while driver runs") .file_for(path) .as_raw_fd(); let flight = self @@ -665,7 +665,7 @@ impl UringDriver { return; }; flight.active = false; - let active_path = flight.active_path.take().unwrap_or(RuntimeIoPath::Buffered); + let active_path = flight.active_path.take().unwrap_or(FileIoPath::Buffered); if result < 0 { let raw_error = result.saturating_neg(); let error = io::Error::from_raw_os_error(raw_error); @@ -683,7 +683,7 @@ impl UringDriver { } else { CompletionStatus::Failed(error) }; - self.shared.finish(flight.task, status, flight.transferred); + self.state.finish(flight.task, status, flight.transferred); return; } @@ -693,9 +693,9 @@ impl UringDriver { operation_length(&flight.task.operation).saturating_sub(flight.transferred); let completed = result as usize; if completed <= remaining && completed != 0 { - self.files + self.handles .as_ref() - .expect("runtime files exist while driver runs") + .expect("runtime handles exist while driver runs") .record( flight.task.operation.kind().io_direction(), active_path, @@ -708,13 +708,13 @@ impl UringDriver { } else { io::ErrorKind::WriteZero }; - self.shared.finish( + self.state.finish( flight.task, CompletionStatus::Failed(io::Error::new(kind, "short io_uring I/O")), flight.transferred, ); } else if completed > remaining { - self.shared.finish( + self.state.finish( flight.task, CompletionStatus::Failed(io::Error::new( io::ErrorKind::InvalidData, @@ -725,7 +725,7 @@ impl UringDriver { } else { flight.transferred += completed; if flight.transferred == operation_length(&flight.task.operation) { - self.shared.finish( + self.state.finish( flight.task, CompletionStatus::Completed, flight.transferred, @@ -737,7 +737,7 @@ impl UringDriver { .load(Ordering::Acquire) { let status = cancelled_before_resubmit_status(&flight); - self.shared.finish(flight.task, status, flight.transferred); + self.state.finish(flight.task, status, flight.transferred); } else { self.flights.insert(request_id, flight); self.pending_targets.push_back(request_id); @@ -748,7 +748,7 @@ impl UringDriver { } fn stop_accepting_and_fail_all(&mut self, error: &io::Error) { - // Submission holds a shared fence through the bounded channel send. + // Submission holds a state fence through the bounded channel send. // Once this exclusive guard is acquired, no task can appear after // the drain. let mut submit_state = self @@ -756,7 +756,7 @@ impl UringDriver { .write() .unwrap_or_else(|poisoned| poisoned.into_inner()); submit_state.accepting = false; - self.shared.stop_accepting_slots(); + self.state.stop_accepting_slots(); // The false state is permanent. Release the fence before waking // completion consumers so a custom waker may safely re-enter the // engine and receive BrokenPipe instead of deadlocking. @@ -778,24 +778,24 @@ impl UringDriver { // duplicated open-file description (and therefore its flock) // for process lifetime. The cache must also observe the flag // and never issue LOCK_UN on another duplicate. - self.shared.mark_unfenced_writes(); - if let Some(files) = self.files.take() { - mem::forget(files); + self.state.mark_unfenced_writes(); + if let Some(handles) = self.handles.take() { + mem::forget(handles); } } for (_, flight) in self.flights.drain() { let status = CompletionStatus::Failed(copy_io_error(error, &message)); if flight.active { - self.shared + self.state .finish_quarantined(flight.task, status, flight.transferred); } else { - self.shared.finish(flight.task, status, flight.transferred); + self.state.finish(flight.task, status, flight.transferred); } } while let Ok(command) = self.receiver.try_recv() { if let DriverCommand::Submit(task) = command { - self.shared.finish( + self.state.finish( task, CompletionStatus::Failed(copy_io_error(error, &message)), 0, @@ -887,7 +887,7 @@ impl UringDriver { } else if user_data & INTERNAL_CQE_BIT == 0 && let Some(flight) = self.flights.remove(&RequestId(user_data)) { - self.shared.finish( + self.state.finish( flight.task, CompletionStatus::Failed(copy_io_error(error, message)), flight.transferred, @@ -978,7 +978,7 @@ mod tests { use crate::managed_memory::ManagedMemoryLimits; struct CancelledCommandProducer { - shared: Arc, + state: Arc, commands: mpsc::SyncSender, managed_memory: ManagedMemory, next: AtomicU64, @@ -1000,7 +1000,7 @@ mod tests { request_id: RequestId(self.next.fetch_add(1, Ordering::Relaxed)), operation: IoOperation::read(buffer, 0), completion, - slot: self.shared.try_reserve_slot(false).unwrap(), + slot: self.state.try_reserve_slot(false).unwrap(), submitted_at: None, }; assert!(self.commands.try_send(DriverCommand::Submit(task)).is_ok()); @@ -1018,11 +1018,11 @@ mod tests { #[test] fn cancelled_command_refills_leave_room_for_io_progress() { let depth = 2; - let shared = Arc::new(RuntimeShared::new(depth, false, false)); + let state = Arc::new(EngineState::new(depth, false, false)); let (commands, receiver) = mpsc::sync_channel(depth * 2 + 1); let (_wake_sender, wake_receiver) = UnixStream::pair().unwrap(); let producer = Arc::new(CancelledCommandProducer { - shared: Arc::clone(&shared), + state: Arc::clone(&state), commands, managed_memory: ManagedMemory::try_new(ManagedMemoryLimits { memory_limit_bytes: 16 * 1024, @@ -1034,14 +1034,14 @@ mod tests { }); // Command processing needs neither a kernel ring nor file descriptors. let mut driver = UringDriver { - files: None, + handles: None, ring: None, wake_receiver, wake_pending: Arc::new(AtomicBool::new(false)), wake_active: false, wake_cancel_submitted: false, wake_cancel_completed: false, - shared, + state, submit_state: Arc::new(RwLock::new(SubmitState { accepting: true })), receiver, flights: HashMap::with_capacity_and_hasher(depth, Xxh3_64Builder::default()), diff --git a/cache2/src/io/backend.rs b/cache2/src/io/file.rs similarity index 76% rename from cache2/src/io/backend.rs rename to cache2/src/io/file.rs index ec0642e..3370631 100644 --- a/cache2/src/io/backend.rs +++ b/cache2/src/io/file.rs @@ -12,8 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Synchronous positioned I/O abstraction used for open, recovery, and the -//! reference runtime path. +//! Positioned reads and writes, owned cache files, and durability operations. //! //! Persistence points are carried through the trait so tests can fail an exact //! record, superblock, or barrier operation without changing the cache @@ -48,53 +47,53 @@ const LINUX_EINTR: i32 = 4; const SAFE_CACHE_OPEN_FLAGS: i32 = libc::O_NOFOLLOW | libc::O_NONBLOCK; #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum RuntimeIoPath { +pub enum FileIoPath { Buffered, Direct, } #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum RuntimeIoDirection { +pub enum FileIoDirection { Read, Write, } #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] -pub struct RuntimeIoDirectionStats { +pub struct FileIoDirectionStats { pub buffered: CacheIoPathSnapshot, pub direct: CacheIoPathSnapshot, } #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] -pub struct RuntimeIoStats { +pub struct FileIoStats { pub direct_active: bool, - pub read: RuntimeIoDirectionStats, - pub write: RuntimeIoDirectionStats, + pub read: FileIoDirectionStats, + pub write: FileIoDirectionStats, } #[derive(Clone)] -pub struct RuntimeIoStatsHandle { - inner: Arc, +pub struct FileIoStatsHandle { + inner: Arc, } -struct RuntimeIoCounters { +struct FileIoCounters { direct_active: bool, activity_counters_enabled: AtomicBool, - read: RuntimeIoDirectionCounters, - write: RuntimeIoDirectionCounters, + read: FileIoDirectionCounters, + write: FileIoDirectionCounters, } -struct RuntimeIoDirectionCounters { - buffered: RuntimeIoPathCounters, - direct: RuntimeIoPathCounters, +struct FileIoDirectionCounters { + buffered: FileIoPathCounters, + direct: FileIoPathCounters, } -struct RuntimeIoPathCounters { +struct FileIoPathCounters { operations: AtomicU64, bytes: AtomicU64, } -impl RuntimeIoPathCounters { +impl FileIoPathCounters { fn new() -> Self { Self { operations: AtomicU64::new(0), @@ -110,46 +109,46 @@ impl RuntimeIoPathCounters { } } -impl RuntimeIoDirectionCounters { +impl FileIoDirectionCounters { fn new() -> Self { Self { - buffered: RuntimeIoPathCounters::new(), - direct: RuntimeIoPathCounters::new(), + buffered: FileIoPathCounters::new(), + direct: FileIoPathCounters::new(), } } - fn snapshot(&self) -> RuntimeIoDirectionStats { - RuntimeIoDirectionStats { + fn snapshot(&self) -> FileIoDirectionStats { + FileIoDirectionStats { buffered: self.buffered.snapshot(), direct: self.direct.snapshot(), } } } -impl RuntimeIoStatsHandle { +impl FileIoStatsHandle { pub fn new(direct_active: bool) -> Self { Self { - inner: Arc::new(RuntimeIoCounters { + inner: Arc::new(FileIoCounters { direct_active, activity_counters_enabled: AtomicBool::new(true), - read: RuntimeIoDirectionCounters::new(), - write: RuntimeIoDirectionCounters::new(), + read: FileIoDirectionCounters::new(), + write: FileIoDirectionCounters::new(), }), } } - fn record(&self, direction: RuntimeIoDirection, path: RuntimeIoPath, length: usize) { + fn record(&self, direction: FileIoDirection, path: FileIoPath, length: usize) { if !self.inner.activity_counters_enabled.load(Ordering::Relaxed) { return; } let bytes = u64::try_from(length).unwrap_or(u64::MAX); let direction = match direction { - RuntimeIoDirection::Read => &self.inner.read, - RuntimeIoDirection::Write => &self.inner.write, + FileIoDirection::Read => &self.inner.read, + FileIoDirection::Write => &self.inner.write, }; let path = match path { - RuntimeIoPath::Buffered => &direction.buffered, - RuntimeIoPath::Direct => &direction.direct, + FileIoPath::Buffered => &direction.buffered, + FileIoPath::Direct => &direction.direct, }; path.operations.fetch_add(1, Ordering::Relaxed); path.bytes.fetch_add(bytes, Ordering::Relaxed); @@ -161,8 +160,8 @@ impl RuntimeIoStatsHandle { .store(enabled, Ordering::Relaxed); } - pub fn snapshot(&self) -> RuntimeIoStats { - RuntimeIoStats { + pub fn snapshot(&self) -> FileIoStats { + FileIoStats { direct_active: self.inner.direct_active, read: self.inner.read.snapshot(), write: self.inner.write.snapshot(), @@ -174,19 +173,19 @@ impl RuntimeIoStatsHandle { /// the descriptor that owns flock, so retaining this set also retains the /// cache lock if an issued write or flush cannot be fenced. `direct`, when present, /// is a separate O_DIRECT open used only for aligned runtime data requests. -pub struct RuntimeFileSet { +pub struct DataFileHandles { buffered: File, direct: Option, - stats: RuntimeIoStatsHandle, + stats: FileIoStatsHandle, } -impl RuntimeFileSet { +impl DataFileHandles { #[cfg(test)] pub fn buffered(file: File) -> Self { Self { buffered: file, direct: None, - stats: RuntimeIoStatsHandle::new(false), + stats: FileIoStatsHandle::new(false), } } @@ -200,7 +199,7 @@ impl RuntimeFileSet { Self { buffered, direct, - stats: RuntimeIoStatsHandle::new(direct_active), + stats: FileIoStatsHandle::new(direct_active), } } @@ -210,26 +209,26 @@ impl RuntimeFileSet { length: usize, offset: u64, allow_direct: bool, - ) -> RuntimeIoPath { + ) -> FileIoPath { if !allow_direct || self.direct.is_none() { - return RuntimeIoPath::Buffered; + return FileIoPath::Buffered; } // Never issue malformed O_DIRECT. Unaligned record fragments and an // unaligned remainder after a positive short completion use the // buffered compatibility path. Direct mode requires the direct // descriptor but does not make 32-byte-aligned records unreadable. if direct_io_aligned(buffer, length, offset) { - RuntimeIoPath::Direct + FileIoPath::Direct } else { - RuntimeIoPath::Buffered + FileIoPath::Buffered } } - pub fn record(&self, direction: RuntimeIoDirection, path: RuntimeIoPath, length: usize) { + pub fn record(&self, direction: FileIoDirection, path: FileIoPath, length: usize) { self.stats.record(direction, path, length); } - pub fn stats_handle(&self) -> RuntimeIoStatsHandle { + pub fn stats_handle(&self) -> FileIoStatsHandle { self.stats.clone() } @@ -242,10 +241,10 @@ impl RuntimeFileSet { } #[cfg(unix)] - pub fn file_for(&self, path: RuntimeIoPath) -> &File { + pub fn file_for(&self, path: FileIoPath) -> &File { match path { - RuntimeIoPath::Buffered => &self.buffered, - RuntimeIoPath::Direct => self + FileIoPath::Buffered => &self.buffered, + FileIoPath::Direct => self .direct .as_ref() .expect("direct path requires a direct descriptor"), @@ -288,12 +287,8 @@ pub enum SyncMode { All, } -pub trait IoBackend: Send + Sync { - fn len(&self) -> io::Result; - fn set_len(&self, len: u64) -> io::Result<()>; - fn preallocate(&self, len: u64) -> io::Result<()> { - self.set_len(len) - } +/// Synchronous reads and writes at explicit byte offsets. +pub trait PositionedIo: Send + Sync { fn read_at(&self, buffer: &mut [u8], offset: u64) -> io::Result; /// Reads into storage that may not yet be initialized. /// @@ -308,9 +303,8 @@ pub trait IoBackend: Send + Sync { length: usize, offset: u64, ) -> io::Result { - // The default keeps fault-injecting and test backends source-compatible. - // Production runtime files override this method with positioned kernel - // I/O that can initialize the destination directly. + // Slice-based implementations require initialized storage. Data-file + // handles override this method to let the kernel initialize it directly. // SAFETY: upheld by the caller; zeroing establishes initialized bytes // before constructing the mutable slice required by `read_at`. unsafe { @@ -319,24 +313,30 @@ pub trait IoBackend: Send + Sync { } } fn write_at(&self, point: WritePoint, buffer: &[u8], offset: u64) -> io::Result; +} + +/// File ownership, extent management, and durability for cache persistence. +/// +/// Recovery reads through [`PositionedIo`], synchronizes through this interface, +/// and clones the same validated descriptor for an immutable private mapping. File identity is +/// intentionally descriptor-based so callers never need to reopen a path between validation and +/// `mmap`. +pub trait StorageFile: PositionedIo { + fn len(&self) -> io::Result; + fn set_len(&self, len: u64) -> io::Result<()>; + fn preallocate(&self, len: u64) -> io::Result<()> { + self.set_len(len) + } fn sync(&self, point: SyncPoint, mode: SyncMode) -> io::Result<()>; fn try_lock_exclusive(&self) -> io::Result<()>; fn unlock(&self) -> io::Result<()>; -} -/// Buffered descriptor access needed by recovery-control code. -/// -/// Recovery uses the [`IoBackend`] methods for injectable positioned I/O and -/// durability barriers, then clones this exact validated descriptor for an -/// immutable private mapping. File identity is intentionally descriptor-based -/// so callers never need to reopen a path between validation and `mmap`. -pub trait ControlIoBackend: IoBackend { - fn try_clone_control_file(&self) -> io::Result; + fn try_clone_mapping_file(&self) -> io::Result; - fn control_file_identity(&self) -> io::Result; + fn identity(&self) -> io::Result; - fn is_same_file(&self, other: &dyn ControlIoBackend) -> io::Result { - Ok(self.control_file_identity()? == other.control_file_identity()?) + fn is_same_file(&self, other: &dyn StorageFile) -> io::Result { + Ok(self.identity()? == other.identity()?) } } @@ -345,19 +345,19 @@ pub trait ControlIoBackend: IoBackend { /// The fields remain opaque: recovery code only needs equality to reject /// aliased data, state, and image descriptors. #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct ControlFileIdentity { +pub struct FileIdentity { device: u64, inode: u64, } -pub struct FileBackend { +pub struct CacheFile { /// Buffered control descriptor and flock owner. file: File, /// Separate Linux O_DIRECT descriptor for aligned runtime data I/O. direct: Option, } -impl FileBackend { +impl CacheFile { #[cfg(test)] pub fn open(path: &Path) -> io::Result { Self::open_with_io_mode(path, IoMode::Buffered) @@ -424,10 +424,10 @@ impl FileBackend { } } - pub fn try_clone_runtime_files(&self) -> io::Result { + pub fn try_clone_data_handles(&self) -> io::Result { let buffered = self.file.try_clone()?; let direct = self.direct.as_ref().map(File::try_clone).transpose()?; - Ok(RuntimeFileSet::with_direct(buffered, direct)) + Ok(DataFileHandles::with_direct(buffered, direct)) } } @@ -456,35 +456,7 @@ fn preallocate_macos(file: &File, len: i64) -> io::Result<()> { } #[cfg(unix)] -impl ControlIoBackend for FileBackend { - fn try_clone_control_file(&self) -> io::Result { - self.file.try_clone() - } - - fn control_file_identity(&self) -> io::Result { - let metadata = self.file.metadata()?; - Ok(ControlFileIdentity { - device: metadata.dev(), - inode: metadata.ino(), - }) - } -} - -/// Positioned-I/O backend used by the POSIX engine. Control, -/// metadata, locking, and recovery continue to use `FileBackend`; this backend -/// routes only aligned runtime reads and record writes to the direct fd. -pub struct RuntimeFileBackend { - files: RuntimeFileSet, -} - -impl RuntimeFileBackend { - pub fn new(files: RuntimeFileSet) -> Self { - Self { files } - } -} - -#[cfg(unix)] -impl IoBackend for FileBackend { +impl StorageFile for CacheFile { fn len(&self) -> io::Result { Ok(self.file.metadata()?.len()) } @@ -549,14 +521,6 @@ impl IoBackend for FileBackend { } } - fn read_at(&self, buffer: &mut [u8], offset: u64) -> io::Result { - self.file.read_at(buffer, offset) - } - - fn write_at(&self, _point: WritePoint, buffer: &[u8], offset: u64) -> io::Result { - self.file.write_at(buffer, offset) - } - fn sync(&self, _point: SyncPoint, mode: SyncMode) -> io::Result<()> { match mode { SyncMode::Data => self.file.sync_data(), @@ -586,31 +550,40 @@ impl IoBackend for FileBackend { Err(io::Error::last_os_error()) } } + + fn try_clone_mapping_file(&self) -> io::Result { + self.file.try_clone() + } + + fn identity(&self) -> io::Result { + let metadata = self.file.metadata()?; + Ok(FileIdentity { + device: metadata.dev(), + inode: metadata.ino(), + }) + } } #[cfg(unix)] -impl IoBackend for RuntimeFileBackend { - fn len(&self) -> io::Result { - Ok(self - .files - .file_for(RuntimeIoPath::Buffered) - .metadata()? - .len()) +impl PositionedIo for CacheFile { + fn read_at(&self, buffer: &mut [u8], offset: u64) -> io::Result { + self.file.read_at(buffer, offset) } - fn set_len(&self, len: u64) -> io::Result<()> { - self.files.file_for(RuntimeIoPath::Buffered).set_len(len) + fn write_at(&self, _point: WritePoint, buffer: &[u8], offset: u64) -> io::Result { + self.file.write_at(buffer, offset) } +} +#[cfg(unix)] +impl PositionedIo for DataFileHandles { fn read_at(&self, buffer: &mut [u8], offset: u64) -> io::Result { - let path = self - .files - .select_path(buffer.as_ptr(), buffer.len(), offset, true); - let result = self.files.file_for(path).read_at(buffer, offset); + let path = self.select_path(buffer.as_ptr(), buffer.len(), offset, true); + let result = self.file_for(path).read_at(buffer, offset); if let Ok(bytes) = result && bytes != 0 { - self.files.record(RuntimeIoDirection::Read, path, bytes); + self.record(FileIoDirection::Read, path, bytes); } result } @@ -621,58 +594,33 @@ impl IoBackend for RuntimeFileBackend { length: usize, offset: u64, ) -> io::Result { - let path = self - .files - .select_path(buffer.cast_const(), length, offset, true); + let path = self.select_path(buffer.cast_const(), length, offset, true); // SAFETY: the caller supplies a writable destination for `length` // bytes; the selected descriptor is held by `self` for this call. - let result = - unsafe { read_file_at_uninit(self.files.file_for(path), buffer, length, offset) }; + let result = unsafe { read_file_at_uninit(self.file_for(path), buffer, length, offset) }; if let Ok(bytes) = result && bytes != 0 { - self.files.record(RuntimeIoDirection::Read, path, bytes); + self.record(FileIoDirection::Read, path, bytes); } result } fn write_at(&self, point: WritePoint, buffer: &[u8], offset: u64) -> io::Result { - let path = self.files.select_path( + let path = self.select_path( buffer.as_ptr(), buffer.len(), offset, point == WritePoint::Record, ); - let result = self.files.file_for(path).write_at(buffer, offset); + let result = self.file_for(path).write_at(buffer, offset); if let Ok(bytes) = result && bytes != 0 { - self.files.record(RuntimeIoDirection::Write, path, bytes); + self.record(FileIoDirection::Write, path, bytes); } result } - - fn sync(&self, _point: SyncPoint, mode: SyncMode) -> io::Result<()> { - let file = self.files.file_for(RuntimeIoPath::Buffered); - match mode { - SyncMode::Data => file.sync_data(), - SyncMode::All => file.sync_all(), - } - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Err(io::Error::new( - io::ErrorKind::Unsupported, - "runtime file backend does not own cache locking", - )) - } - - fn unlock(&self) -> io::Result<()> { - Err(io::Error::new( - io::ErrorKind::Unsupported, - "runtime file backend does not own cache locking", - )) - } } #[cfg(target_os = "linux")] @@ -696,26 +644,22 @@ fn open_direct(path: &Path, buffered: &File) -> io::Result { Ok(direct) } -pub fn read_exact_at(backend: &dyn IoBackend, buffer: &mut [u8], offset: u64) -> io::Result<()> { - read_exact_at_with_progress(backend, buffer, offset).0 +pub fn read_exact_at(io: &dyn PositionedIo, buffer: &mut [u8], offset: u64) -> io::Result<()> { + read_exact_at_with_progress(io, buffer, offset).0 } -pub fn read_at_bounded( - backend: &dyn IoBackend, - buffer: &mut [u8], - offset: u64, -) -> io::Result { - retry_interrupted(|| backend.read_at(buffer, offset)) +pub fn read_at_bounded(io: &dyn PositionedIo, buffer: &mut [u8], offset: u64) -> io::Result { + retry_interrupted(|| io.read_at(buffer, offset)) } fn read_exact_at_with_progress( - backend: &dyn IoBackend, + io: &dyn PositionedIo, mut buffer: &mut [u8], mut offset: u64, ) -> (io::Result<()>, usize) { let mut transferred = 0_usize; while !buffer.is_empty() { - let read = match read_at_bounded(backend, buffer, offset) { + let read = match read_at_bounded(io, buffer, offset) { Err(error) => return (Err(error), transferred), Ok(read) => read, }; @@ -756,7 +700,7 @@ fn read_exact_at_with_progress( } pub fn read_exact_at_uninit_with_progress( - backend: &dyn IoBackend, + io: &dyn PositionedIo, buffer: *mut u8, length: usize, mut offset: u64, @@ -767,7 +711,7 @@ pub fn read_exact_at_uninit_with_progress( let read = match retry_interrupted(|| { // SAFETY: the caller owns a destination valid for `length` bytes, // and the unchanged suffix bounds remain valid across retries. - unsafe { backend.read_at_uninit(buffer.add(transferred), remaining, offset) } + unsafe { io.read_at_uninit(buffer.add(transferred), remaining, offset) } }) { Err(error) => return (Err(error), transferred), Ok(read) => read, @@ -830,23 +774,23 @@ unsafe fn read_file_at_uninit( } pub fn write_all_at( - backend: &dyn IoBackend, + io: &dyn PositionedIo, point: WritePoint, buffer: &[u8], offset: u64, ) -> io::Result<()> { - write_all_at_with_progress(backend, point, buffer, offset).0 + write_all_at_with_progress(io, point, buffer, offset).0 } pub fn write_all_at_with_progress( - backend: &dyn IoBackend, + io: &dyn PositionedIo, point: WritePoint, mut buffer: &[u8], mut offset: u64, ) -> (io::Result<()>, usize) { let mut transferred = 0_usize; while !buffer.is_empty() { - let written = match retry_interrupted(|| backend.write_at(point, buffer, offset)) { + let written = match retry_interrupted(|| io.write_at(point, buffer, offset)) { Err(error) => return (Err(error), transferred), Ok(written) => written, }; @@ -923,19 +867,11 @@ mod tests { struct AlignedBytes([u8; 2 * DIRECT_IO_ALIGNMENT]); #[derive(Default)] - struct InterruptedBackend { + struct InterruptedIo { calls: AtomicUsize, } - impl IoBackend for InterruptedBackend { - fn len(&self) -> io::Result { - Ok(0) - } - - fn set_len(&self, _len: u64) -> io::Result<()> { - Ok(()) - } - + impl PositionedIo for InterruptedIo { fn read_at(&self, _buffer: &mut [u8], _offset: u64) -> io::Result { self.calls.fetch_add(1, Ordering::Relaxed); Err(io::Error::new( @@ -951,50 +887,37 @@ mod tests { "interrupted write", )) } - - fn sync(&self, _point: SyncPoint, _mode: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } #[test] fn exact_io_stops_after_the_interrupted_retry_budget() { let mut initialized = [0_u8; 1]; - let backend = InterruptedBackend::default(); - let (result, transferred) = read_exact_at_with_progress(&backend, &mut initialized, 0); + let io = InterruptedIo::default(); + let (result, transferred) = read_exact_at_with_progress(&io, &mut initialized, 0); assert_eq!(result.unwrap_err().kind(), io::ErrorKind::Interrupted); assert_eq!(transferred, 0); assert_eq!( - backend.calls.load(Ordering::Relaxed), + io.calls.load(Ordering::Relaxed), MAX_INTERRUPTED_RETRIES + 1 ); let mut uninitialized = std::mem::MaybeUninit::::uninit(); - let backend = InterruptedBackend::default(); + let io = InterruptedIo::default(); let (result, transferred) = - read_exact_at_uninit_with_progress(&backend, uninitialized.as_mut_ptr(), 1, 0); + read_exact_at_uninit_with_progress(&io, uninitialized.as_mut_ptr(), 1, 0); assert_eq!(result.unwrap_err().kind(), io::ErrorKind::Interrupted); assert_eq!(transferred, 0); assert_eq!( - backend.calls.load(Ordering::Relaxed), + io.calls.load(Ordering::Relaxed), MAX_INTERRUPTED_RETRIES + 1 ); - let backend = InterruptedBackend::default(); - let (result, transferred) = - write_all_at_with_progress(&backend, WritePoint::Record, &[1], 0); + let io = InterruptedIo::default(); + let (result, transferred) = write_all_at_with_progress(&io, WritePoint::Record, &[1], 0); assert_eq!(result.unwrap_err().kind(), io::ErrorKind::Interrupted); assert_eq!(transferred, 0); assert_eq!( - backend.calls.load(Ordering::Relaxed), + io.calls.load(Ordering::Relaxed), MAX_INTERRUPTED_RETRIES + 1 ); } @@ -1020,58 +943,58 @@ mod tests { } #[test] - fn runtime_files_route_only_fully_aligned_data_to_direct() { + fn data_handles_route_only_fully_aligned_data_to_direct() { let buffered = TestFile::new("buffered-route"); let direct = TestFile::new("direct-route"); - let files = RuntimeFileSet::new(buffered.open(), Some(direct.open())); + let files = DataFileHandles::new(buffered.open(), Some(direct.open())); let bytes = AlignedBytes([0; 2 * DIRECT_IO_ALIGNMENT]); let pointer = bytes.0.as_ptr(); assert_eq!( files.select_path(pointer, DIRECT_IO_ALIGNMENT, 0, true), - RuntimeIoPath::Direct + FileIoPath::Direct ); assert_eq!( files.select_path(pointer, DIRECT_IO_ALIGNMENT, 0, false), - RuntimeIoPath::Buffered + FileIoPath::Buffered ); assert_eq!( files.select_path(pointer, DIRECT_IO_ALIGNMENT - 1, 0, true), - RuntimeIoPath::Buffered + FileIoPath::Buffered ); - let buffered_only = RuntimeFileSet::buffered(buffered.open()); + let buffered_only = DataFileHandles::buffered(buffered.open()); assert_eq!( buffered_only.select_path(pointer, DIRECT_IO_ALIGNMENT, 0, true), - RuntimeIoPath::Buffered + FileIoPath::Buffered ); - let required = RuntimeFileSet::with_direct(buffered.open(), Some(direct.open())); + let required = DataFileHandles::with_direct(buffered.open(), Some(direct.open())); assert_eq!( required.select_path(pointer, DIRECT_IO_ALIGNMENT - 1, 0, true), - RuntimeIoPath::Buffered, + FileIoPath::Buffered, "required mode must preserve the buffered unaligned-I/O path" ); assert_eq!( required.select_path(pointer, DIRECT_IO_ALIGNMENT - 1, 0, false), - RuntimeIoPath::Buffered, + FileIoPath::Buffered, "metadata remains on the buffered control descriptor" ); } #[test] - fn runtime_io_statistics_are_shared_and_can_be_disabled() { + fn file_io_statistics_are_shared_and_can_be_disabled() { let buffered = TestFile::new("buffered-stats"); let direct = TestFile::new("direct-stats"); - let files = RuntimeFileSet::new(buffered.open(), Some(direct.open())); + let files = DataFileHandles::new(buffered.open(), Some(direct.open())); let cloned = files.try_clone().unwrap(); cloned.record( - RuntimeIoDirection::Read, - RuntimeIoPath::Direct, + FileIoDirection::Read, + FileIoPath::Direct, DIRECT_IO_ALIGNMENT, ); - cloned.record(RuntimeIoDirection::Write, RuntimeIoPath::Buffered, 32); + cloned.record(FileIoDirection::Write, FileIoPath::Buffered, 32); let stats = files.stats_handle().snapshot(); assert_eq!(stats.read.direct.operations, 1); assert_eq!(stats.read.direct.bytes, DIRECT_IO_ALIGNMENT as u64); @@ -1080,16 +1003,16 @@ mod tests { cloned.stats_handle().set_activity_counters_enabled(false); files.record( - RuntimeIoDirection::Read, - RuntimeIoPath::Direct, + FileIoDirection::Read, + FileIoPath::Direct, DIRECT_IO_ALIGNMENT, ); - files.record(RuntimeIoDirection::Write, RuntimeIoPath::Buffered, 32); + files.record(FileIoDirection::Write, FileIoPath::Buffered, 32); assert_eq!(files.stats_handle().snapshot(), stats); } #[test] - fn posix_runtime_backend_routes_record_data_and_reports_bytes() { + fn data_handles_route_record_data_and_reports_bytes() { let buffered = TestFile::new("posix-buffered-data"); let direct = TestFile::new("posix-direct-data"); let buffered_file = buffered.open(); @@ -1098,26 +1021,24 @@ mod tests { .set_len(2 * DIRECT_IO_ALIGNMENT as u64) .unwrap(); direct_file.set_len(2 * DIRECT_IO_ALIGNMENT as u64).unwrap(); - let backend = RuntimeFileBackend::new(RuntimeFileSet::new( + let io = DataFileHandles::new( buffered_file.try_clone().unwrap(), Some(direct_file.try_clone().unwrap()), - )); + ); let record = AlignedBytes([0x5a; 2 * DIRECT_IO_ALIGNMENT]); assert_eq!( - backend - .write_at(WritePoint::Record, &record.0[..DIRECT_IO_ALIGNMENT], 0,) + io.write_at(WritePoint::Record, &record.0[..DIRECT_IO_ALIGNMENT], 0,) .unwrap(), DIRECT_IO_ALIGNMENT ); assert_eq!( - backend - .write_at( - WritePoint::DataSuperblock, - &record.0[..DIRECT_IO_ALIGNMENT], - DIRECT_IO_ALIGNMENT as u64, - ) - .unwrap(), + io.write_at( + WritePoint::DataSuperblock, + &record.0[..DIRECT_IO_ALIGNMENT], + DIRECT_IO_ALIGNMENT as u64, + ) + .unwrap(), DIRECT_IO_ALIGNMENT ); @@ -1130,14 +1051,14 @@ mod tests { .unwrap(); assert!(observed.iter().all(|byte| *byte == 0x5a)); - assert_eq!(backend.read_at(&mut [], 0).unwrap(), 0); - assert_eq!(backend.write_at(WritePoint::Record, &[], 0).unwrap(), 0); + assert_eq!(io.read_at(&mut [], 0).unwrap(), 0); + assert_eq!(io.write_at(WritePoint::Record, &[], 0).unwrap(), 0); assert_eq!( - backend.files.stats_handle().snapshot(), - RuntimeIoStats { + io.stats_handle().snapshot(), + FileIoStats { direct_active: true, - write: RuntimeIoDirectionStats { + write: FileIoDirectionStats { buffered: CacheIoPathSnapshot { operations: 1, bytes: DIRECT_IO_ALIGNMENT as u64, @@ -1147,7 +1068,7 @@ mod tests { bytes: DIRECT_IO_ALIGNMENT as u64, }, }, - ..RuntimeIoStats::default() + ..FileIoStats::default() } ); } @@ -1159,12 +1080,12 @@ mod tests { #[test] fn preallocate_sets_the_exact_file_extent() { let file = TestFile::new("preallocate"); - let backend = FileBackend::open(file.path()).unwrap(); + let io = CacheFile::open(file.path()).unwrap(); let len = 2 * DIRECT_IO_ALIGNMENT as u64; - backend.preallocate(len).unwrap(); - assert_eq!(backend.len().unwrap(), len); + io.preallocate(len).unwrap(); + assert_eq!(io.len().unwrap(), len); #[cfg(target_os = "macos")] - assert!(backend.file.metadata().unwrap().blocks() * 512 >= len); + assert!(io.file.metadata().unwrap().blocks() * 512 >= len); } #[cfg(not(any( @@ -1174,24 +1095,24 @@ mod tests { #[test] fn unsupported_physical_preallocation_fails_closed() { let file = TestFile::new("preallocate-unsupported"); - let backend = FileBackend::open(file.path()).unwrap(); - let error = backend.preallocate(DIRECT_IO_ALIGNMENT as u64).unwrap_err(); + let io = CacheFile::open(file.path()).unwrap(); + let error = io.preallocate(DIRECT_IO_ALIGNMENT as u64).unwrap_err(); assert_eq!(error.kind(), io::ErrorKind::Unsupported); - assert_eq!(backend.len().unwrap(), 0); + assert_eq!(io.len().unwrap(), 0); } #[test] - fn recovery_control_backend_clones_exact_file_and_detects_aliases() { + fn storage_file_clones_exact_file_and_detects_aliases() { let primary = TestFile::new("control-primary"); let alias = TestFile::new("control-alias"); let other = TestFile::new("control-other"); drop(primary.open()); std::fs::hard_link(primary.path(), alias.path()).unwrap(); - let primary = FileBackend::open(primary.path()).unwrap(); - let alias = FileBackend::open(alias.path()).unwrap(); - let other = FileBackend::open(other.path()).unwrap(); - let cloned = ControlIoBackend::try_clone_control_file(&primary).unwrap(); + let primary = CacheFile::open(primary.path()).unwrap(); + let alias = CacheFile::open(alias.path()).unwrap(); + let other = CacheFile::open(other.path()).unwrap(); + let cloned = StorageFile::try_clone_mapping_file(&primary).unwrap(); primary .write_at(WritePoint::RecoveryImageHeader, b"image-ok", 0) @@ -1199,19 +1120,18 @@ mod tests { let mut observed = [0_u8; 8]; cloned.read_at(&mut observed, 0).unwrap(); assert_eq!(&observed, b"image-ok"); - assert!(ControlIoBackend::is_same_file(&primary, &alias).unwrap()); - assert!(!ControlIoBackend::is_same_file(&primary, &other).unwrap()); + assert!(StorageFile::is_same_file(&primary, &alias).unwrap()); + assert!(!StorageFile::is_same_file(&primary, &other).unwrap()); } #[test] fn recovery_temp_creation_never_reopens_an_existing_target() { let image = TestFile::new("recovery-create-new"); - let backend = FileBackend::create_new_buffered(image.path()).unwrap(); - backend - .write_at(WritePoint::RecoveryImageMetadata, b"metadata", 0) + let io = CacheFile::create_new_buffered(image.path()).unwrap(); + io.write_at(WritePoint::RecoveryImageMetadata, b"metadata", 0) .unwrap(); - let error = FileBackend::create_new_buffered(image.path()) + let error = CacheFile::create_new_buffered(image.path()) .err() .expect("create_new must reject an existing recovery target"); assert_eq!(error.kind(), io::ErrorKind::AlreadyExists); @@ -1219,19 +1139,18 @@ mod tests { #[test] fn one_fault_handle_controls_multiple_recovery_files() { - use crate::io::backend::testing::FaultAction; - use crate::io::backend::testing::FaultBackend; - use crate::io::backend::testing::FaultEvent; - use crate::io::backend::testing::FaultHandle; + use crate::io::file::testing::FaultAction; + use crate::io::file::testing::FaultEvent; + use crate::io::file::testing::FaultFile; + use crate::io::file::testing::FaultHandle; let state = TestFile::new("shared-fault-state"); let image = TestFile::new("shared-fault-image"); let temp = TestFile::new("shared-fault-temp"); let faults = FaultHandle::default(); - let state = FaultBackend::open_with_handle(state.path(), faults.clone()).unwrap(); - let image = FaultBackend::open_with_handle(image.path(), faults.clone()).unwrap(); - let temp = - FaultBackend::create_new_buffered_with_handle(temp.path(), faults.clone()).unwrap(); + let state = FaultFile::open_with_handle(state.path(), faults.clone()).unwrap(); + let image = FaultFile::open_with_handle(image.path(), faults.clone()).unwrap(); + let temp = FaultFile::create_new_buffered_with_handle(temp.path(), faults.clone()).unwrap(); faults.arm( FaultEvent::Write(WritePoint::State), @@ -1273,7 +1192,7 @@ mod tests { drop(target.open()); symlink(target.path(), link.path()).unwrap(); - assert!(FileBackend::open(link.path()).is_err()); + assert!(CacheFile::open(link.path()).is_err()); } } @@ -1356,22 +1275,22 @@ pub mod testing { } } - pub struct FaultBackend { - inner: FileBackend, + pub struct FaultFile { + inner: CacheFile, handle: FaultHandle, } - impl FaultBackend { + impl FaultFile { pub fn open(path: &Path) -> io::Result<(Self, FaultHandle)> { let handle = FaultHandle::default(); - let backend = Self::open_with_handle(path, handle.clone())?; - Ok((backend, handle)) + let io = Self::open_with_handle(path, handle.clone())?; + Ok((io, handle)) } /// Opens another control file governed by the same fault schedule. pub fn open_with_handle(path: &Path, handle: FaultHandle) -> io::Result { Ok(Self { - inner: FileBackend::open(path)?, + inner: CacheFile::open(path)?, handle, }) } @@ -1379,7 +1298,7 @@ pub mod testing { /// Opens an existing control file without creating a missing path. pub fn open_existing_with_handle(path: &Path, handle: FaultHandle) -> io::Result { Ok(Self { - inner: FileBackend::open_existing_with_io_mode(path, IoMode::Buffered)?, + inner: CacheFile::open_existing_with_io_mode(path, IoMode::Buffered)?, handle, }) } @@ -1391,21 +1310,13 @@ pub mod testing { handle: FaultHandle, ) -> io::Result { Ok(Self { - inner: FileBackend::create_new_buffered(path)?, + inner: CacheFile::create_new_buffered(path)?, handle, }) } } - impl IoBackend for FaultBackend { - fn len(&self) -> io::Result { - self.inner.len() - } - - fn set_len(&self, len: u64) -> io::Result<()> { - self.inner.set_len(len) - } - + impl PositionedIo for FaultFile { fn read_at(&self, buffer: &mut [u8], offset: u64) -> io::Result { match self.handle.action(FaultEvent::Read) { Some(FaultAction::Error(code) | FaultAction::ErrorAlways(code)) => { @@ -1440,6 +1351,17 @@ pub mod testing { None => self.inner.write_at(point, buffer, offset), } } + } + + #[cfg(unix)] + impl StorageFile for FaultFile { + fn len(&self) -> io::Result { + self.inner.len() + } + + fn set_len(&self, len: u64) -> io::Result<()> { + self.inner.set_len(len) + } fn sync(&self, point: SyncPoint, mode: SyncMode) -> io::Result<()> { match self.handle.action(FaultEvent::Sync(point)) { @@ -1482,16 +1404,13 @@ pub mod testing { None => self.inner.unlock(), } } - } - #[cfg(unix)] - impl ControlIoBackend for FaultBackend { - fn try_clone_control_file(&self) -> io::Result { - ControlIoBackend::try_clone_control_file(&self.inner) + fn try_clone_mapping_file(&self) -> io::Result { + StorageFile::try_clone_mapping_file(&self.inner) } - fn control_file_identity(&self) -> io::Result { - self.inner.control_file_identity() + fn identity(&self) -> io::Result { + self.inner.identity() } } diff --git a/cache2/src/io/mod.rs b/cache2/src/io/mod.rs index 9086108..d419983 100644 --- a/cache2/src/io/mod.rs +++ b/cache2/src/io/mod.rs @@ -14,5 +14,5 @@ //! Positioned file access and bounded owned-buffer execution. -pub mod backend; pub mod engine; +pub mod file; diff --git a/cache2/src/region/appender.rs b/cache2/src/region/appender.rs index 1285b66..4371167 100644 --- a/cache2/src/region/appender.rs +++ b/cache2/src/region/appender.rs @@ -23,8 +23,6 @@ use std::io; #[cfg(test)] use std::time::Duration; -use crate::io::backend::DIRECT_IO_ALIGNMENT; -use crate::io::backend::WritePoint; use crate::io::engine::BoundedIoRequest; use crate::io::engine::CACHE_IO_COMPLETION_TIMEOUT; use crate::io::engine::IoBuffer; @@ -36,6 +34,8 @@ use crate::io::engine::RequestId; use crate::io::engine::recovery::BackgroundRecovery; use crate::io::engine::recovery::RecoveryAttempt; use crate::io::engine::submit_background_io; +use crate::io::file::DIRECT_IO_ALIGNMENT; +use crate::io::file::WritePoint; use crate::region::manager::RegionWriteSpan; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; @@ -255,27 +255,17 @@ mod tests { use std::sync::Mutex; use super::*; - use crate::io::backend::IoBackend; - use crate::io::backend::SyncMode; - use crate::io::backend::SyncPoint; use crate::io::engine::IoEngine; + use crate::io::file::PositionedIo; use crate::managed_memory::BufferLease; #[derive(Default)] - struct RecordingBackend { + struct RecordingIo { writes: Mutex)>>, delay: Duration, } - impl IoBackend for RecordingBackend { - fn len(&self) -> io::Result { - Ok(u64::MAX) - } - - fn set_len(&self, _len: u64) -> io::Result<()> { - Ok(()) - } - + impl PositionedIo for RecordingIo { fn read_at(&self, _buffer: &mut [u8], _offset: u64) -> io::Result { Err(io::Error::new(io::ErrorKind::Unsupported, "read unused")) } @@ -288,18 +278,6 @@ mod tests { .push((point, offset, buffer.to_vec())); Ok(buffer.len()) } - - fn sync(&self, _point: SyncPoint, _mode: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } fn geometry() -> DataGeometry { @@ -325,11 +303,11 @@ mod tests { #[test] fn late_span_completion_remains_valid_and_keeps_engine_usable() { - let backend = Arc::new(RecordingBackend { + let io = Arc::new(RecordingIo { delay: CACHE_IO_COMPLETION_TIMEOUT + Duration::from_millis(50), - ..RecordingBackend::default() + ..RecordingIo::default() }); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let mut lease = BufferLease::try_fixed(4096).unwrap(); lease.prepare(4096).unwrap().fill(0x5a); let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; @@ -353,15 +331,15 @@ mod tests { completion.buffer.unwrap().as_slice().unwrap(), &[0x5a; 4096] ); - assert_eq!(backend.writes.lock().unwrap().len(), 1); + assert_eq!(io.writes.lock().unwrap().len(), 1); assert!(engine.try_reserve_read().is_ok()); engine.shutdown().unwrap(); } #[test] fn span_write_preserves_owned_buffer_and_maps_region_offset_exactly() { - let backend = Arc::new(RecordingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(RecordingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let mut lease = BufferLease::try_fixed(4096).unwrap(); lease.prepare(4096).unwrap().fill(0x5a); @@ -377,7 +355,7 @@ mod tests { assert!(completion.buffer.is_some()); drop(completion.buffer); - let writes = backend + let writes = io .writes .lock() .unwrap_or_else(|poisoned| poisoned.into_inner()); @@ -394,8 +372,8 @@ mod tests { #[test] fn invalid_span_returns_the_only_buffer_without_submitting_io() { - let backend = Arc::new(RecordingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(RecordingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let mut invalid = span(); invalid.end_offset += 1; let buffer = IoBuffer::for_write(BufferLease::try_fixed(4096).unwrap(), 4096).unwrap(); @@ -414,8 +392,7 @@ mod tests { assert!(error.buffer.is_some()); drop(error.buffer); assert!( - backend - .writes + io.writes .lock() .unwrap_or_else(|poisoned| poisoned.into_inner()) .is_empty() diff --git a/cache2/src/region/file_backend/mod.rs b/cache2/src/region/file_backend/mod.rs index 9a65ff3..065e413 100644 --- a/cache2/src/region/file_backend/mod.rs +++ b/cache2/src/region/file_backend/mod.rs @@ -32,16 +32,16 @@ use crate::config::runtime::IoMode; use crate::config::runtime::RuntimeOptions; #[cfg(test)] use crate::config::storage::cache_config; -use crate::io::backend::ControlIoBackend; -use crate::io::backend::FileBackend; -use crate::io::backend::IoBackend; -use crate::io::backend::RuntimeFileSet; -use crate::io::backend::SyncMode; -use crate::io::backend::SyncPoint; -use crate::io::backend::WritePoint; -use crate::io::backend::read_at_bounded; -use crate::io::backend::read_exact_at; -use crate::io::backend::write_all_at; +use crate::io::file::CacheFile; +use crate::io::file::DataFileHandles; +use crate::io::file::PositionedIo; +use crate::io::file::StorageFile; +use crate::io::file::SyncMode; +use crate::io::file::SyncPoint; +use crate::io::file::WritePoint; +use crate::io::file::read_at_bounded; +use crate::io::file::read_exact_at; +use crate::io::file::write_all_at; use crate::region::FileRegionCore; use crate::region::RegionAccessState; use crate::region::RegionHealthLatch; @@ -202,7 +202,7 @@ impl FileRegionRuntime { fn attach_data_plane( &mut self, data: DataSuperblock, - files: RuntimeFileSet, + files: DataFileHandles, config: CacheConfig, ) -> io::Result<()> { if self.data_plane.is_some() { @@ -238,7 +238,7 @@ impl FileRegionRuntime { } pub trait RegionFileSystem { - type File: ControlIoBackend; + type File: StorageFile; fn open(&self, path: &Path, create: bool) -> io::Result; @@ -246,7 +246,7 @@ pub trait RegionFileSystem { self.open(path, create) } - fn try_clone_runtime_files(&self, _file: &Self::File) -> io::Result> { + fn try_clone_data_handles(&self, _file: &Self::File) -> io::Result> { Ok(None) } @@ -263,30 +263,30 @@ pub trait RegionFileSystem { pub struct SystemRegionFileSystem; impl RegionFileSystem for SystemRegionFileSystem { - type File = FileBackend; + type File = CacheFile; fn open(&self, path: &Path, create: bool) -> io::Result { if create { - FileBackend::open_with_io_mode(path, IoMode::Buffered) + CacheFile::open_with_io_mode(path, IoMode::Buffered) } else { - FileBackend::open_existing_with_io_mode(path, IoMode::Buffered) + CacheFile::open_existing_with_io_mode(path, IoMode::Buffered) } } fn open_data(&self, path: &Path, create: bool, mode: IoMode) -> io::Result { if create { - FileBackend::open_with_io_mode(path, mode) + CacheFile::open_with_io_mode(path, mode) } else { - FileBackend::open_existing_with_io_mode(path, mode) + CacheFile::open_existing_with_io_mode(path, mode) } } - fn try_clone_runtime_files(&self, file: &Self::File) -> io::Result> { - file.try_clone_runtime_files().map(Some) + fn try_clone_data_handles(&self, file: &Self::File) -> io::Result> { + file.try_clone_data_handles().map(Some) } fn create_new(&self, path: &Path) -> io::Result { - FileBackend::create_new_buffered(path) + CacheFile::create_new_buffered(path) } fn remove_file(&self, path: &Path) -> io::Result<()> { @@ -680,7 +680,7 @@ where if !metadata.matches_image(data, header) { return self.cold_recovery("metadata_identity_mismatch"); } - let file = image.try_clone_control_file()?; + let file = image.try_clone_mapping_file()?; self.cold_reset_needed = false; Ok(Some(CleanFileRegionImage { file, @@ -837,7 +837,7 @@ where .data_file .as_ref() .ok_or_else(|| io::Error::new(io::ErrorKind::NotConnected, "data file is not open"))?; - if let Some(files) = self.file_system.try_clone_runtime_files(data_file)? { + if let Some(files) = self.file_system.try_clone_data_handles(data_file)? { runtime.attach_data_plane(data, files, self.config.clone())?; } Ok(runtime) @@ -1074,14 +1074,14 @@ where let state_result = self .state_file .as_ref() - .map(IoBackend::unlock) + .map(StorageFile::unlock) .unwrap_or(Ok(())); let data_result = if self.retain_lock { Ok(()) } else { self.data_file .as_ref() - .map(IoBackend::unlock) + .map(StorageFile::unlock) .unwrap_or(Ok(())) }; self.locked = false; @@ -1098,8 +1098,8 @@ fn inspect_or_format_data( format_data: DataSuperblock, ) -> io::Result<(DataSuperblock, bool)> where - D: IoBackend, - S: IoBackend, + D: StorageFile, + S: StorageFile, { let file_len = file.len()?; if file_len >= RECOVERY_PAGE_SIZE as u64 { @@ -1134,8 +1134,8 @@ where /// ever matching stale bytes, without scanning the file or its old records. fn format_empty_data(file: &D, state: &S, format_data: DataSuperblock) -> io::Result<()> where - D: IoBackend, - S: IoBackend, + D: StorageFile, + S: StorageFile, { let encoded = format_data .encode() @@ -1155,7 +1155,7 @@ where fn read_state_pages(file: &B) -> io::Result<[[u8; RECOVERY_PAGE_SIZE]; STATE_SLOT_COUNT]> where - B: IoBackend, + B: PositionedIo, { let mut pages = [[0_u8; RECOVERY_PAGE_SIZE]; STATE_SLOT_COUNT]; for (slot, page) in pages.iter_mut().enumerate() { @@ -1195,7 +1195,7 @@ fn select_state_for_fence( fn write_state_page(file: &B, page: &[u8; RECOVERY_PAGE_SIZE], offset: u64) -> io::Result<()> where - B: IoBackend, + B: PositionedIo, { write_all_at(file, WritePoint::State, page, offset) } @@ -1462,13 +1462,13 @@ const fn anonymous_index_backing_name() -> &'static str { } } -struct PositionedIoWriter<'a, B: IoBackend + ?Sized> { +struct PositionedIoWriter<'a, B: PositionedIo + ?Sized> { backend: &'a B, point: WritePoint, offset: u64, } -impl<'a, B: IoBackend + ?Sized> PositionedIoWriter<'a, B> { +impl<'a, B: PositionedIo + ?Sized> PositionedIoWriter<'a, B> { const fn new(backend: &'a B, point: WritePoint, offset: u64) -> Self { Self { backend, @@ -1482,7 +1482,7 @@ impl<'a, B: IoBackend + ?Sized> PositionedIoWriter<'a, B> { } } -impl Write for PositionedIoWriter<'_, B> { +impl Write for PositionedIoWriter<'_, B> { fn write(&mut self, buffer: &[u8]) -> io::Result { let written = self.backend.write_at(self.point, buffer, self.offset)?; self.offset = self diff --git a/cache2/src/region/file_backend/tests.rs b/cache2/src/region/file_backend/tests.rs index 6160c64..ffb8812 100644 --- a/cache2/src/region/file_backend/tests.rs +++ b/cache2/src/region/file_backend/tests.rs @@ -38,13 +38,13 @@ use crate::StatsOptions; use crate::config::runtime::MAX_WRITE_FLUSH_THRESHOLD_BYTES; use crate::config::runtime::PosixIoOptions; use crate::config::runtime::ReadAdmission; -use crate::io::backend::MAX_INTERRUPTED_RETRIES; -use crate::io::backend::testing::FaultAction; -use crate::io::backend::testing::FaultBackend; -use crate::io::backend::testing::FaultEvent; -use crate::io::backend::testing::FaultHandle; -use crate::io::backend::testing::kill_process; use crate::io::engine::IoEngine; +use crate::io::file::MAX_INTERRUPTED_RETRIES; +use crate::io::file::testing::FaultAction; +use crate::io::file::testing::FaultEvent; +use crate::io::file::testing::FaultFile; +use crate::io::file::testing::FaultHandle; +use crate::io::file::testing::kill_process; use crate::managed_memory::ManagedMemory; use crate::managed_memory::ManagedMemoryLimits; use crate::region::RegionStageValue; @@ -161,7 +161,7 @@ impl Drop for TestDirectory { #[test] fn state_page_reads_stop_after_the_interrupted_retry_budget() { let directory = TestDirectory::new(); - let (state, faults) = FaultBackend::open(&directory.files.state).unwrap(); + let (state, faults) = FaultFile::open(&directory.files.state).unwrap(); state.set_len(STATE_FILE_SIZE as u64).unwrap(); faults.arm(FaultEvent::Read, 1, FaultAction::ErrorAlways(libc::EINTR)); @@ -227,23 +227,23 @@ impl FaultRegionFileSystem { } impl RegionFileSystem for FaultRegionFileSystem { - type File = FaultBackend; + type File = FaultFile; fn open(&self, path: &Path, create: bool) -> io::Result { if create { - FaultBackend::open_with_handle(path, self.io.clone()) + FaultFile::open_with_handle(path, self.io.clone()) } else { - FaultBackend::open_existing_with_handle(path, self.io.clone()) + FaultFile::open_existing_with_handle(path, self.io.clone()) } } - fn try_clone_runtime_files(&self, _file: &Self::File) -> io::Result> { + fn try_clone_data_handles(&self, _file: &Self::File) -> io::Result> { self.file_system.check(FileSystemFault::CloneRuntimeFiles)?; Ok(None) } fn create_new(&self, path: &Path) -> io::Result { - FaultBackend::create_new_buffered_with_handle(path, self.io.clone()) + FaultFile::create_new_buffered_with_handle(path, self.io.clone()) } fn remove_file(&self, path: &Path) -> io::Result<()> { @@ -942,7 +942,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, faults) = FaultBackend::open(&directory.files.data).unwrap(); + let (backend, faults) = FaultFile::open(&directory.files.data).unwrap(); backend.set_len(data.geometry.data_file_len).unwrap(); let engine = IoEngine::for_test(Arc::new(backend), 2).unwrap(); let value = vec![0x5a; 16 * 1024]; @@ -1106,7 +1106,7 @@ fn same_hash_candidate_requires_full_key() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, _) = FaultBackend::open(&directory.files.data).unwrap(); + let (backend, _) = FaultFile::open(&directory.files.data).unwrap(); backend.set_len(data.geometry.data_file_len).unwrap(); let engine = IoEngine::for_test(Arc::new(backend), 1).unwrap(); let owner_key = b"collision-owner"; @@ -1251,7 +1251,7 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, faults) = FaultBackend::open(&directory.files.data).unwrap(); + let (backend, faults) = FaultFile::open(&directory.files.data).unwrap(); backend.set_len(data.geometry.data_file_len).unwrap(); let engine = IoEngine::for_test(Arc::new(backend), 1).unwrap(); let hash = hash_key(data.hash_seed, b"key"); diff --git a/cache2/src/region/manager.rs b/cache2/src/region/manager.rs index 4e7b9bd..592acd4 100644 --- a/cache2/src/region/manager.rs +++ b/cache2/src/region/manager.rs @@ -21,7 +21,7 @@ use std::collections::VecDeque; -use crate::io::backend::DIRECT_IO_ALIGNMENT; +use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::region::record::RECORD_ALIGNMENT; use crate::region::recovery::PersistentId; use crate::region::recovery::metadata::PartitionMetadataRecord; diff --git a/cache2/src/region/mod.rs b/cache2/src/region/mod.rs index fa3af82..6f5db5c 100644 --- a/cache2/src/region/mod.rs +++ b/cache2/src/region/mod.rs @@ -54,10 +54,10 @@ use self::staging::StagingEncodeError; use self::staging::StagingError; use crate::checksum::crc32c; use crate::hashing::route_hash; -use crate::io::backend::DIRECT_IO_ALIGNMENT; use crate::io::engine::IoBuffer; use crate::io::engine::IoEngine; use crate::io::engine::ReadSlot; +use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::managed_memory::BufferLease; use crate::region::appender::RegionSpanCompletion; #[cfg(test)] diff --git a/cache2/src/region/reader.rs b/cache2/src/region/reader.rs index 5e57d2c..5d6c4d5 100644 --- a/cache2/src/region/reader.rs +++ b/cache2/src/region/reader.rs @@ -24,7 +24,6 @@ use std::io; use std::ops::Range; use std::sync::Arc; -use crate::io::backend::DIRECT_IO_ALIGNMENT; use crate::io::engine::BoundedIoRequest; use crate::io::engine::IoBuffer; use crate::io::engine::IoCompletion; @@ -35,6 +34,7 @@ use crate::io::engine::OperationKind; use crate::io::engine::ReadSlot; use crate::io::engine::RequestId; use crate::io::engine::submit_cache_read; +use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::managed_memory::BufferLease; use crate::region::index::packed::IndexEntry; use crate::region::record::RECORD_ALIGNMENT; @@ -322,29 +322,19 @@ mod tests { use std::sync::Mutex; use super::*; - use crate::io::backend::IoBackend; - use crate::io::backend::SyncMode; - use crate::io::backend::SyncPoint; - use crate::io::backend::WritePoint; use crate::io::engine::IoEngine; + use crate::io::file::PositionedIo; + use crate::io::file::WritePoint; use crate::managed_memory::ManagedMemory; use crate::managed_memory::ManagedMemoryLimits; use crate::region::index::packed::PackedLocation; #[derive(Default)] - struct RecordingBackend { + struct RecordingIo { reads: Mutex>, } - impl IoBackend for RecordingBackend { - fn len(&self) -> io::Result { - Ok(u64::MAX) - } - - fn set_len(&self, _len: u64) -> io::Result<()> { - Ok(()) - } - + impl PositionedIo for RecordingIo { fn read_at(&self, buffer: &mut [u8], offset: u64) -> io::Result { self.reads .lock() @@ -359,18 +349,6 @@ mod tests { fn write_at(&self, _point: WritePoint, _buffer: &[u8], _offset: u64) -> io::Result { Err(io::Error::new(io::ErrorKind::Unsupported, "write unused")) } - - fn sync(&self, _point: SyncPoint, _mode: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } fn geometry() -> DataGeometry { @@ -396,8 +374,8 @@ mod tests { #[test] fn unaligned_record_uses_one_aligned_read_and_returns_its_exact_slice() { - let backend = Arc::new(RecordingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(RecordingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = ManagedMemory::try_new(ManagedMemoryLimits { memory_limit_bytes: DIRECT_IO_ALIGNMENT, reserved_memory_bytes: 0, @@ -421,7 +399,7 @@ mod tests { assert_eq!(completion.record_bytes().unwrap().len(), 64); let record_absolute = DATA_REGION_AREA_OFFSET + geometry().region_size + 32; - let reads = backend + let reads = io .reads .lock() .unwrap_or_else(|poisoned| poisoned.into_inner()); @@ -438,8 +416,8 @@ mod tests { #[test] fn buffered_record_uses_one_size_class_upper_bound_read() { - let backend = Arc::new(RecordingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(RecordingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = ManagedMemory::try_new(ManagedMemoryLimits { memory_limit_bytes: DIRECT_IO_ALIGNMENT, reserved_memory_bytes: 0, @@ -461,8 +439,7 @@ mod tests { assert_eq!(completion.descriptor.record_range, 0..1120); assert_eq!(completion.record_bytes().unwrap().len(), 1120); assert_eq!( - backend - .reads + io.reads .lock() .unwrap_or_else(|poisoned| poisoned.into_inner()) .as_slice(), @@ -490,15 +467,14 @@ mod tests { #[test] fn invalid_entry_is_rejected_before_allocating_or_issuing_io() { - let backend = Arc::new(RecordingBackend::default()); - let engine = IoEngine::for_test(backend.clone(), 1).unwrap(); + let io = Arc::new(RecordingIo::default()); + let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let invalid = entry(PackedLocation::new(geometry().region_count, 0, 32).unwrap()); let error = describe_read(geometry(), 7, candidate(invalid), true).unwrap_err(); assert_eq!(error.kind(), io::ErrorKind::InvalidInput); assert!( - backend - .reads + io.reads .lock() .unwrap_or_else(|poisoned| poisoned.into_inner()) .is_empty() diff --git a/cache2/src/region/record/codec.rs b/cache2/src/region/record/codec.rs index 43fb73a..923689b 100644 --- a/cache2/src/region/record/codec.rs +++ b/cache2/src/region/record/codec.rs @@ -25,7 +25,7 @@ use hashcrew::xxhash::xxh3_64_with_seed; use crate::checksum::crc32c; #[cfg(test)] -use crate::io::backend::DIRECT_IO_ALIGNMENT; +use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::region::index::packed::IndexEntry; use crate::region::index::packed::MAX_RECORD_LEN; use crate::region::index::packed::PackedLocation; diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/region/runtime/mod.rs index 6bd22c6..950df5f 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/region/runtime/mod.rs @@ -52,7 +52,6 @@ use crate::config::runtime::read_io_wait_capacity; use crate::config::runtime::read_io_wait_timeout; use crate::config::storage_geometry; use crate::hashing::route_hash; -use crate::io::backend::RuntimeFileSet; use crate::io::engine::IoBuffer; use crate::io::engine::IoEngine; use crate::io::engine::IoOperation; @@ -61,6 +60,7 @@ use crate::io::engine::ReadSlotWaiter; use crate::io::engine::build_file_engine; use crate::io::engine::recovery::BackgroundRecovery; use crate::io::engine::submit_background_io; +use crate::io::file::DataFileHandles; use crate::managed_memory::BufferLease; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::ManagedMemory; @@ -748,7 +748,7 @@ impl RegionDataPlane { pub fn new( core: Arc, data: DataSuperblock, - files: RuntimeFileSet, + files: DataFileHandles, config: CacheConfig, ) -> io::Result { // Recovery supplies independently validated metadata. It must still @@ -1305,7 +1305,7 @@ impl RegionDataPlane { } /// Fences admission, drains all workers, and shuts down the I/O engine. - /// The return value asks the backend to retain flock for process lifetime + /// The return value asks the io to retain flock for process lifetime /// because an issued write or flush could not be fenced. pub fn shutdown(&self) -> io::Result { self.start_close(); @@ -1361,10 +1361,10 @@ fn aggregate_io_stats( // File-set clones intentionally share one path counter. Read it once // rather than multiplying the same totals by the number of workers. if engine_index == 0 { - aggregate.read.buffered = snapshot.runtime.read.buffered; - aggregate.read.direct = snapshot.runtime.read.direct; - aggregate.write.buffered = snapshot.runtime.write.buffered; - aggregate.write.direct = snapshot.runtime.write.direct; + aggregate.read.buffered = snapshot.file_io.read.buffered; + aggregate.read.direct = snapshot.file_io.read.direct; + aggregate.write.buffered = snapshot.file_io.write.buffered; + aggregate.write.direct = snapshot.file_io.write.direct; } } for engine in reclaim_engines { @@ -1401,7 +1401,7 @@ fn add_io_direction(aggregate: &mut CacheIoDirectionSnapshot, snapshot: CacheIoD fn start_running( core: Arc, data: DataSuperblock, - files: RuntimeFileSet, + files: DataFileHandles, config: CacheConfig, metrics: Arc, operations: Arc, @@ -1593,7 +1593,7 @@ fn start_running( } fn build_engine_pool( - files: RuntimeFileSet, + files: DataFileHandles, runtime: &RuntimeOptions, topology: IoPoolTopology, read_wait_enabled: bool, @@ -2246,10 +2246,10 @@ mod tests { #[test] fn read_lane_uses_one_bounded_alternate_on_primary_pressure() { let file = TestFile::new("read-lane"); - let backend = file.backend(); + let io = file.io(); let engines: Box<[Arc]> = vec![ - Arc::new(IoEngine::for_test(Arc::clone(&backend), 1).unwrap()), - Arc::new(IoEngine::for_test(Arc::clone(&backend), 1).unwrap()), + Arc::new(IoEngine::for_test(Arc::clone(&io), 1).unwrap()), + Arc::new(IoEngine::for_test(Arc::clone(&io), 1).unwrap()), ] .into_boxed_slice(); let pressure_cursor = AtomicUsize::new(0); @@ -2277,15 +2277,15 @@ mod tests { engine.shutdown().unwrap(); } drop(engines); - drop(backend); + drop(io); } #[test] fn hot_read_route_rotates_pressure_fallback_across_all_lanes() { let file = TestFile::new("read-lane-rotation"); - let backend = file.backend(); + let io = file.io(); let engines: Box<[Arc]> = (0..4) - .map(|_| Arc::new(IoEngine::for_test(Arc::clone(&backend), 1).unwrap())) + .map(|_| Arc::new(IoEngine::for_test(Arc::clone(&io), 1).unwrap())) .collect::>() .into_boxed_slice(); let pressure_cursor = AtomicUsize::new(0); @@ -2303,7 +2303,7 @@ mod tests { engine.shutdown().unwrap(); } drop(engines); - drop(backend); + drop(io); } #[test] diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/region/runtime/shutdown_tests.rs index d0310a0..11d8cd2 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/region/runtime/shutdown_tests.rs @@ -17,12 +17,10 @@ use std::sync::mpsc; use super::*; use crate::IoEngineOptions; -use crate::io::backend::IoBackend; -use crate::io::backend::SyncMode; -use crate::io::backend::SyncPoint; -use crate::io::backend::WritePoint; use crate::io::engine::IoEngine; use crate::io::engine::IoRequest; +use crate::io::file::PositionedIo; +use crate::io::file::WritePoint; #[derive(Default)] struct BlockedReadState { @@ -52,15 +50,7 @@ impl BlockedRead { } } -impl IoBackend for BlockedRead { - fn len(&self) -> io::Result { - Ok(4096) - } - - fn set_len(&self, _: u64) -> io::Result<()> { - Ok(()) - } - +impl PositionedIo for BlockedRead { fn read_at(&self, bytes: &mut [u8], _: u64) -> io::Result { let mut state = self.state.lock().unwrap(); state.started = true; @@ -75,18 +65,6 @@ impl IoBackend for BlockedRead { fn write_at(&self, _: WritePoint, bytes: &[u8], _: u64) -> io::Result { Ok(bytes.len()) } - - fn sync(&self, _: SyncPoint, _: SyncMode) -> io::Result<()> { - Ok(()) - } - - fn try_lock_exclusive(&self) -> io::Result<()> { - Ok(()) - } - - fn unlock(&self) -> io::Result<()> { - Ok(()) - } } #[test] @@ -142,10 +120,10 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { store.close_fast().unwrap(); // Reuse a stopped runtime's fixed resources without unrelated workers. let shared = Arc::get_mut(&mut plane.shared).unwrap(); - let backend = Arc::new(BlockedRead::default()); - let engine = Arc::new(IoEngine::for_test(backend.clone(), 1).unwrap()); + let io = Arc::new(BlockedRead::default()); + let engine = Arc::new(IoEngine::for_test(io.clone(), 1).unwrap()); let read_engine = Arc::clone(&engine); - let read_backend = Arc::clone(&backend); + let read_io = Arc::clone(&io); let managed_memory = Arc::clone(&shared.managed_memory); let submit_read = move || -> io::Result { let slot = read_engine.try_reserve_read()?; @@ -154,7 +132,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { let request = read_engine .submit_reserved_read(slot, IoOperation::read(buffer, 0)) .map_err(|error| error.error)?; - read_backend.wait_started(); + read_io.wait_started(); Ok(request) }; let (submitted, submission) = mpsc::channel(); @@ -182,7 +160,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { tx.send(result).unwrap(); }); let result = rx.recv_timeout(Duration::from_secs(1)); - backend.release(); + io.release(); thread.join().unwrap(); engine.shutdown().unwrap(); std::fs::remove_dir_all(root).unwrap(); diff --git a/cache2/src/region/staging.rs b/cache2/src/region/staging.rs index 4218b8f..03d4493 100644 --- a/cache2/src/region/staging.rs +++ b/cache2/src/region/staging.rs @@ -21,8 +21,8 @@ use std::mem; use std::sync::Mutex; use std::sync::MutexGuard; -use crate::io::backend::DIRECT_IO_ALIGNMENT; use crate::io::engine::IoBuffer; +use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::managed_memory::BUFFER_ALIGNMENT; use crate::managed_memory::BufferLease; use crate::managed_memory::ManagedMemory; From a07b7b2ff3f89abba5f48b2bd50b2df224fdb112 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:05:11 +0800 Subject: [PATCH 03/12] refactor: place filesystem operations below region persistence Move path operations to io::fs and name the OS implementation OsFileSystem. RegionPaths now consistently denotes paths; DataFileHandles denotes descriptors for one open data file. Move descriptor cloning onto StorageFile and remove the filesystem capability that silently skipped runtime startup. Fault-injecting files now follow the same startup path and can fail descriptor cloning at the file boundary. Validation: all-target workspace check, 276 library tests, and both extended recovery qualification tests. --- cache2/src/cache.rs | 6 +- cache2/src/io/file.rs | 44 +++-- cache2/src/io/fs.rs | 79 +++++++++ cache2/src/io/mod.rs | 1 + cache2/src/region/file_backend/mod.rs | 185 ++++++-------------- cache2/src/region/file_backend/tests.rs | 179 +++++++++---------- cache2/src/region/runtime/mod.rs | 30 ++-- cache2/src/region/runtime/shutdown_tests.rs | 12 +- cache2/src/region/store.rs | 10 +- 9 files changed, 281 insertions(+), 265 deletions(-) create mode 100644 cache2/src/io/fs.rs diff --git a/cache2/src/cache.rs b/cache2/src/cache.rs index f178274..1226676 100644 --- a/cache2/src/cache.rs +++ b/cache2/src/cache.rs @@ -45,7 +45,7 @@ use crate::error::Error; use crate::error::ErrorOperation; use crate::error::from_io; use crate::region::file_backend::FileRegionBackend; -use crate::region::file_backend::RegionFiles; +use crate::region::file_backend::RegionPaths; use crate::region::recovery::DataSuperblock; use crate::region::recovery::PersistentId; use crate::region::recovery::RECOVERY_IMAGE_INDEX_OFFSET; @@ -240,7 +240,7 @@ impl Cache { hash_seed: KEY_HASH_SEED, storage_fingerprint: storage_fingerprint(config.storage()), }; - let files = RegionFiles::new( + let paths = RegionPaths::new( &path, sidecar_path(&path, ".state"), sidecar_path(&path, ".image"), @@ -253,7 +253,7 @@ impl Cache { || stats.l2_latency != LatencyMode::Off; let mutation_recording = stats.request_counters || stats.mutation_latency != LatencyMode::Off; - let backend = FileRegionBackend::new(files, format_data, config); + let backend = FileRegionBackend::new(paths, format_data, config); let store = RegionStore::open(index_slots, backend)?; let startup = store.startup(); let data_plane = store.data_plane_handle()?; diff --git a/cache2/src/io/file.rs b/cache2/src/io/file.rs index 3370631..f2a5f7b 100644 --- a/cache2/src/io/file.rs +++ b/cache2/src/io/file.rs @@ -333,6 +333,8 @@ pub trait StorageFile: PositionedIo { fn try_clone_mapping_file(&self) -> io::Result; + fn try_clone_data_handles(&self) -> io::Result; + fn identity(&self) -> io::Result; fn is_same_file(&self, other: &dyn StorageFile) -> io::Result { @@ -423,12 +425,6 @@ impl CacheFile { )) } } - - pub fn try_clone_data_handles(&self) -> io::Result { - let buffered = self.file.try_clone()?; - let direct = self.direct.as_ref().map(File::try_clone).transpose()?; - Ok(DataFileHandles::with_direct(buffered, direct)) - } } #[cfg(target_os = "macos")] @@ -457,6 +453,12 @@ fn preallocate_macos(file: &File, len: i64) -> io::Result<()> { #[cfg(unix)] impl StorageFile for CacheFile { + fn try_clone_data_handles(&self) -> io::Result { + let buffered = self.file.try_clone()?; + let direct = self.direct.as_ref().map(File::try_clone).transpose()?; + Ok(DataFileHandles::with_direct(buffered, direct)) + } + fn len(&self) -> io::Result { Ok(self.file.metadata()?.len()) } @@ -1206,6 +1208,7 @@ pub mod testing { #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum FaultEvent { Read, + CloneDataHandles, Write(WritePoint), Sync(SyncPoint), Lock, @@ -1281,6 +1284,13 @@ pub mod testing { } impl FaultFile { + pub fn with_handle(file: CacheFile, handle: FaultHandle) -> Self { + Self { + inner: file, + handle, + } + } + pub fn open(path: &Path) -> io::Result<(Self, FaultHandle)> { let handle = FaultHandle::default(); let io = Self::open_with_handle(path, handle.clone())?; @@ -1295,14 +1305,6 @@ pub mod testing { }) } - /// Opens an existing control file without creating a missing path. - pub fn open_existing_with_handle(path: &Path, handle: FaultHandle) -> io::Result { - Ok(Self { - inner: CacheFile::open_existing_with_io_mode(path, IoMode::Buffered)?, - handle, - }) - } - /// Atomically creates a new control file governed by an existing fault /// schedule. This is used for unpublished recovery-image temporaries. pub fn create_new_buffered_with_handle( @@ -1355,6 +1357,20 @@ pub mod testing { #[cfg(unix)] impl StorageFile for FaultFile { + fn try_clone_data_handles(&self) -> io::Result { + match self.handle.action(FaultEvent::CloneDataHandles) { + Some(FaultAction::Error(code) | FaultAction::ErrorAlways(code)) => { + Err(io::Error::from_raw_os_error(code)) + } + Some(FaultAction::Torn { .. }) => Err(io::Error::new( + io::ErrorKind::InvalidInput, + "torn actions apply only to positioned I/O", + )), + Some(FaultAction::KillAfter) => kill_after(self.inner.try_clone_data_handles()), + None => self.inner.try_clone_data_handles(), + } + } + fn len(&self) -> io::Result { self.inner.len() } diff --git a/cache2/src/io/fs.rs b/cache2/src/io/fs.rs new file mode 100644 index 0000000..b8e3c36 --- /dev/null +++ b/cache2/src/io/fs.rs @@ -0,0 +1,79 @@ +// Copyright 2026 ScopeDB, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Path operations used to open cache files and atomically install recovery images. + +use std::fs; +use std::fs::File; +use std::io; +use std::path::Path; + +use crate::config::runtime::IoMode; +use crate::io::file::CacheFile; +use crate::io::file::StorageFile; + +pub trait FileSystem { + type File: StorageFile; + + fn open(&self, path: &Path, create: bool, mode: IoMode) -> io::Result; + + fn create_new(&self, path: &Path) -> io::Result; + + fn remove_file(&self, path: &Path) -> io::Result<()>; + + fn rename(&self, source: &Path, destination: &Path) -> io::Result<()>; + + fn sync_parent(&self, path: &Path) -> io::Result<()>; +} + +#[derive(Clone, Copy, Debug, Default)] +pub struct OsFileSystem; + +impl FileSystem for OsFileSystem { + type File = CacheFile; + + fn open(&self, path: &Path, create: bool, mode: IoMode) -> io::Result { + if create { + CacheFile::open_with_io_mode(path, mode) + } else { + CacheFile::open_existing_with_io_mode(path, mode) + } + } + + fn create_new(&self, path: &Path) -> io::Result { + CacheFile::create_new_buffered(path) + } + + fn remove_file(&self, path: &Path) -> io::Result<()> { + match fs::remove_file(path) { + Ok(()) => Ok(()), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(()), + Err(error) => Err(error), + } + } + + fn rename(&self, source: &Path, destination: &Path) -> io::Result<()> { + fs::rename(source, destination) + } + + fn sync_parent(&self, path: &Path) -> io::Result<()> { + File::open(parent_directory(path))?.sync_all() + } +} + +pub fn parent_directory(path: &Path) -> &Path { + path.parent() + .filter(|parent| !parent.as_os_str().is_empty()) + .unwrap_or_else(|| Path::new(".")) +} diff --git a/cache2/src/io/mod.rs b/cache2/src/io/mod.rs index d419983..7364595 100644 --- a/cache2/src/io/mod.rs +++ b/cache2/src/io/mod.rs @@ -16,3 +16,4 @@ pub mod engine; pub mod file; +pub mod fs; diff --git a/cache2/src/region/file_backend/mod.rs b/cache2/src/region/file_backend/mod.rs index 065e413..c1d925d 100644 --- a/cache2/src/region/file_backend/mod.rs +++ b/cache2/src/region/file_backend/mod.rs @@ -15,7 +15,6 @@ //! File ownership, recovery, and lifecycle adapter for the Region core. use std::fmt; -use std::fs; use std::fs::File; use std::io::Write; use std::io::{self}; @@ -32,7 +31,6 @@ use crate::config::runtime::IoMode; use crate::config::runtime::RuntimeOptions; #[cfg(test)] use crate::config::storage::cache_config; -use crate::io::file::CacheFile; use crate::io::file::DataFileHandles; use crate::io::file::PositionedIo; use crate::io::file::StorageFile; @@ -42,6 +40,9 @@ use crate::io::file::WritePoint; use crate::io::file::read_at_bounded; use crate::io::file::read_exact_at; use crate::io::file::write_all_at; +use crate::io::fs::FileSystem; +use crate::io::fs::OsFileSystem; +use crate::io::fs::parent_directory; use crate::region::FileRegionCore; use crate::region::RegionAccessState; use crate::region::RegionHealthLatch; @@ -93,15 +94,15 @@ use crate::region::runtime::RegionDataPlane; #[cfg(test)] const REGION_SHARDS: u32 = 4; -/// Data and recovery sidecars owned by one concrete Region backend. +/// Paths to the data file and its state and recovery-image sidecars. #[derive(Clone, Debug, Eq, PartialEq)] -pub struct RegionFiles { +pub struct RegionPaths { pub data: PathBuf, pub state: PathBuf, pub image: PathBuf, } -impl RegionFiles { +impl RegionPaths { pub fn new( data: impl Into, state: impl Into, @@ -202,7 +203,7 @@ impl FileRegionRuntime { fn attach_data_plane( &mut self, data: DataSuperblock, - files: DataFileHandles, + handles: DataFileHandles, config: CacheConfig, ) -> io::Result<()> { if self.data_plane.is_some() { @@ -214,7 +215,7 @@ impl FileRegionRuntime { self.data_plane = Some(RegionDataPlane::new( Arc::clone(&self.core), data, - files, + handles, config, )?); Ok(()) @@ -237,84 +238,15 @@ impl FileRegionRuntime { } } -pub trait RegionFileSystem { - type File: StorageFile; - - fn open(&self, path: &Path, create: bool) -> io::Result; - - fn open_data(&self, path: &Path, create: bool, _mode: IoMode) -> io::Result { - self.open(path, create) - } - - fn try_clone_data_handles(&self, _file: &Self::File) -> io::Result> { - Ok(None) - } - - fn create_new(&self, path: &Path) -> io::Result; - - fn remove_file(&self, path: &Path) -> io::Result<()>; - - fn rename(&self, source: &Path, destination: &Path) -> io::Result<()>; - - fn sync_parent(&self, path: &Path) -> io::Result<()>; -} - -#[derive(Clone, Copy, Debug, Default)] -pub struct SystemRegionFileSystem; - -impl RegionFileSystem for SystemRegionFileSystem { - type File = CacheFile; - - fn open(&self, path: &Path, create: bool) -> io::Result { - if create { - CacheFile::open_with_io_mode(path, IoMode::Buffered) - } else { - CacheFile::open_existing_with_io_mode(path, IoMode::Buffered) - } - } - - fn open_data(&self, path: &Path, create: bool, mode: IoMode) -> io::Result { - if create { - CacheFile::open_with_io_mode(path, mode) - } else { - CacheFile::open_existing_with_io_mode(path, mode) - } - } - - fn try_clone_data_handles(&self, file: &Self::File) -> io::Result> { - file.try_clone_data_handles().map(Some) - } - - fn create_new(&self, path: &Path) -> io::Result { - CacheFile::create_new_buffered(path) - } - - fn remove_file(&self, path: &Path) -> io::Result<()> { - match fs::remove_file(path) { - Ok(()) => Ok(()), - Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(()), - Err(error) => Err(error), - } - } - - fn rename(&self, source: &Path, destination: &Path) -> io::Result<()> { - fs::rename(source, destination) - } - - fn sync_parent(&self, path: &Path) -> io::Result<()> { - File::open(parent_directory(path))?.sync_all() - } -} - /// Concrete state/index lifecycle backed by one data file and two sidecars. /// /// It owns the append/read runtime, persists one complete index plus the /// Region/FIFO physical view, and never scans records during open. -pub struct FileRegionBackend +pub struct FileRegionBackend where - F: RegionFileSystem, + F: FileSystem, { - files: RegionFiles, + paths: RegionPaths, /// Used when the data file is missing or empty. Existing /// files retain their on-disk identities but must match this geometry and /// storage-layout fingerprint. @@ -331,51 +263,51 @@ where retain_lock: bool, } -impl FileRegionBackend { +impl FileRegionBackend { #[cfg(test)] - fn for_test(files: RegionFiles, format_data: DataSuperblock, index_slots: usize) -> Self { - Self::for_test_with_options(files, format_data, index_slots, RuntimeOptions::default()) + fn for_test(paths: RegionPaths, format_data: DataSuperblock, index_slots: usize) -> Self { + Self::for_test_with_options(paths, format_data, index_slots, RuntimeOptions::default()) } #[cfg(test)] pub fn for_test_with_options( - files: RegionFiles, + paths: RegionPaths, format_data: DataSuperblock, index_slots: usize, runtime_options: RuntimeOptions, ) -> Self { let config = cache_config(format_data.geometry, index_slots, runtime_options); - Self::new(files, format_data, config) + Self::new(paths, format_data, config) } - pub fn new(files: RegionFiles, format_data: DataSuperblock, config: CacheConfig) -> Self { - Self::new_with_file_system(files, format_data, SystemRegionFileSystem, config) + pub fn new(paths: RegionPaths, format_data: DataSuperblock, config: CacheConfig) -> Self { + Self::new_with_file_system(paths, format_data, OsFileSystem, config) } } impl FileRegionBackend where - F: RegionFileSystem, + F: FileSystem, { #[cfg(test)] fn for_test_with_file_system( - files: RegionFiles, + paths: RegionPaths, format_data: DataSuperblock, index_slots: usize, file_system: F, ) -> Self { let config = cache_config(format_data.geometry, index_slots, RuntimeOptions::default()); - Self::new_with_file_system(files, format_data, file_system, config) + Self::new_with_file_system(paths, format_data, file_system, config) } fn new_with_file_system( - files: RegionFiles, + paths: RegionPaths, format_data: DataSuperblock, file_system: F, config: CacheConfig, ) -> Self { Self { - files, + paths, format_data, config, file_system, @@ -409,8 +341,8 @@ where log::info!( target: "cache2::recovery", event = "cache_recovery_cold", - path:% = self.files.data.display(), - image_path:% = self.files.image.display(), + path:% = self.paths.data.display(), + image_path:% = self.paths.image.display(), reason; "cache recovery selected cold start" ); @@ -426,8 +358,8 @@ where log::warn!( target: "cache2::recovery", event = "cache_recovery_cold", - path:% = self.files.data.display(), - image_path:% = self.files.image.display(), + path:% = self.paths.data.display(), + image_path:% = self.paths.image.display(), reason, index_backing = "file_private_mmap", index_slots, @@ -441,7 +373,7 @@ where log::info!( target: "cache2::recovery", event = "cache_append_shards_rebind_planned", - path:% = self.files.data.display(), + path:% = self.paths.data.display(), previous_append_shards = previous, append_shards = current, reused_active_regions = previous.min(current), @@ -456,7 +388,7 @@ where Ok(None) } - /// Acquire exclusive ownership of all files before inspection. + /// Acquire exclusive ownership of the data and state files before inspection. pub fn acquire_exclusive(&mut self) -> io::Result<()> { if self.locked { return Err(io::Error::new( @@ -464,38 +396,41 @@ where "RegionStore backend is already locked", )); } - if self.files.data == self.files.state - || self.files.data == self.files.image - || self.files.state == self.files.image + if self.paths.data == self.paths.state + || self.paths.data == self.paths.image + || self.paths.state == self.paths.image { return Err(io::Error::new( io::ErrorKind::InvalidInput, "RegionStore data/state/image paths must be distinct", )); } - if parent_directory(&self.files.data) != parent_directory(&self.files.state) - || parent_directory(&self.files.data) != parent_directory(&self.files.image) + if parent_directory(&self.paths.data) != parent_directory(&self.paths.state) + || parent_directory(&self.paths.data) != parent_directory(&self.paths.image) { return Err(io::Error::new( io::ErrorKind::InvalidInput, "RegionStore data/state/image files must share one directory", )); } - let temporary = recovery_temporary_path(&self.files.image); - if temporary == self.files.data - || temporary == self.files.state - || temporary == self.files.image + let temporary = recovery_temporary_path(&self.paths.image); + if temporary == self.paths.data + || temporary == self.paths.state + || temporary == self.paths.image { return Err(io::Error::new( io::ErrorKind::InvalidInput, "RegionStore recovery temporary path collides with a cache file", )); } - let data = - self.file_system - .open_data(&self.files.data, true, self.config.runtime().io_mode)?; + let data = self + .file_system + .open(&self.paths.data, true, self.config.runtime().io_mode)?; data.try_lock_exclusive()?; - let state = match self.file_system.open(&self.files.state, true) { + let state = match self + .file_system + .open(&self.paths.state, true, IoMode::Buffered) + { Ok(state) => state, Err(error) => { let _ = data.unlock(); @@ -533,7 +468,7 @@ where index_slots: usize, ) -> io::Result> { self.file_system - .remove_file(&recovery_temporary_path(&self.files.image))?; + .remove_file(&recovery_temporary_path(&self.paths.image))?; let format_data = self.format_data; let (data, fresh) = { let data_file = self.data_file.as_ref().ok_or_else(|| { @@ -574,7 +509,10 @@ where return self.cold_recovery("state_data_mismatch"); } - let image = match self.file_system.open(&self.files.image, false) { + let image = match self + .file_system + .open(&self.paths.image, false, IoMode::Buffered) + { Ok(image) => image, Err(error) if error.kind() == io::ErrorKind::NotFound => { return self.cold_recovery("image_missing"); @@ -691,9 +629,9 @@ where /// Construct a provisional empty runtime without starting workers. pub fn anonymous_runtime(&mut self, index_slots: usize) -> io::Result { - self.file_system.remove_file(&self.files.image)?; + self.file_system.remove_file(&self.paths.image)?; self.file_system - .remove_file(&recovery_temporary_path(&self.files.image))?; + .remove_file(&recovery_temporary_path(&self.paths.image))?; let data = self.data_superblock()?; if self.cold_reset_needed { let data_file = self.data_file.as_ref().ok_or_else(|| { @@ -718,7 +656,7 @@ where log::error!( target: "cache2::recovery", event = "cache_index_backing_failed", - path:% = self.files.data.display(), + path:% = self.paths.data.display(), index_backing = anonymous_index_backing_name(), index_slots, index_mapping_bytes, @@ -837,9 +775,8 @@ where .data_file .as_ref() .ok_or_else(|| io::Error::new(io::ErrorKind::NotConnected, "data file is not open"))?; - if let Some(files) = self.file_system.try_clone_data_handles(data_file)? { - runtime.attach_data_plane(data, files, self.config.clone())?; - } + let handles = data_file.try_clone_data_handles()?; + runtime.attach_data_plane(data, handles, self.config.clone())?; Ok(runtime) } @@ -984,7 +921,7 @@ where .ok_or_else(|| io::Error::new(io::ErrorKind::NotConnected, "data file is not open"))?; data_file.sync(SyncPoint::WarmData, SyncMode::Data)?; - let temporary = recovery_temporary_path(&self.files.image); + let temporary = recovery_temporary_path(&self.paths.image); self.file_system.remove_file(&temporary)?; let persisted = (|| { let image = self.file_system.create_new(&temporary)?; @@ -1015,8 +952,8 @@ where )?; image.sync(SyncPoint::RecoveryImage, SyncMode::Data)?; health.require_healthy()?; - self.file_system.rename(&temporary, &self.files.image)?; - self.file_system.sync_parent(&self.files.image)?; + self.file_system.rename(&temporary, &self.paths.image)?; + self.file_system.sync_parent(&self.paths.image)?; health.require_healthy() })(); if persisted.is_err() { @@ -1445,12 +1382,6 @@ fn recovery_temporary_path(image: &Path) -> PathBuf { PathBuf::from(path) } -fn parent_directory(path: &Path) -> &Path { - path.parent() - .filter(|parent| !parent.as_os_str().is_empty()) - .unwrap_or_else(|| Path::new(".")) -} - const fn anonymous_index_backing_name() -> &'static str { #[cfg(any(target_os = "linux", target_os = "macos"))] { diff --git a/cache2/src/region/file_backend/tests.rs b/cache2/src/region/file_backend/tests.rs index ffb8812..93ad1fa 100644 --- a/cache2/src/region/file_backend/tests.rs +++ b/cache2/src/region/file_backend/tests.rs @@ -133,7 +133,7 @@ async fn assert_pending(mut future: Pin<&mut F>, message: &str) { struct TestDirectory { root: PathBuf, - files: RegionFiles, + paths: RegionPaths, } impl TestDirectory { @@ -142,12 +142,12 @@ impl TestDirectory { let root = env::temp_dir().join(format!("cache2-region-{}-{ordinal}", std::process::id())); let _ = fs::remove_dir_all(&root); fs::create_dir(&root).unwrap(); - let files = RegionFiles::new( + let paths = RegionPaths::new( root.join("data"), root.join("state"), root.join("recovery.image"), ); - Self { root, files } + Self { root, paths } } } @@ -161,7 +161,7 @@ impl Drop for TestDirectory { #[test] fn state_page_reads_stop_after_the_interrupted_retry_budget() { let directory = TestDirectory::new(); - let (state, faults) = FaultFile::open(&directory.files.state).unwrap(); + let (state, faults) = FaultFile::open(&directory.paths.state).unwrap(); state.set_len(STATE_FILE_SIZE as u64).unwrap(); faults.arm(FaultEvent::Read, 1, FaultAction::ErrorAlways(libc::EINTR)); @@ -179,7 +179,6 @@ fn state_page_reads_stop_after_the_interrupted_retry_budget() { #[derive(Clone, Copy, Debug, Eq, PartialEq)] enum FileSystemFault { - CloneRuntimeFiles, Rename, SyncParent, } @@ -206,12 +205,12 @@ impl FileSystemFaultHandle { } #[derive(Clone)] -struct FaultRegionFileSystem { +struct FaultFileSystem { io: FaultHandle, file_system: FileSystemFaultHandle, } -impl FaultRegionFileSystem { +impl FaultFileSystem { fn new() -> (Self, FaultHandle, FileSystemFaultHandle) { let io = FaultHandle::default(); let file_system = FileSystemFaultHandle::default(); @@ -226,20 +225,12 @@ impl FaultRegionFileSystem { } } -impl RegionFileSystem for FaultRegionFileSystem { +impl FileSystem for FaultFileSystem { type File = FaultFile; - fn open(&self, path: &Path, create: bool) -> io::Result { - if create { - FaultFile::open_with_handle(path, self.io.clone()) - } else { - FaultFile::open_existing_with_handle(path, self.io.clone()) - } - } - - fn try_clone_data_handles(&self, _file: &Self::File) -> io::Result> { - self.file_system.check(FileSystemFault::CloneRuntimeFiles)?; - Ok(None) + fn open(&self, path: &Path, create: bool, mode: IoMode) -> io::Result { + let file = OsFileSystem.open(path, create, mode)?; + Ok(FaultFile::with_handle(file, self.io.clone())) } fn create_new(&self, path: &Path) -> io::Result { @@ -261,7 +252,7 @@ impl RegionFileSystem for FaultRegionFileSystem { fn sync_parent(&self, path: &Path) -> io::Result<()> { self.file_system.check(FileSystemFault::SyncParent)?; - SystemRegionFileSystem.sync_parent(path) + OsFileSystem.sync_parent(path) } } @@ -318,12 +309,12 @@ fn external_process_kill_recovery_contract() { if let Ok(case) = env::var(CHILD_CASE) { let root = PathBuf::from(env::var_os(CHILD_ROOT).expect("child root is set")); - let files = RegionFiles::new( + let paths = RegionPaths::new( root.join("data"), root.join("state"), root.join("recovery.image"), ); - run_crash_child(&case, files); + run_crash_child(&case, paths); } for (case, expect_clean) in [ @@ -338,7 +329,7 @@ fn external_process_kill_recovery_contract() { let data = data_path_superblock(); let mut initial = RegionStore::open( 4096, - FileRegionBackend::for_test(directory.files.clone(), data, 4096), + FileRegionBackend::for_test(directory.paths.clone(), data, 4096), ) .unwrap(); eventually_admitted(|| initial.put_value(b"survivor", b"old")); @@ -364,7 +355,7 @@ fn external_process_kill_recovery_contract() { let mut reopened = RegionStore::open( 4096, - FileRegionBackend::for_test(directory.files.clone(), data, 4096), + FileRegionBackend::for_test(directory.paths.clone(), data, 4096), ) .unwrap(); if expect_clean { @@ -383,17 +374,17 @@ fn external_process_kill_recovery_contract() { } #[cfg(unix)] -fn run_crash_child(case: &str, files: RegionFiles) -> ! { +fn run_crash_child(case: &str, paths: RegionPaths) -> ! { let data = data_path_superblock(); match case { "open" => { let _store = - RegionStore::open(4096, FileRegionBackend::for_test(files, data, 4096)).unwrap(); + RegionStore::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); kill_process(); } "write" | "drain" => { let store = - RegionStore::open(4096, FileRegionBackend::for_test(files, data, 4096)).unwrap(); + RegionStore::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); eventually_admitted(|| store.put_value(b"replacement", b"new")); if case == "drain" { store.drain().unwrap(); @@ -401,10 +392,10 @@ fn run_crash_child(case: &str, files: RegionFiles) -> ! { kill_process(); } "warm-data" | "warm-image" | "clean-state" => { - let (file_system, faults, _) = FaultRegionFileSystem::new(); + let (file_system, faults, _) = FaultFileSystem::new(); let mut store = RegionStore::open( 4096, - FileRegionBackend::for_test_with_file_system(files, data, 4096, file_system), + FileRegionBackend::for_test_with_file_system(paths, data, 4096, file_system), ) .unwrap(); let point = match case { @@ -471,7 +462,7 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { let mut store = RegionStore::open( 4096, FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, 4096, runtime_options, @@ -558,7 +549,7 @@ fn queued_l2_read_does_not_pin_warm_close() { let mut store = RegionStore::open( 4096, FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, 4096, runtime_options.clone(), @@ -590,7 +581,7 @@ fn queued_l2_read_does_not_pin_warm_close() { let mut reopened = RegionStore::open( 4096, FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, 4096, runtime_options, @@ -616,7 +607,7 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { let mut store = RegionStore::open( 4096, FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, 4096, runtime_options, @@ -694,7 +685,7 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { drop(retained_hits); let mut recovered = RegionStore::open( 4096, - FileRegionBackend::for_test(directory.files.clone(), data, 4096), + FileRegionBackend::for_test(directory.paths.clone(), data, 4096), ) .unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); @@ -728,7 +719,7 @@ fn poisoned_runtime_gates_stop_workers_and_reject_warm_close() { let mut store = RegionStore::open( 4096, FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, 4096, runtime_options.clone(), @@ -752,7 +743,7 @@ fn poisoned_runtime_gates_stop_workers_and_reject_warm_close() { let mut reopened = RegionStore::open( 4096, FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, 4096, runtime_options, @@ -942,7 +933,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, faults) = FaultFile::open(&directory.files.data).unwrap(); + let (backend, faults) = FaultFile::open(&directory.paths.data).unwrap(); backend.set_len(data.geometry.data_file_len).unwrap(); let engine = IoEngine::for_test(Arc::new(backend), 2).unwrap(); let value = vec![0x5a; 16 * 1024]; @@ -1106,7 +1097,7 @@ fn same_hash_candidate_requires_full_key() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, _) = FaultFile::open(&directory.files.data).unwrap(); + let (backend, _) = FaultFile::open(&directory.paths.data).unwrap(); backend.set_len(data.geometry.data_file_len).unwrap(); let engine = IoEngine::for_test(Arc::new(backend), 1).unwrap(); let owner_key = b"collision-owner"; @@ -1251,7 +1242,7 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, faults) = FaultFile::open(&directory.files.data).unwrap(); + let (backend, faults) = FaultFile::open(&directory.paths.data).unwrap(); backend.set_len(data.geometry.data_file_len).unwrap(); let engine = IoEngine::for_test(Arc::new(backend), 1).unwrap(); let hash = hash_key(data.hash_seed, b"key"); @@ -1334,11 +1325,11 @@ fn fresh_and_dirty_startup_do_not_write_runtime_region_metadata() { let config = 8; let data = test_data_superblock(); - let (fresh_file_system, fresh_io, _) = FaultRegionFileSystem::new(); + let (fresh_file_system, fresh_io, _) = FaultFileSystem::new(); let mut fresh = RegionStore::open( config, FileRegionBackend::for_test_with_file_system( - directory.files.clone(), + directory.paths.clone(), data, config, fresh_file_system, @@ -1349,11 +1340,11 @@ fn fresh_and_dirty_startup_do_not_write_runtime_region_metadata() { assert_no_runtime_data_write_during_startup(&fresh_io.events()); fresh.close_fast().unwrap(); - let (dirty_file_system, dirty_io, _) = FaultRegionFileSystem::new(); + let (dirty_file_system, dirty_io, _) = FaultFileSystem::new(); let mut dirty = RegionStore::open( config, FileRegionBackend::for_test_with_file_system( - directory.files.clone(), + directory.paths.clone(), data, config, dirty_file_system, @@ -1377,7 +1368,7 @@ fn publish_custom_clean_image( ..RuntimeOptions::default() }; let mut backend = FileRegionBackend::for_test_with_options( - directory.files.clone(), + directory.paths.clone(), data, index_slots, runtime_options, @@ -1408,7 +1399,7 @@ fn clean_image_rebinds_a_different_append_shard_topology() { let mut reopened = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(reopened.startup(), StartupMode::Warm); @@ -1420,7 +1411,7 @@ fn clean_image_rebinds_a_different_append_shard_topology() { let mut stable = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(stable.startup(), StartupMode::Warm); @@ -1457,7 +1448,7 @@ fn append_shard_growth_without_free_regions_cold_starts_safely() { let mut reopened = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(reopened.startup(), StartupMode::Cold); @@ -1477,7 +1468,7 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { let data = test_data_superblock(); let mut first = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); first.close_fast().unwrap(); @@ -1486,19 +1477,19 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { let file = File::options() .read(true) .write(true) - .open(&directory.files.data) + .open(&directory.paths.data) .unwrap(); file.write_all_at(b"stale-record", stale_offset).unwrap(); file.sync_data().unwrap(); let mut cold = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); let mut observed = [0xff_u8; 12]; - File::open(&directory.files.data) + File::open(&directory.paths.data) .unwrap() .read_exact_at(&mut observed, stale_offset) .unwrap(); @@ -1512,7 +1503,7 @@ fn invalid_capacity_is_rejected_before_creating_files() { let directory = TestDirectory::new(); let opened = RegionStore::open( index_slots, - FileRegionBackend::for_test(directory.files.clone(), test_data_superblock(), 8), + FileRegionBackend::for_test(directory.paths.clone(), test_data_superblock(), 8), ); assert_eq!(opened.err().unwrap().kind(), io::ErrorKind::InvalidInput); assert_eq!(fs::read_dir(&directory.root).unwrap().count(), 0); @@ -1525,14 +1516,14 @@ fn dropping_a_warm_runtime_releases_ownership_and_reopens_cold() { let data = test_data_superblock(); let mut initial = RegionStore::open( 8, - FileRegionBackend::for_test(directory.files.clone(), data, 8), + FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) .unwrap(); initial.close_warm().unwrap(); let recovered = RegionStore::open( 8, - FileRegionBackend::for_test(directory.files.clone(), data, 8), + FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) .unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); @@ -1540,7 +1531,7 @@ fn dropping_a_warm_runtime_releases_ownership_and_reopens_cold() { let mut cold = RegionStore::open( 8, - FileRegionBackend::for_test(directory.files.clone(), data, 8), + FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) .unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); @@ -1562,7 +1553,7 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { let mut first = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); let runtime = first.runtime_mut().unwrap(); @@ -1573,7 +1564,7 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { .write_slot(config - 1, value) .unwrap(); first.close_warm().unwrap(); - assert!(directory.files.image.exists()); + assert!(directory.paths.image.exists()); first.close_warm().unwrap(); first.close_fast().unwrap(); assert_eq!( @@ -1584,7 +1575,7 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { let mut recovered = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); @@ -1610,7 +1601,7 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { let data = test_data_superblock_with_regions(REGION_SHARDS + 1); let mut first = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); first.close_warm().unwrap(); @@ -1618,7 +1609,7 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { let image = File::options() .read(true) .write(true) - .open(&directory.files.image) + .open(&directory.paths.image) .unwrap(); let mut page = [0_u8; RECOVERY_PAGE_SIZE]; image.read_exact_at(&mut page, 0).unwrap(); @@ -1632,7 +1623,7 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { let mut rejected = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(rejected.startup(), StartupMode::Cold); @@ -1652,7 +1643,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { let data = test_data_superblock_with_regions(REGION_SHARDS + 1); let mut first = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); first.close_warm().unwrap(); @@ -1660,7 +1651,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { let image = File::options() .read(true) .write(true) - .open(&directory.files.image) + .open(&directory.paths.image) .unwrap(); image .write_all_at( @@ -1672,7 +1663,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { let mut recovered = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); @@ -1687,7 +1678,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { let mut cold = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); @@ -1755,9 +1746,9 @@ fn every_prepublication_failure_leaves_no_selectable_clean_state() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock_with_regions(REGION_SHARDS + 1); - let (file_system, io_faults, file_system_faults) = FaultRegionFileSystem::new(); + let (file_system, io_faults, file_system_faults) = FaultFileSystem::new(); let backend = FileRegionBackend::for_test_with_file_system( - directory.files.clone(), + directory.paths.clone(), data, config, file_system, @@ -1773,7 +1764,7 @@ fn every_prepublication_failure_leaves_no_selectable_clean_state() { let mut reopened = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert_eq!(reopened.startup(), StartupMode::Cold, "failure case {case}"); @@ -1808,28 +1799,26 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let (file_system, faults, file_system_faults) = FaultRegionFileSystem::new(); + let (file_system, faults, _) = FaultFileSystem::new(); faults.arm(event, occurrence, action); - file_system_faults.arm(FileSystemFault::CloneRuntimeFiles); let opened = RegionStore::open( config, FileRegionBackend::for_test_with_file_system( - directory.files.clone(), + directory.paths.clone(), data, config, file_system, ), ); assert!(opened.is_err(), "RUNNING barrier case {case}"); - assert_eq!( - *file_system_faults.armed.lock().unwrap(), - Some(FileSystemFault::CloneRuntimeFiles), - "runtime files must not be requested before RUNNING is durable" + assert!( + !faults.events().contains(&FaultEvent::CloneDataHandles), + "runtime paths must not be requested before RUNNING is durable" ); let mut cold = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert!(matches!(cold.startup(), StartupMode::Cold)); @@ -1841,14 +1830,14 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { let directory = TestDirectory::new(); let data = test_data_superblock(); - let (file_system, faults, file_system_faults) = FaultRegionFileSystem::new(); - file_system_faults.arm(FileSystemFault::CloneRuntimeFiles); + let (file_system, faults, _) = FaultFileSystem::new(); + faults.arm(FaultEvent::CloneDataHandles, 1, FaultAction::Error(5)); let opened = RegionStore::open( 8, - FileRegionBackend::for_test_with_file_system(directory.files.clone(), data, 8, file_system), + FileRegionBackend::for_test_with_file_system(directory.paths.clone(), data, 8, file_system), ); assert_eq!(opened.err().unwrap().raw_os_error(), Some(5)); - assert_eq!(*file_system_faults.armed.lock().unwrap(), None); + assert!(faults.events().contains(&FaultEvent::CloneDataHandles)); assert!( faults .events() @@ -1857,7 +1846,7 @@ fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { let mut cold = RegionStore::open( 8, - FileRegionBackend::for_test(directory.files.clone(), data, 8), + FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) .unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); @@ -1869,11 +1858,11 @@ fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let (file_system, faults, _) = FaultRegionFileSystem::new(); + let (file_system, faults, _) = FaultFileSystem::new(); let mut store = RegionStore::open( config, FileRegionBackend::for_test_with_file_system( - directory.files.clone(), + directory.paths.clone(), data, config, file_system, @@ -1889,7 +1878,7 @@ fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { let mut reopened = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); assert!(matches!( @@ -1909,18 +1898,18 @@ fn data_and_state_inode_alias_is_rejected_without_truncation() { let config = 8; let data = test_data_superblock(); let marker = b"do-not-truncate"; - fs::write(&directory.files.data, marker).unwrap(); - fs::hard_link(&directory.files.data, &directory.files.state).unwrap(); + fs::write(&directory.paths.data, marker).unwrap(); + fs::hard_link(&directory.paths.data, &directory.paths.state).unwrap(); let opened = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ); assert!(matches!( opened, Err(error) if error.kind() == io::ErrorKind::InvalidInput )); - assert_eq!(fs::read(&directory.files.data).unwrap(), marker); + assert_eq!(fs::read(&directory.paths.data).unwrap(), marker); } #[test] @@ -1930,12 +1919,12 @@ fn recovery_temporary_path_cannot_name_the_data_or_state_file() { let image = directory.root.join("recovery"); let data_path = directory.root.join("recovery.next"); fs::write(&data_path, marker).unwrap(); - let files = RegionFiles::new(&data_path, directory.root.join("state"), image); + let paths = RegionPaths::new(&data_path, directory.root.join("state"), image); let opened = RegionStore::open( 8, FileRegionBackend::for_test( - files, + paths, test_data_superblock_with_regions(REGION_SHARDS + 1), 8, ), @@ -1952,14 +1941,14 @@ fn recovery_sidecars_must_share_one_directory() { let directory = TestDirectory::new(); let other = directory.root.join("other"); fs::create_dir(&other).unwrap(); - let files = RegionFiles::new( + let paths = RegionPaths::new( directory.root.join("data"), other.join("state"), directory.root.join("image"), ); let opened = RegionStore::open( 8, - FileRegionBackend::for_test(files, test_data_superblock(), 8), + FileRegionBackend::for_test(paths, test_data_superblock(), 8), ); assert!(matches!( opened, @@ -1976,13 +1965,13 @@ fn state_sidecar_lock_prevents_cross_data_file_races() { let data = test_data_superblock(); let mut first = RegionStore::open( config, - FileRegionBackend::for_test(directory.files.clone(), data, config), + FileRegionBackend::for_test(directory.paths.clone(), data, config), ) .unwrap(); - let conflicting_files = RegionFiles::new( + let conflicting_files = RegionPaths::new( directory.root.join("other-data"), - directory.files.state.clone(), + directory.paths.state.clone(), directory.root.join("other-image"), ); let opened = RegionStore::open( diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/region/runtime/mod.rs index 950df5f..1937d73 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/region/runtime/mod.rs @@ -748,7 +748,7 @@ impl RegionDataPlane { pub fn new( core: Arc, data: DataSuperblock, - files: DataFileHandles, + handles: DataFileHandles, config: CacheConfig, ) -> io::Result { // Recovery supplies independently validated metadata. It must still @@ -772,7 +772,7 @@ impl RegionDataPlane { let running = start_running( Arc::clone(&core), data, - files, + handles, config, Arc::clone(&metrics), Arc::clone(&operations), @@ -1401,7 +1401,7 @@ fn add_io_direction(aggregate: &mut CacheIoDirectionSnapshot, snapshot: CacheIoD fn start_running( core: Arc, data: DataSuperblock, - files: DataFileHandles, + handles: DataFileHandles, config: CacheConfig, metrics: Arc, operations: Arc, @@ -1458,11 +1458,11 @@ fn start_running( })?, ); } - let reclaim_files = files.try_clone()?; - let write_files = files.try_clone()?; + let reclaim_handles = handles.try_clone()?; + let write_handles = handles.try_clone()?; let read_wait_enabled = !read_io_wait_timeout(runtime).is_zero(); let read_engines = build_engine_pool( - files, + handles, runtime, IoPoolTopology::read(runtime.io_engine), read_wait_enabled, @@ -1470,13 +1470,13 @@ fn start_running( let read_waiters = read_wait_enabled.then(|| Arc::new(Semaphore::new(read_io_wait_capacity(runtime)))); let write_engines = build_engine_pool( - write_files, + write_handles, runtime, IoPoolTopology::write(runtime.io_engine), false, )?; let reclaim_engines = build_engine_pool( - reclaim_files, + reclaim_handles, runtime, IoPoolTopology::reclaim(runtime.io_engine), false, @@ -1593,12 +1593,12 @@ fn start_running( } fn build_engine_pool( - files: DataFileHandles, + handles: DataFileHandles, runtime: &RuntimeOptions, topology: IoPoolTopology, read_wait_enabled: bool, ) -> io::Result]>> { - let mut source = Some(files); + let mut source = Some(handles); let engine_count = topology.engine_count(); let mut engines = Vec::new(); engines @@ -2512,7 +2512,7 @@ mod tests { use crate::config::runtime::IoEngineOptions; use crate::config::runtime::PosixIoOptions; use crate::region::file_backend::FileRegionBackend; - use crate::region::file_backend::RegionFiles; + use crate::region::file_backend::RegionPaths; use crate::region::index::packed::IndexEntry; use crate::region::index::packed::PackedLocation; use crate::region::recovery::DATA_REGION_AREA_OFFSET; @@ -2524,7 +2524,7 @@ mod tests { "cache2-completion-timeout-{}-{id}", std::process::id() )); - let files = RegionFiles::new( + let paths = RegionPaths::new( path.with_extension("cache"), path.with_extension("state"), path.with_extension("image"), @@ -2566,7 +2566,7 @@ mod tests { }; let mut store = RegionStore::open( 8, - FileRegionBackend::for_test_with_options(files.clone(), data, 8, config), + FileRegionBackend::for_test_with_options(paths.clone(), data, 8, config), ) .unwrap(); let plane = store.data_plane_handle().unwrap(); @@ -2607,8 +2607,8 @@ mod tests { assert_eq!(snapshot.l2_read_overloads, 1); } } - std::fs::remove_file(files.data).unwrap(); - std::fs::remove_file(files.state).unwrap(); + std::fs::remove_file(paths.data).unwrap(); + std::fs::remove_file(paths.state).unwrap(); } #[test] diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/region/runtime/shutdown_tests.rs index 11d8cd2..d0a3e87 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/region/runtime/shutdown_tests.rs @@ -80,7 +80,7 @@ fn submitted_read_must_not_pin_close() { fn assert_close_does_not_wait_for_read(submit_before_close: bool) { use crate::config::runtime::PosixIoOptions; use crate::region::file_backend::FileRegionBackend; - use crate::region::file_backend::RegionFiles; + use crate::region::file_backend::RegionPaths; use crate::region::recovery::PersistentId; use crate::region::store::RegionStore; let root = env::temp_dir().join(format!( @@ -88,7 +88,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { std::process::id() )); std::fs::create_dir_all(&root).unwrap(); - let files = RegionFiles::new(root.join("data"), root.join("state"), root.join("image")); + let paths = RegionPaths::new(root.join("data"), root.join("state"), root.join("image")); let data = DataSuperblock { generation: 1, cache_uuid: PersistentId::from_bytes([1; 16]).unwrap(), @@ -113,7 +113,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { }; let mut store = RegionStore::open( 8, - FileRegionBackend::for_test_with_options(files, data, 8, config), + FileRegionBackend::for_test_with_options(paths, data, 8, config), ) .unwrap(); let mut plane = store.data_plane_handle().unwrap(); @@ -182,13 +182,13 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { #[test] fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { use crate::region::file_backend::FileRegionBackend; - use crate::region::file_backend::RegionFiles; + use crate::region::file_backend::RegionPaths; use crate::region::recovery::PersistentId; use crate::region::store::RegionStore; use crate::snapshot::CacheHealth; let root = env::temp_dir().join(format!("cache2-recovery-admission-{}", std::process::id())); std::fs::create_dir_all(&root).unwrap(); - let files = RegionFiles::new(root.join("data"), root.join("state"), root.join("image")); + let paths = RegionPaths::new(root.join("data"), root.join("state"), root.join("image")); let data = DataSuperblock { generation: 1, cache_uuid: PersistentId::from_bytes([1; 16]).unwrap(), @@ -208,7 +208,7 @@ fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { }; let mut store = RegionStore::open( 8, - FileRegionBackend::for_test_with_options(files, data, 8, config), + FileRegionBackend::for_test_with_options(paths, data, 8, config), ) .unwrap(); let plane = store.data_plane_handle().unwrap(); diff --git a/cache2/src/region/store.rs b/cache2/src/region/store.rs index 895c79b..6d1b213 100644 --- a/cache2/src/region/store.rs +++ b/cache2/src/region/store.rs @@ -25,23 +25,23 @@ use std::io; +use crate::io::fs::FileSystem; +use crate::io::fs::OsFileSystem; use crate::region::file_backend::FileRegionBackend; use crate::region::file_backend::FileRegionRuntime; -use crate::region::file_backend::RegionFileSystem; -use crate::region::file_backend::SystemRegionFileSystem; use crate::region::index::storage::validated_index_partition_ranges; use crate::region::runtime::RegionDataPlane; use crate::snapshot::StartupMode; /// Owns the files and runtime for one Region-backed cache. -pub struct RegionStore { +pub struct RegionStore { backend: FileRegionBackend, runtime: Option, startup: StartupMode, closed: bool, } -impl RegionStore { +impl RegionStore { pub fn open(index_slots: usize, mut backend: FileRegionBackend) -> io::Result { validate_index_slots(index_slots)?; backend.acquire_exclusive()?; @@ -129,7 +129,7 @@ impl RegionStore { } } -impl Drop for RegionStore { +impl Drop for RegionStore { fn drop(&mut self) { if !self.closed { let _ = self.close_fast(); From 6b4d45f3f190d088b33dcc9fb05b3173c9815457 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:09:21 +0800 Subject: [PATCH 04/12] refactor: distinguish region storage from cache session state Use RegionStore for the live L2 index, region allocation, and record operations. Move the open/close coordinator to cache::session as CacheSession, and update its owning field and call sites. Name per-region records RegionDesc and read descriptions ReadDesc, including their variables and methods. Use RegionState, AppendStaging, AppendShardGate, AppendWorkerControl, and RegionManagerLock for their concrete roles. Replace the single-field region-access wrapper with the same atomic generation array. Validation: 276 library tests and all-target workspace checks. Persistent state discriminants, layouts, synchronization, and public APIs are unchanged. --- cache2/src/cache.rs | 24 +- .../src/{region/store.rs => cache/session.rs} | 10 +- cache2/src/config/runtime.rs | 4 +- cache2/src/region/file_backend/mod.rs | 98 ++++---- cache2/src/region/file_backend/tests.rs | 231 +++++++++--------- cache2/src/region/index/storage/mod.rs | 4 +- cache2/src/region/manager.rs | 86 ++++--- cache2/src/region/mod.rs | 116 ++++----- cache2/src/region/reader.rs | 60 ++--- cache2/src/region/recovery/metadata.rs | 40 ++- cache2/src/region/runtime/mod.rs | 189 +++++++------- cache2/src/region/runtime/shutdown_tests.rs | 8 +- cache2/src/region/staging.rs | 24 +- 13 files changed, 437 insertions(+), 457 deletions(-) rename cache2/src/{region/store.rs => cache/session.rs} (94%) diff --git a/cache2/src/cache.rs b/cache2/src/cache.rs index 1226676..2541aa3 100644 --- a/cache2/src/cache.rs +++ b/cache2/src/cache.rs @@ -37,6 +37,7 @@ use tokio::task::JoinError; use crate::LatencyMode; use crate::RequestOperation; +use crate::cache::session::CacheSession; use crate::config::CacheConfig; use crate::config::storage::KEY_HASH_SEED; use crate::config::storage_fingerprint; @@ -52,11 +53,12 @@ use crate::region::recovery::RECOVERY_IMAGE_INDEX_OFFSET; use crate::region::recovery::recovery_image_index_len; use crate::region::runtime::HybridValueRead; use crate::region::runtime::RegionDataPlane; -use crate::region::store::RegionStore; use crate::snapshot::CacheSnapshot; use crate::snapshot::DetailedCacheSnapshot; use crate::snapshot::StartupMode; +pub mod session; + /// Storage tier that served a lookup. /// /// An L2 value promoted before return still reports L2, even though its owned @@ -119,7 +121,7 @@ pub struct Cache { read_recording: bool, mutation_recording: bool, data_plane: RegionDataPlane, - owner: Arc>, + session: Arc>, startup: StartupMode, path: PathBuf, logical_disk_peak_bytes: u64, @@ -254,15 +256,15 @@ impl Cache { let mutation_recording = stats.request_counters || stats.mutation_latency != LatencyMode::Off; let backend = FileRegionBackend::new(paths, format_data, config); - let store = RegionStore::open(index_slots, backend)?; - let startup = store.startup(); - let data_plane = store.data_plane_handle()?; + let session = CacheSession::open(index_slots, backend)?; + let startup = session.startup(); + let data_plane = session.data_plane_handle()?; Ok(Cache { closed: AtomicBool::new(false), read_recording, mutation_recording, data_plane, - owner: Arc::new(Mutex::new(store)), + session: Arc::new(Mutex::new(session)), startup, path, logical_disk_peak_bytes, @@ -516,18 +518,18 @@ impl Cache { let tokio_handle = self.tokio_handle.clone(); let close = (!self.closed.swap(true, Ordering::AcqRel)).then(|| { self.data_plane.start_close(); - let owner = Arc::clone(&self.owner); + let session = Arc::clone(&self.session); let path = self.path.clone(); let started = Instant::now(); tokio_handle.spawn_blocking(move || { - let result = owner + let result = session .lock() .map_err(|_| cache_lifecycle_poisoned()) - .and_then(|mut store| { + .and_then(|mut session| { if warm { - store.close_warm() + session.close_warm() } else { - store.close_fast() + session.close_fast() } }); log_cache_close(&path, mode, started.elapsed(), &result); diff --git a/cache2/src/region/store.rs b/cache2/src/cache/session.rs similarity index 94% rename from cache2/src/region/store.rs rename to cache2/src/cache/session.rs index 6d1b213..3c03e0d 100644 --- a/cache2/src/region/store.rs +++ b/cache2/src/cache/session.rs @@ -34,14 +34,14 @@ use crate::region::runtime::RegionDataPlane; use crate::snapshot::StartupMode; /// Owns the files and runtime for one Region-backed cache. -pub struct RegionStore { +pub struct CacheSession { backend: FileRegionBackend, runtime: Option, startup: StartupMode, closed: bool, } -impl RegionStore { +impl CacheSession { pub fn open(index_slots: usize, mut backend: FileRegionBackend) -> io::Result { validate_index_slots(index_slots)?; backend.acquire_exclusive()?; @@ -129,7 +129,7 @@ impl RegionStore { } } -impl Drop for RegionStore { +impl Drop for CacheSession { fn drop(&mut self) { if !self.closed { let _ = self.close_fast(); @@ -138,14 +138,14 @@ impl Drop for RegionStore { } fn closed_error() -> io::Error { - io::Error::new(io::ErrorKind::BrokenPipe, "RegionStore is closed") + io::Error::new(io::ErrorKind::BrokenPipe, "CacheSession is closed") } fn validate_index_slots(index_slots: usize) -> io::Result<()> { if index_slots < 8 { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore requires at least 8 index slots", + "CacheSession requires at least 8 index slots", )); } validated_index_partition_ranges(index_slots) diff --git a/cache2/src/config/runtime.rs b/cache2/src/config/runtime.rs index 56a0d90..cab9dde 100644 --- a/cache2/src/config/runtime.rs +++ b/cache2/src/config/runtime.rs @@ -32,7 +32,7 @@ use crate::memory::MemoryStore; use crate::region::recovery::DataGeometry; use crate::region::runtime::metrics::ActivityMetrics; use crate::region::runtime_fixed_memory_bytes; -use crate::region::staging::RegionStaging; +use crate::region::staging::AppendStaging; use crate::stats::recording::Recorder; const DEFAULT_L1_SHARDS: usize = 32; @@ -694,7 +694,7 @@ impl RuntimeOptions { .map_err(|_| invalid_config("Region size does not fit the memory requirements"))?; let chunk_bytes = usable_region; let write_buffer_reservation = - RegionStaging::reservation_bytes(shard_count, chunk_bytes) + AppendStaging::reservation_bytes(shard_count, chunk_bytes) .ok_or_else(|| invalid_config("write buffer memory requirements overflow"))?; let reserved_memory = fixed_bytes .checked_add(self.l1_capacity_bytes) diff --git a/cache2/src/region/file_backend/mod.rs b/cache2/src/region/file_backend/mod.rs index c1d925d..d30b74f 100644 --- a/cache2/src/region/file_backend/mod.rs +++ b/cache2/src/region/file_backend/mod.rs @@ -12,7 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! File ownership, recovery, and lifecycle adapter for the Region core. +//! File ownership, recovery, and lifecycle adapter for the Region regions. use std::fmt; use std::fs::File; @@ -43,11 +43,10 @@ use crate::io::file::write_all_at; use crate::io::fs::FileSystem; use crate::io::fs::OsFileSystem; use crate::io::fs::parent_directory; -use crate::region::FileRegionCore; -use crate::region::RegionAccessState; +use crate::region::AppendShardGate; use crate::region::RegionHealthLatch; -use crate::region::RegionManagerAuthority; -use crate::region::RegionShard; +use crate::region::RegionManagerLock; +use crate::region::RegionStore; use crate::region::guarded_index_result; use crate::region::index::RegionIndex; use crate::region::index::packed::MAX_INDEX_PARTITIONS; @@ -83,7 +82,7 @@ use crate::region::recovery::metadata::RegionMetadata; use crate::region::recovery::metadata::RegionMetadataError; use crate::region::recovery::metadata::RegionMetadataRecord; use crate::region::recovery::metadata::RegionMetadataRoot; -use crate::region::recovery::metadata::RegionMetadataState; +use crate::region::recovery::metadata::RegionState; use crate::region::recovery::prepare_next_state; use crate::region::recovery::prepare_running_barrier; use crate::region::recovery::recovery_image_index_len; @@ -116,19 +115,19 @@ impl RegionPaths { } } pub struct FileRegionRuntime { - core: Arc, + regions: Arc, data_plane: Option, } impl Deref for FileRegionRuntime { - type Target = FileRegionCore; + type Target = RegionStore; fn deref(&self) -> &Self::Target { - &self.core + &self.regions } } pub struct FrozenFileRegionView { - core: Arc, + regions: Arc, metadata: RegionMetadata, } @@ -161,8 +160,8 @@ impl FileRegionRuntime { )); } let manager = RegionManager::from_metadata(metadata).map_err(region_metadata_io_error)?; - let mut shards = Vec::new(); - shards + let mut append_gates = Vec::new(); + append_gates .try_reserve_exact(manager.active_regions().len()) .map_err(|_| { io::Error::new( @@ -170,9 +169,9 @@ impl FileRegionRuntime { "cannot allocate data shard gates", ) })?; - shards.resize_with(manager.active_regions().len(), RegionShard::default); - let mut region_access = Vec::new(); - region_access + append_gates.resize_with(manager.active_regions().len(), AppendShardGate::default); + let mut region_generations = Vec::new(); + region_generations .try_reserve_exact(manager.regions().len()) .map_err(|_| { io::Error::new( @@ -181,18 +180,16 @@ impl FileRegionRuntime { ) })?; for region in manager.regions() { - region_access.push(RegionAccessState { - generation: AtomicU64::new(region.created_seqno), - }); + region_generations.push(AtomicU64::new(region.created_seqno)); } let health = RegionHealthLatch::healthy(); let index = RegionIndex::from_storage(index).map_err(index_storage_io_error)?; Ok(Self { - core: Arc::new(FileRegionCore { + regions: Arc::new(RegionStore { index, - manager: RegionManagerAuthority::new(manager, health.clone()), - shards: shards.into_boxed_slice(), - region_access: region_access.into_boxed_slice(), + manager: RegionManagerLock::new(manager, health.clone()), + append_gates: append_gates.into_boxed_slice(), + region_generations: region_generations.into_boxed_slice(), rotation: Mutex::new(()), health, }), @@ -213,7 +210,7 @@ impl FileRegionRuntime { )); } self.data_plane = Some(RegionDataPlane::new( - Arc::clone(&self.core), + Arc::clone(&self.regions), data, handles, config, @@ -393,7 +390,7 @@ where if self.locked { return Err(io::Error::new( io::ErrorKind::AlreadyExists, - "RegionStore backend is already locked", + "CacheSession backend is already locked", )); } if self.paths.data == self.paths.state @@ -402,7 +399,7 @@ where { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore data/state/image paths must be distinct", + "CacheSession data/state/image paths must be distinct", )); } if parent_directory(&self.paths.data) != parent_directory(&self.paths.state) @@ -410,7 +407,7 @@ where { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore data/state/image files must share one directory", + "CacheSession data/state/image files must share one directory", )); } let temporary = recovery_temporary_path(&self.paths.image); @@ -420,7 +417,7 @@ where { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore recovery temporary path collides with a cache file", + "CacheSession recovery temporary path collides with a cache file", )); } let data = self @@ -448,7 +445,7 @@ where let _ = data.unlock(); return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore data and state paths resolve to the same file", + "CacheSession data and state paths resolve to the same file", )); } if let Err(error) = state.try_lock_exclusive() { @@ -530,7 +527,7 @@ where if image.is_same_file(data_file)? || image.is_same_file(state_file)? { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore image aliases the data or state file", + "CacheSession image aliases the data or state file", )); } @@ -817,24 +814,27 @@ where } } runtime.health.require_healthy()?; - let core = runtime.core; - if core.shards.iter().any(|shard| shard.mutation.is_poisoned()) - || core.rotation.is_poisoned() + let regions = runtime.regions; + if regions + .append_gates + .iter() + .any(|shard| shard.mutation.is_poisoned()) + || regions.rotation.is_poisoned() { - core.health.enter_miss_only(); + regions.health.enter_miss_only(); return Err(io::Error::new( io::ErrorKind::InvalidData, "data shard gate is poisoned", )); } - let partitions = index_partition_metadata(core.index.storage(), &core.health)?; - let metadata = core + let partitions = index_partition_metadata(regions.index.storage(), ®ions.health)?; + let metadata = regions .manager .lock()? .freeze_metadata(partitions) .map_err(region_metadata_io_error)?; - core.health.require_healthy()?; - Ok(FrozenFileRegionView { core, metadata }) + regions.health.require_healthy()?; + Ok(FrozenFileRegionView { regions, metadata }) } /// Make completed data and one complete image durable. @@ -842,13 +842,13 @@ where &mut self, view: &FrozenFileRegionView, ) -> io::Result { - let health = &view.core.health; + let health = &view.regions.health; health.require_healthy()?; let source_metadata = &view.metadata; source_metadata .validate() .map_err(region_metadata_io_error)?; - let storage = view.core.index.storage(); + let storage = view.regions.index.storage(); let physical_stats = guarded_index_result(health, storage.physical_stats())?; let partition_stats = guarded_index_result(health, storage.partition_stats())?; if source_metadata.root.index_slots @@ -1269,12 +1269,12 @@ fn metadata_partition_stats_match(metadata: &RegionMetadata, stats: &[IndexPhysi fn empty_region_metadata( data: DataSuperblock, index_slots: usize, - shards: u32, + append_gates: u32, ) -> io::Result { - if shards == 0 || data.geometry.region_count <= shards { + if append_gates == 0 || data.geometry.region_count <= append_gates { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "RegionStore requires one Active Region per shard plus one spare", + "CacheSession requires one Active Region per shard plus one spare", )); } let partition_ranges = @@ -1292,7 +1292,7 @@ fn empty_region_metadata( .map_err(|_| io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate Region table"))?; let mut free_ordinal = 0_u32; for region_id in 0..data.geometry.region_count { - let active = region_id < shards; + let active = region_id < append_gates; let queue_ordinal = if active { region_id } else { @@ -1304,9 +1304,9 @@ fn empty_region_metadata( }; regions.push(RegionMetadataRecord { state: if active { - RegionMetadataState::Active + RegionState::Active } else { - RegionMetadataState::Free + RegionState::Free }, queue_ordinal, created_seqno: if active { u64::from(region_id) + 1 } else { 0 }, @@ -1330,10 +1330,10 @@ fn empty_region_metadata( partition_count: u32::try_from(partition_ranges.len()).map_err(|_| { io::Error::new(io::ErrorKind::InvalidInput, "too many index partitions") })?, - shard_count: shards, - max_seqno: u64::from(shards), - free_region_count: data.geometry.region_count - shards, - active_region_count: shards, + shard_count: append_gates, + max_seqno: u64::from(append_gates), + free_region_count: data.geometry.region_count - append_gates, + active_region_count: append_gates, sealed_region_count: 0, }, regions: regions.into_boxed_slice(), diff --git a/cache2/src/region/file_backend/tests.rs b/cache2/src/region/file_backend/tests.rs index 93ad1fa..b49fa61 100644 --- a/cache2/src/region/file_backend/tests.rs +++ b/cache2/src/region/file_backend/tests.rs @@ -35,6 +35,7 @@ use std::time::Instant; use super::*; use crate::IoEngineOptions; use crate::StatsOptions; +use crate::cache::session::CacheSession; use crate::config::runtime::MAX_WRITE_FLUSH_THRESHOLD_BYTES; use crate::config::runtime::PosixIoOptions; use crate::config::runtime::ReadAdmission; @@ -55,7 +56,7 @@ use crate::region::index::storage::IndexSlotState; use crate::region::index::storage::page_format::INDEX_IMAGE_SLOTS_PER_PAGE; use crate::region::reader::ReadCandidate; use crate::region::reader::ReadCompletion; -use crate::region::reader::ReadDescriptor; +use crate::region::reader::ReadDesc; use crate::region::reader::describe_read; use crate::region::record::RECORD_ALIGNMENT; use crate::region::record::codec::hash_key; @@ -64,14 +65,13 @@ use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; use crate::region::recovery::PersistentId; use crate::region::runtime::HybridValueRead; -use crate::region::staging::RegionStaging; +use crate::region::staging::AppendStaging; use crate::region::staging::StagedRecord; -use crate::region::store::RegionStore; use crate::snapshot::CacheSnapshot; use crate::snapshot::DetailedCacheSnapshot; use crate::snapshot::StartupMode; -impl RegionStore { +impl CacheSession { fn put_value(&self, key: &[u8], value: &[u8]) -> io::Result { self.runtime()?.data_plane()?.put(key, value) } @@ -327,7 +327,7 @@ fn external_process_kill_recovery_contract() { ] { let directory = TestDirectory::new(); let data = data_path_superblock(); - let mut initial = RegionStore::open( + let mut initial = CacheSession::open( 4096, FileRegionBackend::for_test(directory.paths.clone(), data, 4096), ) @@ -353,7 +353,7 @@ fn external_process_kill_recovery_contract() { "crash case {case} did not SIGKILL" ); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( 4096, FileRegionBackend::for_test(directory.paths.clone(), data, 4096), ) @@ -379,21 +379,21 @@ fn run_crash_child(case: &str, paths: RegionPaths) -> ! { match case { "open" => { let _store = - RegionStore::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); + CacheSession::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); kill_process(); } "write" | "drain" => { - let store = - RegionStore::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); - eventually_admitted(|| store.put_value(b"replacement", b"new")); + let session = + CacheSession::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); + eventually_admitted(|| session.put_value(b"replacement", b"new")); if case == "drain" { - store.drain().unwrap(); + session.drain().unwrap(); } kill_process(); } "warm-data" | "warm-image" | "clean-state" => { let (file_system, faults, _) = FaultFileSystem::new(); - let mut store = RegionStore::open( + let mut session = CacheSession::open( 4096, FileRegionBackend::for_test_with_file_system(paths, data, 4096, file_system), ) @@ -405,7 +405,7 @@ fn run_crash_child(case: &str, paths: RegionPaths) -> ! { _ => unreachable!(), }; faults.arm(FaultEvent::Sync(point), 1, FaultAction::KillAfter); - let _ = store.close_warm(); + let _ = session.close_warm(); panic!("crash fault did not terminate the child"); } _ => panic!("unknown crash child case: {case}"), @@ -459,7 +459,7 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { }, ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut session = CacheSession::open( 4096, FileRegionBackend::for_test_with_options( directory.paths.clone(), @@ -474,12 +474,13 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { .enable_time() .build() .unwrap(); - eventually_admitted(|| store.put_value(b"queued-read", b"local-value")); - store.drain().unwrap(); - let plane = store.data_plane_handle().unwrap(); + eventually_admitted(|| session.put_value(b"queued-read", b"local-value")); + session.drain().unwrap(); + let plane = session.data_plane_handle().unwrap(); let mut slots: Vec<_> = (0..2).map(|_| plane.reserve_read_slot_for_test()).collect(); tokio_runtime.block_on(async { - let mut cancelled = Box::pin(store.get_value_async(b"queued-read", tokio_runtime.handle())); + let mut cancelled = + Box::pin(session.get_value_async(b"queued-read", tokio_runtime.handle())); assert_pending( cancelled.as_mut(), "saturated read must enter the wait queue", @@ -488,7 +489,7 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { drop(cancelled); }); let value = tokio_runtime.block_on(async { - let mut waiting = Box::pin(store.get_value_async(b"queued-read", tokio_runtime.handle())); + let mut waiting = Box::pin(session.get_value_async(b"queued-read", tokio_runtime.handle())); assert_pending( waiting.as_mut(), "a cancelled read must release its wait-queue permit", @@ -502,13 +503,13 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { drop(value); let blocked: Vec<_> = (0..2).map(|_| plane.reserve_read_slot_for_test()).collect(); tokio_runtime.block_on(async { - let mut waiting = Box::pin(store.get_value_async(b"queued-read", tokio_runtime.handle())); + let mut waiting = Box::pin(session.get_value_async(b"queued-read", tokio_runtime.handle())); assert_pending( waiting.as_mut(), "saturated read must enter the bounded wait queue", ) .await; - let queue_full = match store + let queue_full = match session .get_value_async(b"queued-read", tokio_runtime.handle()) .await { @@ -523,10 +524,10 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { assert_eq!(timed_out.kind(), io::ErrorKind::TimedOut); }); drop(blocked); - let snapshot = store.snapshot().unwrap(); + let snapshot = session.snapshot().unwrap(); assert_eq!(snapshot.l2_read_overloads, 2); assert!(snapshot.l2_read_wait_ns > 0); - store.close_fast().unwrap(); + session.close_fast().unwrap(); } #[test] @@ -546,7 +547,7 @@ fn queued_l2_read_does_not_pin_warm_close() { }, ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut session = CacheSession::open( 4096, FileRegionBackend::for_test_with_options( directory.paths.clone(), @@ -560,16 +561,16 @@ fn queued_l2_read_does_not_pin_warm_close() { .enable_time() .build() .unwrap(); - eventually_admitted(|| store.put_value(b"queued-close", b"value")); - store.drain().unwrap(); - let plane = store.data_plane_handle().unwrap(); + eventually_admitted(|| session.put_value(b"queued-close", b"value")); + session.drain().unwrap(); + let plane = session.data_plane_handle().unwrap(); let slot = plane.reserve_read_slot_for_test(); tokio_runtime.block_on(async { let mut waiting = Box::pin(plane.get_async(b"queued-close", tokio_runtime.handle(), None)); assert_pending(waiting.as_mut(), "saturated read must enter the wait queue").await; - store.close_warm().unwrap(); + session.close_warm().unwrap(); drop(slot); let error = match waiting.await { Ok(_) => panic!("queued read must stop after warm close"), @@ -578,7 +579,7 @@ fn queued_l2_read_does_not_pin_warm_close() { assert_eq!(error.kind(), io::ErrorKind::BrokenPipe); }); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( 4096, FileRegionBackend::for_test_with_options( directory.paths.clone(), @@ -604,7 +605,7 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { }, ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut session = CacheSession::open( 4096, FileRegionBackend::for_test_with_options( directory.paths.clone(), @@ -620,13 +621,13 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { for (ordinal, size) in mixed.into_iter().enumerate() { let key = key_for_shard(data, ordinal as u64, ordinal as u64); let value = vec![(ordinal as u8) + 1; size]; - eventually_admitted(|| store.put_value(&key, &value)); + eventually_admitted(|| session.put_value(&key, &value)); expected.push((key, value)); } - store.drain().unwrap(); + session.drain().unwrap(); for (key, value) in &expected { - assert_eq!(store.get_value(key).unwrap().unwrap().value(), value); - assert_eq!(store.get_value(key).unwrap().unwrap().value(), value); + assert_eq!(session.get_value(key).unwrap().unwrap().value(), value); + assert_eq!(session.get_value(key).unwrap().unwrap().value(), value); } // A 256 KiB record leaves insufficient room for another same-shard @@ -635,19 +636,19 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { // Region, only the newest same-shard record is guaranteed to survive. let rotation_value = vec![0xa5; 256 * 1024]; let mut recent = Vec::new(); - let rotations_before = store.detailed_snapshot().unwrap().region.rotations; + let rotations_before = session.detailed_snapshot().unwrap().region.rotations; for ordinal in 0..32 { let key = key_for_shard(data, 0, 100 + ordinal); - eventually_admitted(|| store.put_value(&key, &rotation_value)); - store.drain().unwrap(); + eventually_admitted(|| session.put_value(&key, &rotation_value)); + session.drain().unwrap(); recent.push(key); } assert_eq!( - store.detailed_snapshot().unwrap().region.rotations - rotations_before, + session.detailed_snapshot().unwrap().region.rotations - rotations_before, 31, "one full-Region signal must cause exactly one rotation" ); - let reclaim = store.detailed_snapshot().unwrap().summary.reclaim; + let reclaim = session.detailed_snapshot().unwrap().summary.reclaim; // The explicit drain may fence the optional reinsertion before its // reclaimer enters the mutation gate. Heat classification must still // offer the candidate, while either admission or a bounded skip is a @@ -660,13 +661,13 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { assert!(reclaim.reinsert_bytes.saturating_mul(8) <= reclaim.bytes_read); for key in recent.iter().rev().take(1) { assert_eq!( - store.get_value(key).unwrap().unwrap().value(), + session.get_value(key).unwrap().unwrap().value(), rotation_value ); } let retained_hits: Vec<_> = (0..129) - .map(|_| store.get_value(recent.last().unwrap()).unwrap().unwrap()) + .map(|_| session.get_value(recent.last().unwrap()).unwrap().unwrap()) .collect(); assert!( retained_hits @@ -674,16 +675,16 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { .all(|hit| hit.value() == rotation_value) ); assert!( - store.get_value(b"definite-index-miss").unwrap().is_none(), + session.get_value(b"definite-index-miss").unwrap().is_none(), "an index miss must not acquire a retained-hit buffer" ); // Retained zero-copy hits own their transient aligned allocations, but // cannot pin the runtime operation barrier or prevent a warm shutdown. - store.close_warm().unwrap(); + session.close_warm().unwrap(); assert_eq!(retained_hits[0].value(), rotation_value); drop(retained_hits); - let mut recovered = RegionStore::open( + let mut recovered = CacheSession::open( 4096, FileRegionBackend::for_test(directory.paths.clone(), data, 4096), ) @@ -716,7 +717,7 @@ fn poisoned_runtime_gates_stop_workers_and_reject_warm_close() { write_flush_threshold_bytes: 128 * 1024, ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut session = CacheSession::open( 4096, FileRegionBackend::for_test_with_options( directory.paths.clone(), @@ -727,20 +728,20 @@ fn poisoned_runtime_gates_stop_workers_and_reject_warm_close() { ) .unwrap(); - let runtime = store.runtime().unwrap(); + let runtime = session.runtime().unwrap(); match case { "shard" => runtime.data_plane().unwrap().poison_shard_for_test(0), "index" => runtime - .core + .regions .index .storage() .poison_hash_partition_for_test(0), _ => unreachable!(), } - let error = store.close_warm().unwrap_err(); + let error = session.close_warm().unwrap_err(); assert_eq!(error.kind(), io::ErrorKind::InvalidData, "{case}"); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( 4096, FileRegionBackend::for_test_with_options( directory.paths.clone(), @@ -767,12 +768,12 @@ fn fresh_metadata_assigns_four_active_shards_and_a_free_victim() { assert_eq!(metadata.root.max_seqno, u64::from(REGION_SHARDS)); for shard_id in 0..REGION_SHARDS { let region = metadata.regions[usize::try_from(shard_id).unwrap()]; - assert_eq!(region.state, RegionMetadataState::Active); + assert_eq!(region.state, RegionState::Active); assert_eq!(region.queue_ordinal, shard_id); assert_eq!(region.created_seqno, u64::from(shard_id) + 1); } let free = metadata.regions[shards]; - assert_eq!(free.state, RegionMetadataState::Free); + assert_eq!(free.state, RegionState::Free); assert_eq!(free.queue_ordinal, 0); let manager = RegionManager::from_metadata(metadata).unwrap(); @@ -809,7 +810,7 @@ fn four_tib_region_metadata_round_trips_into_runtime_authority() { assert_eq!(manager.next_seqno(), u64::from(REGION_SHARDS) + 1); } -fn foreground_stage_fixture() -> (DataSuperblock, FileRegionRuntime, RegionStaging) { +fn foreground_stage_fixture() -> (DataSuperblock, FileRegionRuntime, AppendStaging) { let data = data_path_superblock(); let runtime = FileRegionRuntime::install( PartitionedIndexStorage::anonymous(64).unwrap(), @@ -817,7 +818,7 @@ fn foreground_stage_fixture() -> (DataSuperblock, FileRegionRuntime, RegionStagi ) .unwrap(); let managed_memory = data_path_memory(); - let staging = RegionStaging::try_new( + let staging = AppendStaging::try_new( 1, MAX_WRITE_FLUSH_THRESHOLD_BYTES, data.geometry.region_size, @@ -830,7 +831,7 @@ fn foreground_stage_fixture() -> (DataSuperblock, FileRegionRuntime, RegionStagi #[test] fn foreground_stage_bypasses_busy_manager_without_consuming_a_sequence() { let (data, runtime, staging) = foreground_stage_fixture(); - let manager = runtime.manager.inner.lock().unwrap(); + let manager = runtime.manager.state.lock().unwrap(); let next_seqno = manager.next_seqno(); let hash = hash_key(data.hash_seed, b"key"); let record_bytes = required_record_bytes(b"key".len(), b"value".len()).unwrap(); @@ -847,14 +848,14 @@ fn foreground_stage_bypasses_busy_manager_without_consuming_a_sequence() { #[test] fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { let (data, runtime, staging) = foreground_stage_fixture(); - let next_seqno = runtime.manager.inner.lock().unwrap().next_seqno(); - let mutation = runtime.core.shards[0].mutation.lock().unwrap(); + let next_seqno = runtime.manager.state.lock().unwrap().next_seqno(); + let mutation = runtime.regions.append_gates[0].mutation.lock().unwrap(); let hash = hash_key(data.hash_seed, b"key"); let record_bytes = required_record_bytes(b"key".len(), b"value".len()).unwrap(); let (sender, receiver) = mpsc::sync_channel(1); - let core = Arc::clone(&runtime.core); + let regions = Arc::clone(&runtime.regions); let writer = std::thread::spawn(move || { - let result = core.try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value"); + let result = regions.try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value"); sender.send((result, staging)).unwrap(); }); @@ -864,13 +865,13 @@ fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { let (result, staging) = early.expect("a busy shard must reject without waiting for its lock"); assert!(matches!(result.unwrap(), RegionStageValue::NeedsProgress)); assert_eq!( - runtime.manager.inner.lock().unwrap().next_seqno(), + runtime.manager.state.lock().unwrap().next_seqno(), next_seqno ); assert!(staging.shard_fill_snapshot(0).unwrap().is_none()); let staged = runtime - .core + .regions .try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value") .unwrap(); assert!(matches!( @@ -878,7 +879,7 @@ fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { RegionStageValue::Staged { seqno, .. } if seqno == next_seqno )); assert_eq!( - runtime.manager.inner.lock().unwrap().next_seqno(), + runtime.manager.state.lock().unwrap().next_seqno(), next_seqno + 1 ); } @@ -891,8 +892,8 @@ fn completed_record_publication_does_not_enter_region_manager() { empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) .unwrap(); - let core = Arc::clone(&runtime.core); - let manager = core.manager.inner.lock().unwrap(); + let regions = Arc::clone(&runtime.regions); + let manager = regions.manager.state.lock().unwrap(); let record = StagedRecord::new( 7, IndexEntry { @@ -901,7 +902,7 @@ fn completed_record_publication_does_not_enter_region_manager() { 1, ); let (sender, receiver) = mpsc::sync_channel(1); - let publisher_core = Arc::clone(&core); + let publisher_core = Arc::clone(®ions); let publisher = std::thread::spawn(move || { sender .send(publisher_core.publish_completed_records(&[record])) @@ -925,7 +926,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { ) .unwrap(); let managed_memory = data_path_memory(); - let staging = RegionStaging::try_new( + let staging = AppendStaging::try_new( 1, MAX_WRITE_FLUSH_THRESHOLD_BYTES, data.geometry.region_size, @@ -1061,7 +1062,7 @@ fn read_availability_errors_do_not_latch_miss_only() { io::ErrorKind::BrokenPipe, ] { let completion = ReadCompletion { - descriptor: ReadDescriptor { + desc: ReadDesc { hash: 7, entry, region_generation: 1, @@ -1089,7 +1090,7 @@ fn same_hash_candidate_requires_full_key() { ) .unwrap(); let managed_memory = data_path_memory(); - let staging = RegionStaging::try_new( + let staging = AppendStaging::try_new( 1, MAX_WRITE_FLUSH_THRESHOLD_BYTES, data.geometry.region_size, @@ -1138,17 +1139,17 @@ fn same_hash_candidate_requires_full_key() { let entry = runtime .begin_point_read(owner_hash) .expect("hash lookup must return the collision candidate"); - let descriptor = describe_read(data.geometry, owner_hash, entry, true).unwrap(); + let desc = describe_read(data.geometry, owner_hash, entry, true).unwrap(); // Supplying a different key after the hash lookup precisely models a // 64-bit collision at the L2 record-validation boundary. assert!( runtime - .read_value_from_descriptor( + .read_value_from_desc( &engine, engine.try_reserve_read().unwrap(), read_buffer, - descriptor, + desc, foreign_key, ) .unwrap() @@ -1164,15 +1165,14 @@ fn same_hash_candidate_requires_full_key() { region_generation: current.region_generation + 1, ..current }; - let stale_descriptor = - describe_read(data.geometry, owner_hash, stale_generation, true).unwrap(); + let stale_desc = describe_read(data.geometry, owner_hash, stale_generation, true).unwrap(); assert!( runtime - .read_value_from_descriptor( + .read_value_from_desc( &engine, engine.try_reserve_read().unwrap(), managed_memory.try_read_buffer(read_buffer_bytes).unwrap(), - stale_descriptor, + stale_desc, owner_key, ) .unwrap() @@ -1191,18 +1191,17 @@ fn same_hash_candidate_requires_full_key() { }, ..current }; - let wrong_length_descriptor = - describe_read(data.geometry, owner_hash, wrong_length, true).unwrap(); - let wrong_length_read_bytes = wrong_length_descriptor.read_len; + let wrong_length_desc = describe_read(data.geometry, owner_hash, wrong_length, true).unwrap(); + let wrong_length_read_bytes = wrong_length_desc.read_len; assert!( runtime - .read_value_from_descriptor( + .read_value_from_desc( &engine, engine.try_reserve_read().unwrap(), managed_memory .try_read_buffer(wrong_length_read_bytes) .unwrap(), - wrong_length_descriptor, + wrong_length_desc, owner_key, ) .unwrap() @@ -1234,7 +1233,7 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { ) .unwrap(); let managed_memory = data_path_memory(); - let staging = RegionStaging::try_new( + let staging = AppendStaging::try_new( 1, MAX_WRITE_FLUSH_THRESHOLD_BYTES, data.geometry.region_size, @@ -1277,7 +1276,7 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { assert_eq!(runtime.lookup_snapshot(hash).unwrap(), None); assert_eq!(runtime.index.lookup_raw(hash).unwrap(), None); assert_eq!( - runtime.manager.inner.lock().unwrap().regions()[0].completed_used, + runtime.manager.state.lock().unwrap().regions()[0].completed_used, 0 ); engine.shutdown().unwrap(); @@ -1293,7 +1292,7 @@ fn rotation_is_committed_without_a_metadata_io_boundary() { .unwrap(); runtime .manager - .inner + .state .lock() .unwrap() .request_rotation_for_test(0) @@ -1326,7 +1325,7 @@ fn fresh_and_dirty_startup_do_not_write_runtime_region_metadata() { let data = test_data_superblock(); let (fresh_file_system, fresh_io, _) = FaultFileSystem::new(); - let mut fresh = RegionStore::open( + let mut fresh = CacheSession::open( config, FileRegionBackend::for_test_with_file_system( directory.paths.clone(), @@ -1341,7 +1340,7 @@ fn fresh_and_dirty_startup_do_not_write_runtime_region_metadata() { fresh.close_fast().unwrap(); let (dirty_file_system, dirty_io, _) = FaultFileSystem::new(); - let mut dirty = RegionStore::open( + let mut dirty = CacheSession::open( config, FileRegionBackend::for_test_with_file_system( directory.paths.clone(), @@ -1397,7 +1396,7 @@ fn clean_image_rebinds_a_different_append_shard_topology() { let metadata = empty_region_metadata(data, config, 1).unwrap(); publish_custom_clean_image(&directory, config, data, metadata); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1409,7 +1408,7 @@ fn clean_image_rebinds_a_different_append_shard_topology() { drop(manager); reopened.close_warm().unwrap(); - let mut stable = RegionStore::open( + let mut stable = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1435,7 +1434,7 @@ fn append_shard_growth_without_free_regions_cold_starts_safely() { let data = test_data_superblock(); let mut metadata = empty_region_metadata(data, config, 1).unwrap(); for (ordinal, region) in metadata.regions[1..4].iter_mut().enumerate() { - region.state = RegionMetadataState::Sealed; + region.state = RegionState::Sealed; region.queue_ordinal = u32::try_from(ordinal).unwrap(); region.created_seqno = u64::try_from(ordinal + 2).unwrap(); } @@ -1446,7 +1445,7 @@ fn append_shard_growth_without_free_regions_cold_starts_safely() { metadata.validate().unwrap(); publish_custom_clean_image(&directory, config, data, metadata); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1466,7 +1465,7 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let mut first = RegionStore::open( + let mut first = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1482,7 +1481,7 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { file.write_all_at(b"stale-record", stale_offset).unwrap(); file.sync_data().unwrap(); - let mut cold = RegionStore::open( + let mut cold = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1501,7 +1500,7 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { fn invalid_capacity_is_rejected_before_creating_files() { for index_slots in [0, 1, 7, usize::MAX] { let directory = TestDirectory::new(); - let opened = RegionStore::open( + let opened = CacheSession::open( index_slots, FileRegionBackend::for_test(directory.paths.clone(), test_data_superblock(), 8), ); @@ -1514,14 +1513,14 @@ fn invalid_capacity_is_rejected_before_creating_files() { fn dropping_a_warm_runtime_releases_ownership_and_reopens_cold() { let directory = TestDirectory::new(); let data = test_data_superblock(); - let mut initial = RegionStore::open( + let mut initial = CacheSession::open( 8, FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) .unwrap(); initial.close_warm().unwrap(); - let recovered = RegionStore::open( + let recovered = CacheSession::open( 8, FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) @@ -1529,7 +1528,7 @@ fn dropping_a_warm_runtime_releases_ownership_and_reopens_cold() { assert_eq!(recovered.startup(), StartupMode::Warm); drop(recovered); - let mut cold = RegionStore::open( + let mut cold = CacheSession::open( 8, FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) @@ -1551,7 +1550,7 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { }, }); - let mut first = RegionStore::open( + let mut first = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1573,7 +1572,7 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { ); drop(first); - let mut recovered = RegionStore::open( + let mut recovered = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1599,7 +1598,7 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { let directory = TestDirectory::new(); let config = 130; let data = test_data_superblock_with_regions(REGION_SHARDS + 1); - let mut first = RegionStore::open( + let mut first = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1621,7 +1620,7 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { .unwrap(); image.sync_data().unwrap(); - let mut rejected = RegionStore::open( + let mut rejected = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1641,7 +1640,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { let directory = TestDirectory::new(); let config = INDEX_IMAGE_SLOTS_PER_PAGE + 8; let data = test_data_superblock_with_regions(REGION_SHARDS + 1); - let mut first = RegionStore::open( + let mut first = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1661,7 +1660,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { .unwrap(); image.sync_data().unwrap(); - let mut recovered = RegionStore::open( + let mut recovered = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1676,7 +1675,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { assert_eq!(runtime.lookup_snapshot(0).unwrap(), None); assert!(recovered.close_warm().is_err()); - let mut cold = RegionStore::open( + let mut cold = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1753,16 +1752,16 @@ fn every_prepublication_failure_leaves_no_selectable_clean_state() { config, file_system, ); - let mut store = RegionStore::open(config, backend).unwrap(); + let mut session = CacheSession::open(config, backend).unwrap(); if let Some((event, action)) = io_fault { io_faults.arm(event, 1, action); } if let Some(fault) = file_system_fault { file_system_faults.arm(fault); } - assert!(store.close_warm().is_err(), "failure case {case}"); + assert!(session.close_warm().is_err(), "failure case {case}"); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1801,7 +1800,7 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { let data = test_data_superblock(); let (file_system, faults, _) = FaultFileSystem::new(); faults.arm(event, occurrence, action); - let opened = RegionStore::open( + let opened = CacheSession::open( config, FileRegionBackend::for_test_with_file_system( directory.paths.clone(), @@ -1816,7 +1815,7 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { "runtime paths must not be requested before RUNNING is durable" ); - let mut cold = RegionStore::open( + let mut cold = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1832,7 +1831,7 @@ fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { let data = test_data_superblock(); let (file_system, faults, _) = FaultFileSystem::new(); faults.arm(FaultEvent::CloneDataHandles, 1, FaultAction::Error(5)); - let opened = RegionStore::open( + let opened = CacheSession::open( 8, FileRegionBackend::for_test_with_file_system(directory.paths.clone(), data, 8, file_system), ); @@ -1844,7 +1843,7 @@ fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { .contains(&FaultEvent::Sync(SyncPoint::RunningState)) ); - let mut cold = RegionStore::open( + let mut cold = CacheSession::open( 8, FileRegionBackend::for_test(directory.paths.clone(), data, 8), ) @@ -1859,7 +1858,7 @@ fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { let config = 8; let data = test_data_superblock(); let (file_system, faults, _) = FaultFileSystem::new(); - let mut store = RegionStore::open( + let mut session = CacheSession::open( config, FileRegionBackend::for_test_with_file_system( directory.paths.clone(), @@ -1874,9 +1873,9 @@ fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { 1, FaultAction::Error(5), ); - assert!(store.close_warm().is_err()); + assert!(session.close_warm().is_err()); - let mut reopened = RegionStore::open( + let mut reopened = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1901,7 +1900,7 @@ fn data_and_state_inode_alias_is_rejected_without_truncation() { fs::write(&directory.paths.data, marker).unwrap(); fs::hard_link(&directory.paths.data, &directory.paths.state).unwrap(); - let opened = RegionStore::open( + let opened = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ); @@ -1921,7 +1920,7 @@ fn recovery_temporary_path_cannot_name_the_data_or_state_file() { fs::write(&data_path, marker).unwrap(); let paths = RegionPaths::new(&data_path, directory.root.join("state"), image); - let opened = RegionStore::open( + let opened = CacheSession::open( 8, FileRegionBackend::for_test( paths, @@ -1946,7 +1945,7 @@ fn recovery_sidecars_must_share_one_directory() { other.join("state"), directory.root.join("image"), ); - let opened = RegionStore::open( + let opened = CacheSession::open( 8, FileRegionBackend::for_test(paths, test_data_superblock(), 8), ); @@ -1963,7 +1962,7 @@ fn state_sidecar_lock_prevents_cross_data_file_races() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let mut first = RegionStore::open( + let mut first = CacheSession::open( config, FileRegionBackend::for_test(directory.paths.clone(), data, config), ) @@ -1974,7 +1973,7 @@ fn state_sidecar_lock_prevents_cross_data_file_races() { directory.paths.state.clone(), directory.root.join("other-image"), ); - let opened = RegionStore::open( + let opened = CacheSession::open( config, FileRegionBackend::for_test(conflicting_files, data, config), ); diff --git a/cache2/src/region/index/storage/mod.rs b/cache2/src/region/index/storage/mod.rs index 0f95e46..6a4e059 100644 --- a/cache2/src/region/index/storage/mod.rs +++ b/cache2/src/region/index/storage/mod.rs @@ -1778,7 +1778,7 @@ impl Mapping { Self::map(length, libc::MAP_PRIVATE, file.as_raw_fd()) } - fn map(length: usize, flags: i32, descriptor: i32) -> io::Result { + fn map(length: usize, flags: i32, fd: i32) -> io::Result { // SAFETY: the requested range is non-zero and bounded by `isize::MAX`. // Anonymous mappings ignore descriptor; file mappings use offset zero // after the caller verifies the file range. @@ -1788,7 +1788,7 @@ impl Mapping { length, libc::PROT_READ | libc::PROT_WRITE, flags, - descriptor, + fd, 0, ) }; diff --git a/cache2/src/region/manager.rs b/cache2/src/region/manager.rs index 592acd4..b08a112 100644 --- a/cache2/src/region/manager.rs +++ b/cache2/src/region/manager.rs @@ -29,7 +29,7 @@ use crate::region::recovery::metadata::RegionMetadata; use crate::region::recovery::metadata::RegionMetadataError; use crate::region::recovery::metadata::RegionMetadataRecord; use crate::region::recovery::metadata::RegionMetadataRoot; -use crate::region::recovery::metadata::RegionMetadataState; +use crate::region::recovery::metadata::RegionState; use crate::snapshot::RegionSnapshot; const UNASSIGNED_REGION: u32 = u32::MAX; @@ -184,8 +184,8 @@ impl ShardMutation { } #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub struct RegionRuntime { - pub state: RegionMetadataState, +pub struct RegionDesc { + pub state: RegionState, pub created_seqno: u64, /// Last byte covered by a successful write completion. A buffered or /// io_uring CQE is not a durability barrier; CLEAN later syncs this prefix. @@ -201,7 +201,7 @@ pub struct RegionManager { binding: RegionMetadataBinding, region_size: u64, next_seqno: u64, - regions: Vec, + regions: Vec, active_regions: Vec, shard_mutations: Vec, free_regions: VecDeque, @@ -249,7 +249,7 @@ impl RegionManager { for (region_id, encoded) in encoded_regions.iter().copied().enumerate() { let region_id = u32::try_from(region_id).map_err(|_| RegionMetadataError::ArithmeticOverflow)?; - let runtime = RegionRuntime { + let desc = RegionDesc { state: encoded.state, created_seqno: encoded.created_seqno, completed_used: encoded.durable_used_offset, @@ -257,14 +257,14 @@ impl RegionManager { physical_record_count: encoded.physical_record_count, }; install_recovered_queue_entry( - runtime.state, + desc.state, encoded.queue_ordinal, region_id, &mut active_regions, &mut free_regions, &mut sealed_regions, )?; - regions.push(runtime); + regions.push(desc); } if regions.len() != region_count || active_regions.contains(&UNASSIGNED_REGION) @@ -311,7 +311,7 @@ impl RegionManager { self.next_seqno } - pub fn regions(&self) -> &[RegionRuntime] { + pub fn regions(&self) -> &[RegionDesc] { &self.regions } @@ -431,7 +431,7 @@ impl RegionManager { .ok_or(RegionMutationError::Invariant( "active Region id is out of bounds", ))?; - if region.state != RegionMetadataState::Active { + if region.state != RegionState::Active { return Err(RegionMutationError::Invariant( "data shard does not own an Active Region", )); @@ -567,7 +567,7 @@ impl RegionManager { "open write span Region is out of bounds", ))?; if self.active_regions.get(shard_id) != Some(&open.region_id) - || region.state != RegionMetadataState::Active + || region.state != RegionState::Active || open.max_seqno < region.created_seqno || region.reserved_used != open.end_offset { @@ -630,7 +630,7 @@ impl RegionManager { .get(region_index) .ok_or(RegionMutationError::StaleReceipt)?; if self.active_regions.get(receipt.shard_id) != Some(&receipt.region_id) - || region.state != RegionMetadataState::Active + || region.state != RegionState::Active || receipt.seqno < region.created_seqno || region.reserved_used != receipt @@ -709,7 +709,7 @@ impl RegionManager { .get(region_index) .ok_or(RegionMutationError::StaleReceipt)?; if self.active_regions.get(padding.shard_id) != Some(&padding.region_id) - || region.state != RegionMetadataState::Active + || region.state != RegionState::Active || padding.max_seqno < region.created_seqno || region.reserved_used != padding.padded_end_offset { @@ -776,7 +776,7 @@ impl RegionManager { .copied() .ok_or(RegionMutationError::StaleReceipt)?; if self.active_regions.get(receipt.shard_id) != Some(&receipt.region_id) - || region.state != RegionMetadataState::Active + || region.state != RegionState::Active || receipt.max_seqno < region.created_seqno || region.completed_used != receipt.start_offset || receipt.end_offset > region.reserved_used @@ -850,9 +850,9 @@ impl RegionManager { )); } - self.regions[old_index].state = RegionMetadataState::Sealed; - self.regions[victim_index] = RegionRuntime { - state: RegionMetadataState::Active, + self.regions[old_index].state = RegionState::Sealed; + self.regions[victim_index] = RegionDesc { + state: RegionState::Active, created_seqno, completed_used: 0, reserved_used: 0, @@ -901,7 +901,7 @@ impl RegionManager { .ok_or(RegionMutationError::Invariant( "active Region id is out of bounds", ))?; - if old.state != RegionMetadataState::Active || old.reserved_used != old.completed_used { + if old.state != RegionState::Active || old.reserved_used != old.completed_used { return Err(RegionMutationError::WouldBlock); } @@ -919,7 +919,7 @@ impl RegionManager { .ok_or(RegionMutationError::Invariant( "rotation victim id is out of bounds", ))?; - if victim.state != RegionMetadataState::Free || victim_region_id == old_region_id { + if victim.state != RegionState::Free || victim_region_id == old_region_id { return Err(RegionMutationError::Invariant( "rotation victim queue is inconsistent", )); @@ -962,8 +962,8 @@ impl RegionManager { .get(activated_index) .ok_or(RegionMutationError::StaleReceipt)?; if self.active_regions.get(receipt.shard_id) != Some(&receipt.activated_region_id) - || sealed.state != RegionMetadataState::Sealed - || activated.state != RegionMetadataState::Active + || sealed.state != RegionState::Sealed + || activated.state != RegionState::Active || activated.created_seqno != receipt.activated_created_seqno { return Err(RegionMutationError::StaleReceipt); @@ -1006,9 +1006,7 @@ impl RegionManager { .ok_or(RegionMutationError::Invariant( "reclaim Region id is out of bounds", ))?; - if region.state != RegionMetadataState::Sealed - || region.completed_used != region.reserved_used - { + if region.state != RegionState::Sealed || region.completed_used != region.reserved_used { return Err(RegionMutationError::Invariant( "reclaim victim is not a completed sealed Region", )); @@ -1043,7 +1041,7 @@ impl RegionManager { .regions .get_mut(index) .ok_or(RegionMutationError::StaleReceipt)?; - if region.state != RegionMetadataState::Sealed + if region.state != RegionState::Sealed || region.created_seqno != receipt.created_seqno || region.completed_used != receipt.used_offset || region.physical_record_count != receipt.physical_record_count @@ -1060,8 +1058,8 @@ impl RegionManager { "free Region queue exceeded its reserved capacity", )); } - *region = RegionRuntime { - state: RegionMetadataState::Free, + *region = RegionDesc { + state: RegionState::Free, created_seqno: 0, completed_used: 0, reserved_used: 0, @@ -1147,19 +1145,19 @@ impl RegionManager { let mut ordinals = try_unassigned_vec(self.regions.len())?; install_live_queue( &self.regions, - RegionMetadataState::Active, + RegionState::Active, self.active_regions.iter().copied(), &mut ordinals, )?; install_live_queue( &self.regions, - RegionMetadataState::Free, + RegionState::Free, self.free_regions.iter().copied(), &mut ordinals, )?; install_live_queue( &self.regions, - RegionMetadataState::Sealed, + RegionState::Sealed, self.sealed_regions.iter().copied(), &mut ordinals, )?; @@ -1201,7 +1199,7 @@ impl PartitionTotals { } fn install_recovered_queue_entry( - state: RegionMetadataState, + state: RegionState, ordinal: u32, region_id: u32, active: &mut [u32], @@ -1209,9 +1207,9 @@ fn install_recovered_queue_entry( sealed: &mut VecDeque, ) -> Result<(), RegionMetadataError> { let target: &mut [u32] = match state { - RegionMetadataState::Active => active, - RegionMetadataState::Free => free.make_contiguous(), - RegionMetadataState::Sealed => sealed.make_contiguous(), + RegionState::Active => active, + RegionState::Free => free.make_contiguous(), + RegionState::Sealed => sealed.make_contiguous(), }; let ordinal = usize::try_from(ordinal).map_err(|_| RegionMetadataError::ArithmeticOverflow)?; let slot = target @@ -1225,8 +1223,8 @@ fn install_recovered_queue_entry( } fn install_live_queue( - regions: &[RegionRuntime], - expected_state: RegionMetadataState, + regions: &[RegionDesc], + expected_state: RegionState, queue: I, ordinals: &mut [u32], ) -> Result<(), RegionMetadataError> @@ -1325,12 +1323,12 @@ mod tests { sealed_region_count: 2, }, regions: vec![ - region(RegionMetadataState::Active, 1, 2, 0), - region(RegionMetadataState::Free, 1, 0, 0), - region(RegionMetadataState::Sealed, 1, 7, 64), - region(RegionMetadataState::Active, 0, 1, 0), - region(RegionMetadataState::Sealed, 0, 4, 128), - region(RegionMetadataState::Free, 0, 0, 0), + region(RegionState::Active, 1, 2, 0), + region(RegionState::Free, 1, 0, 0), + region(RegionState::Sealed, 1, 7, 64), + region(RegionState::Active, 0, 1, 0), + region(RegionState::Sealed, 0, 4, 128), + region(RegionState::Free, 0, 0, 0), ] .into_boxed_slice(), partitions: shards.into_boxed_slice(), @@ -1338,7 +1336,7 @@ mod tests { } fn region( - state: RegionMetadataState, + state: RegionState, queue_ordinal: u32, created_seqno: u64, used_bytes: u64, @@ -1360,11 +1358,11 @@ mod tests { metadata.regions[4].created_seqno = 3; metadata.regions[4].queue_ordinal = 0; - metadata.regions[1].state = RegionMetadataState::Sealed; + metadata.regions[1].state = RegionState::Sealed; metadata.regions[1].created_seqno = 4; metadata.regions[1].queue_ordinal = 1; - metadata.regions[5].state = RegionMetadataState::Sealed; + metadata.regions[5].state = RegionState::Sealed; metadata.regions[5].created_seqno = 6; metadata.regions[5].queue_ordinal = 2; diff --git a/cache2/src/region/mod.rs b/cache2/src/region/mod.rs index 6f5db5c..28e6593 100644 --- a/cache2/src/region/mod.rs +++ b/cache2/src/region/mod.rs @@ -38,7 +38,7 @@ use self::manager::RegionReclaimReceipt; use self::reader::PendingRead; use self::reader::ReadCandidate; use self::reader::ReadCompletion; -use self::reader::ReadDescriptor; +use self::reader::ReadDesc; #[cfg(test)] use self::reader::describe_read; use self::reader::submit_read; @@ -76,7 +76,7 @@ use crate::region::recovery::metadata::REGION_METADATA_PAGE_SIZE; use crate::region::recovery::metadata::REGION_METADATA_PARTITIONS_PER_PAGE; use crate::region::recovery::metadata::REGION_METADATA_REGIONS_PER_PAGE; use crate::region::recovery::metadata::RegionMetadataError; -use crate::region::staging::RegionStaging; +use crate::region::staging::AppendStaging; use crate::snapshot::CacheIndexSnapshot; use crate::snapshot::RegionSnapshot; @@ -87,7 +87,6 @@ pub mod record; pub mod recovery; pub mod runtime; pub mod staging; -pub mod store; mod appender; mod reader; @@ -151,7 +150,7 @@ impl RegionHealthLatch { self.is_healthy().then_some(()).ok_or_else(|| { io::Error::new( io::ErrorKind::InvalidData, - "RegionStore is miss-only and cannot publish CLEAN", + "Region store is miss-only and cannot publish CLEAN", ) }) } @@ -160,32 +159,32 @@ impl RegionHealthLatch { /// The steady-state owner of Region allocation, FIFO rotation, and write-span /// accounting. Index publication is deliberately independent; reads validate /// the physical record locally and may observe an older valid completion. -struct RegionManagerAuthority { - inner: Mutex, +struct RegionManagerLock { + state: Mutex, health: RegionHealthLatch, } -impl RegionManagerAuthority { +impl RegionManagerLock { fn new(manager: RegionManager, health: RegionHealthLatch) -> Self { Self { - inner: Mutex::new(manager), + state: Mutex::new(manager), health, } } fn lock(&self) -> io::Result> { self.health.require_healthy()?; - match self.inner.lock() { + match self.state.lock() { Ok(guard) if self.health.is_healthy() => Ok(guard), Ok(_) => Err(io::Error::new( io::ErrorKind::InvalidData, - "RegionStore became miss-only while acquiring Region authority", + "Region store became miss-only while acquiring Region authority", )), Err(_) => { self.health.enter_miss_only(); Err(io::Error::new( io::ErrorKind::InvalidData, - "RegionStore Region authority is poisoned", + "Region manager is poisoned", )) } } @@ -193,37 +192,33 @@ impl RegionManagerAuthority { fn try_lock(&self) -> io::Result>> { self.health.require_healthy()?; - match self.inner.try_lock() { + match self.state.try_lock() { Ok(guard) if self.health.is_healthy() => Ok(Some(guard)), Ok(_) => Err(io::Error::new( io::ErrorKind::InvalidData, - "RegionStore became miss-only while acquiring Region authority", + "Region store became miss-only while acquiring Region authority", )), Err(TryLockError::WouldBlock) => Ok(None), Err(TryLockError::Poisoned(_)) => { self.health.enter_miss_only(); Err(io::Error::new( io::ErrorKind::InvalidData, - "RegionStore Region authority is poisoned", + "Region manager is poisoned", )) } } } } -pub struct FileRegionCore { +pub struct RegionStore { index: RegionIndex, - manager: RegionManagerAuthority, - shards: Box<[RegionShard]>, - region_access: Box<[RegionAccessState]>, + manager: RegionManagerLock, + append_gates: Box<[AppendShardGate]>, + region_generations: Box<[AtomicU64]>, rotation: Mutex<()>, health: RegionHealthLatch, } -struct RegionAccessState { - generation: AtomicU64, -} - #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] pub struct RegionReclaimStats { pub records_scanned: u64, @@ -248,7 +243,7 @@ pub struct RegionReinsertRecord<'a> { /// staging transitions one operation. Span completion and rotation are already /// ordered by the shard's single production worker. #[derive(Default)] -struct RegionShard { +struct AppendShardGate { mutation: Mutex<()>, } @@ -288,9 +283,9 @@ impl RegionValueRead { } } -impl FileRegionCore { +impl RegionStore { pub const fn shard_count(&self) -> usize { - self.shards.len() + self.append_gates.len() } pub const fn index_slot_count(&self) -> usize { @@ -309,7 +304,7 @@ impl FileRegionCore { } pub fn append_shard(&self, hash: u64) -> usize { - route_hash(hash, self.shards.len()) + route_hash(hash, self.append_gates.len()) } pub fn region_snapshot(&self) -> io::Result { @@ -509,8 +504,8 @@ impl FileRegionCore { /// Releases one fully scanned source only after every accepted replacement /// batch has completed and conditionally published. pub fn complete_reclaim(&self, receipt: RegionReclaimReceipt) -> io::Result<()> { - let access = self - .region_access + let generation = self + .region_generations .get(receipt.region_id as usize) .ok_or_else(|| { io::Error::new( @@ -518,7 +513,7 @@ impl FileRegionCore { "reclaim Region id is out of bounds", ) })?; - access.generation.store(0, Ordering::Release); + generation.store(0, Ordering::Release); self.manager .lock()? .finish_reclaim(receipt) @@ -539,12 +534,12 @@ impl FileRegionCore { } fn lock_shard_mutation(&self, shard_id: usize) -> io::Result> { - let shard = self.shard(shard_id)?; + let shard = self.append_gate(shard_id)?; self.lock_shard_gate(&shard.mutation) } - fn shard(&self, shard_id: usize) -> io::Result<&RegionShard> { - self.shards.get(shard_id).ok_or_else(|| { + fn append_gate(&self, shard_id: usize) -> io::Result<&AppendShardGate> { + self.append_gates.get(shard_id).ok_or_else(|| { io::Error::new(io::ErrorKind::InvalidInput, "data shard is out of bounds") }) } @@ -592,10 +587,10 @@ impl FileRegionCore { return None; } }; - let access = self - .region_access + let generation = self + .region_generations .get(entry.location.region_id() as usize)?; - let region_generation = access.generation.load(Ordering::Acquire); + let region_generation = generation.load(Ordering::Acquire); if region_generation == 0 { return None; } @@ -618,21 +613,21 @@ impl FileRegionCore { let Some(candidate) = self.begin_point_read(hash) else { return Ok(None); }; - let descriptor = describe_read(geometry, hash, candidate, true)?; + let desc = describe_read(geometry, hash, candidate, true)?; let slot = engine.try_reserve_read()?; - self.read_value_from_descriptor(engine, slot, buffer, descriptor, key) + self.read_value_from_desc(engine, slot, buffer, desc, key) } #[cfg(test)] - fn read_value_from_descriptor( + fn read_value_from_desc( &self, engine: &IoEngine, slot: ReadSlot, buffer: BufferLease, - descriptor: ReadDescriptor, + desc: ReadDesc, key: &[u8], ) -> io::Result> { - let pending = self.submit_value_read(engine, slot, buffer, descriptor)?; + let pending = self.submit_value_read(engine, slot, buffer, desc)?; let completion = pending.wait(engine); self.finish_value_read(completion, key) } @@ -642,9 +637,9 @@ impl FileRegionCore { engine: &IoEngine, slot: ReadSlot, buffer: BufferLease, - descriptor: ReadDescriptor, + desc: ReadDesc, ) -> io::Result { - match submit_read(engine, slot, descriptor, buffer) { + match submit_read(engine, slot, desc, buffer) { Ok(pending) => Ok(pending), Err(error) => { if !is_read_pressure(error.kind()) { @@ -661,7 +656,7 @@ impl FileRegionCore { completion: ReadCompletion, key: &[u8], ) -> io::Result> { - let hash = completion.descriptor.hash; + let hash = completion.desc.hash; if let Err(error) = completion.result { if !is_read_pressure(error.kind()) { self.health @@ -684,12 +679,11 @@ impl FileRegionCore { else { return Ok(None); }; - if header.region_generation != completion.descriptor.region_generation - || header.key_hash != hash + if header.region_generation != completion.desc.region_generation || header.key_hash != hash { return Ok(None); } - let indexed_location = completion.descriptor.entry.location; + let indexed_location = completion.desc.entry.location; let Ok(exact_location) = PackedLocation::new( indexed_location.region_id(), indexed_location.offset(), @@ -722,7 +716,7 @@ impl FileRegionCore { if crc32c(&[&record[RECORD_HEADER_SIZE..payload_end]]) != header.payload_crc { return Ok(None); } - let value_start = completion.descriptor.record_range.start + RECORD_HEADER_SIZE + key_len; + let value_start = completion.desc.record_range.start + RECORD_HEADER_SIZE + key_len; let Some(value_end) = value_start.checked_add(value_len) else { return Ok(None); }; @@ -737,7 +731,7 @@ impl FileRegionCore { }; Ok(Some(RegionValueRead { buffer, - buffer_len: completion.descriptor.read_len, + buffer_len: completion.desc.read_len, value_range: value_start..value_end, seqno: header.seqno, })) @@ -750,7 +744,7 @@ impl FileRegionCore { /// This method performs no device I/O and never publishes an index entry. pub fn try_stage_value( &self, - staging: &RegionStaging, + staging: &AppendStaging, shard_id: usize, hash: u64, record_bytes: u32, @@ -762,7 +756,7 @@ impl FileRegionCore { pub fn try_stage_reinsert( &self, - staging: &RegionStaging, + staging: &AppendStaging, shard_id: usize, record: RegionReinsertRecord<'_>, ) -> io::Result { @@ -780,7 +774,7 @@ impl FileRegionCore { #[allow(clippy::too_many_arguments)] fn try_stage_record( &self, - staging: &RegionStaging, + staging: &AppendStaging, shard_id: usize, hash: u64, record_bytes: u32, @@ -796,7 +790,7 @@ impl FileRegionCore { )); } let payload = RecordPayload::new(key, value); - let _shard_mutation = match self.shard(shard_id)?.mutation.try_lock() { + let _shard_mutation = match self.append_gate(shard_id)?.mutation.try_lock() { Ok(guard) => guard, Err(TryLockError::WouldBlock) => return Ok(RegionStageValue::NeedsProgress), Err(TryLockError::Poisoned(_)) => { @@ -928,7 +922,7 @@ impl FileRegionCore { fn fail_preflighted_stage( &self, - staging: &RegionStaging, + staging: &AppendStaging, message: &'static str, ) -> io::Result { self.health.enter_miss_only(); @@ -942,7 +936,7 @@ impl FileRegionCore { /// exact owned-buffer write completion succeeds. pub fn flush_staging_shard( &self, - staging: &RegionStaging, + staging: &AppendStaging, engine: &IoEngine, shard_id: usize, recovery: &crate::io::engine::recovery::BackgroundRecovery, @@ -1119,8 +1113,8 @@ impl FileRegionCore { .lock()? .begin_rotation(candidate) .map_err(|error| region_mutation_context("rotation begin", error))?; - let access = self - .region_access + let generation = self + .region_generations .get(receipt.activated_region_id as usize) .ok_or_else(|| { self.health.enter_miss_only(); @@ -1129,9 +1123,7 @@ impl FileRegionCore { "rotation activated an untracked Region", ) })?; - access - .generation - .store(receipt.activated_created_seqno, Ordering::Release); + generation.store(receipt.activated_created_seqno, Ordering::Release); // Manager authority now carries the exact in-progress rotation // receipt, so foreground staging fails fast on this shard. Release the // shard gates before publishing the completed in-memory rotation. @@ -1165,7 +1157,7 @@ impl FileRegionCore { fn fail_staged_span( &self, - staging: &RegionStaging, + staging: &AppendStaging, span: RegionWriteSpan, buffer: Option, records: Vec, @@ -1260,14 +1252,14 @@ fn region_metadata_io_error(error: RegionMetadataError) -> io::Error { fn region_mutation_io_error(error: RegionMutationError) -> io::Error { io::Error::new( io::ErrorKind::InvalidData, - format!("RegionStore authority mutation failed: {error:?}"), + format!("Region state mutation failed: {error:?}"), ) } fn region_mutation_context(context: &'static str, error: RegionMutationError) -> io::Error { io::Error::new( io::ErrorKind::InvalidData, - format!("RegionStore {context} failed: {error:?}"), + format!("Region {context} failed: {error:?}"), ) } diff --git a/cache2/src/region/reader.rs b/cache2/src/region/reader.rs index 5d6c4d5..4245027 100644 --- a/cache2/src/region/reader.rs +++ b/cache2/src/region/reader.rs @@ -50,7 +50,7 @@ pub struct ReadCandidate { } #[derive(Clone, Debug, Eq, PartialEq)] -pub struct ReadDescriptor { +pub struct ReadDesc { pub hash: u64, pub entry: IndexEntry, pub region_generation: u64, @@ -60,13 +60,13 @@ pub struct ReadDescriptor { } pub struct PendingRead { - descriptor: ReadDescriptor, + desc: ReadDesc, request_id: RequestId, request: BoundedIoRequest, } pub struct ReadCompletion { - pub descriptor: ReadDescriptor, + pub desc: ReadDesc, pub result: io::Result<()>, pub buffer: Option, } @@ -80,9 +80,9 @@ impl ReadCompletion { } self.buffer .as_ref()? - .prepared(self.descriptor.read_len) + .prepared(self.desc.read_len) .ok()? - .get(self.descriptor.record_range.clone()) + .get(self.desc.record_range.clone()) } } @@ -90,12 +90,12 @@ impl PendingRead { #[cfg(test)] pub fn wait(self, engine: &IoEngine) -> ReadCompletion { let Self { - descriptor, + desc, request_id, request, } = self; let completion = request.wait(engine); - Self::finish(descriptor, request_id, completion) + Self::finish(desc, request_id, completion) } pub async fn wait_async( @@ -104,16 +104,16 @@ impl PendingRead { tokio_handle: &tokio::runtime::Handle, ) -> ReadCompletion { let Self { - descriptor, + desc, request_id, request, } = self; let completion = request.wait_async(engine, tokio_handle).await; - Self::finish(descriptor, request_id, completion) + Self::finish(desc, request_id, completion) } fn finish( - descriptor: ReadDescriptor, + desc: ReadDesc, request_id: RequestId, completion: Result, ) -> ReadCompletion { @@ -122,7 +122,7 @@ impl PendingRead { Err(timeout) => { let (error, buffer) = timeout.into_lease(); return ReadCompletion { - descriptor, + desc, result: Err(error), buffer, }; @@ -145,7 +145,7 @@ impl PendingRead { )) } else if buffer .as_ref() - .is_some_and(|buffer| buffer.prepared(descriptor.read_len).is_err()) + .is_some_and(|buffer| buffer.prepared(desc.read_len).is_err()) { Some(io::Error::new( io::ErrorKind::InvalidData, @@ -158,7 +158,7 @@ impl PendingRead { let result = match protocol_error { Some(error) => Err(error), None => io_result.and_then(|completed| { - if completed != descriptor.read_len || bytes_transferred != descriptor.read_len { + if completed != desc.read_len || bytes_transferred != desc.read_len { return Err(io::Error::new( io::ErrorKind::UnexpectedEof, "Region record read completed with the wrong byte count", @@ -168,7 +168,7 @@ impl PendingRead { }), }; ReadCompletion { - descriptor, + desc, result, buffer, } @@ -182,14 +182,14 @@ impl PendingRead { pub fn submit_read( engine: &IoEngine, slot: ReadSlot, - descriptor: ReadDescriptor, + desc: ReadDesc, buffer: BufferLease, ) -> io::Result { - let buffer = IoBuffer::for_read(buffer, descriptor.read_len).map_err(|error| error.error)?; - let request = submit_cache_read(engine, slot, IoOperation::read(buffer, descriptor.absolute)) + let buffer = IoBuffer::for_read(buffer, desc.read_len).map_err(|error| error.error)?; + let request = submit_cache_read(engine, slot, IoOperation::read(buffer, desc.absolute)) .map_err(|error| error.into_lease().0)?; Ok(PendingRead { - descriptor, + desc, request_id: request.id(), request, }) @@ -200,7 +200,7 @@ pub fn describe_read( hash: u64, candidate: ReadCandidate, align_for_direct_io: bool, -) -> io::Result { +) -> io::Result { let ReadCandidate { entry, region_generation, @@ -300,7 +300,7 @@ pub fn describe_read( )); } - Ok(ReadDescriptor { + Ok(ReadDesc { hash, entry, region_generation, @@ -383,19 +383,19 @@ mod tests { .unwrap(); let location = PackedLocation::new(1, 32, 64).unwrap(); let entry = entry(location); - let descriptor = describe_read(geometry(), 7, candidate(entry), true).unwrap(); + let desc = describe_read(geometry(), 7, candidate(entry), true).unwrap(); let slot = engine.try_reserve_read().unwrap(); let completion = submit_read( &engine, slot, - descriptor, + desc, managed_memory.try_read_buffer(DIRECT_IO_ALIGNMENT).unwrap(), ) .unwrap() .wait(&engine); assert!(completion.result.is_ok()); - assert_eq!(completion.descriptor.record_range, 32..96); + assert_eq!(completion.desc.record_range, 32..96); assert_eq!(completion.record_bytes().unwrap().len(), 64); let record_absolute = DATA_REGION_AREA_OFFSET + geometry().region_size + 32; @@ -424,19 +424,19 @@ mod tests { }) .unwrap(); let entry = entry(PackedLocation::new(1, 32, 1120).unwrap()); - let descriptor = describe_read(geometry(), 7, candidate(entry), false).unwrap(); + let desc = describe_read(geometry(), 7, candidate(entry), false).unwrap(); let record_absolute = DATA_REGION_AREA_OFFSET + geometry().region_size + 32; let completion = submit_read( &engine, engine.try_reserve_read().unwrap(), - descriptor, + desc, managed_memory.try_read_buffer(1120).unwrap(), ) .unwrap() .wait(&engine); assert!(completion.result.is_ok()); - assert_eq!(completion.descriptor.record_range, 0..1120); + assert_eq!(completion.desc.record_range, 0..1120); assert_eq!(completion.record_bytes().unwrap().len(), 1120); assert_eq!( io.reads @@ -455,12 +455,12 @@ mod tests { let record_len = 1056; let offset = geometry().region_size as u32 - record_len; let entry = entry(PackedLocation::new(1, offset, 1120).unwrap()); - let descriptor = describe_read(geometry(), 7, candidate(entry), false).unwrap(); + let desc = describe_read(geometry(), 7, candidate(entry), false).unwrap(); - assert_eq!(descriptor.record_range, 0..record_len as usize); - assert_eq!(descriptor.read_len, record_len as usize); + assert_eq!(desc.record_range, 0..record_len as usize); + assert_eq!(desc.read_len, record_len as usize); assert_eq!( - descriptor.absolute, + desc.absolute, DATA_REGION_AREA_OFFSET + geometry().region_size + u64::from(offset) ); } diff --git a/cache2/src/region/recovery/metadata.rs b/cache2/src/region/recovery/metadata.rs index a060ba7..27099ef 100644 --- a/cache2/src/region/recovery/metadata.rs +++ b/cache2/src/region/recovery/metadata.rs @@ -132,13 +132,13 @@ impl PageKind { /// Only stable, quiescent Region states can appear in a CLEAN image. #[repr(u8)] #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum RegionMetadataState { +pub enum RegionState { Free = 0, Active = 1, Sealed = 2, } -impl RegionMetadataState { +impl RegionState { fn decode(value: u8) -> Option { match value { 0 => Some(Self::Free), @@ -172,7 +172,7 @@ pub struct RegionMetadataRoot { #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub struct RegionMetadataRecord { - pub state: RegionMetadataState, + pub state: RegionState, /// Free queue position, Active shard id, or Sealed FIFO position. pub queue_ordinal: u32, pub created_seqno: u64, @@ -469,11 +469,9 @@ impl RegionMetadata { .checked_add(removed) .ok_or(RegionMetadataError::ArithmeticOverflow)?; for region in &mut self.regions { - if region.state == RegionMetadataState::Active - && region.queue_ordinal >= shard_count - { + if region.state == RegionState::Active && region.queue_ordinal >= shard_count { let removed_ordinal = region.queue_ordinal - shard_count; - region.state = RegionMetadataState::Sealed; + region.state = RegionState::Sealed; region.queue_ordinal = first_new_sealed .checked_add(removed_ordinal) .ok_or(RegionMetadataError::ArithmeticOverflow)?; @@ -499,11 +497,9 @@ impl RegionMetadata { .checked_add(1) .ok_or(RegionMetadataError::ArithmeticOverflow)?; for region in &mut self.regions { - if region.state == RegionMetadataState::Free - && region.queue_ordinal >= free_region_count - { + if region.state == RegionState::Free && region.queue_ordinal >= free_region_count { let added_ordinal = region.queue_ordinal - free_region_count; - region.state = RegionMetadataState::Active; + region.state = RegionState::Active; region.queue_ordinal = old_shard_count .checked_add(added_ordinal) .ok_or(RegionMetadataError::ArithmeticOverflow)?; @@ -652,16 +648,16 @@ fn validate_regions( return Err(RegionMetadataError::InvalidField("region_geometry")); } let (seen, state_count) = match region.state { - RegionMetadataState::Free => (&mut free_seen, root.free_region_count), - RegionMetadataState::Active => (&mut active_seen, root.active_region_count), - RegionMetadataState::Sealed => (&mut sealed_seen, root.sealed_region_count), + RegionState::Free => (&mut free_seen, root.free_region_count), + RegionState::Active => (&mut active_seen, root.active_region_count), + RegionState::Sealed => (&mut sealed_seen, root.sealed_region_count), }; if region.queue_ordinal >= state_count || mem::replace(&mut seen[region.queue_ordinal as usize], 1) != 0 { return Err(RegionMetadataError::InvalidField("region_queue_ordinal")); } - if region.state == RegionMetadataState::Free { + if region.state == RegionState::Free { if region.created_seqno != 0 || region.durable_used_offset != 0 || region.physical_record_count != 0 @@ -1088,7 +1084,7 @@ fn decode_region(input: &[u8]) -> Result, slot_waiter: ReadSlotWaiter, - descriptor: ReadDescriptor, + desc: ReadDesc, read_token: MemoryReadToken, hash: u64, deadline: Instant, @@ -332,7 +332,7 @@ struct WaitingGet { struct ReservedGet { engine: Arc, slot: ReadSlot, - descriptor: ReadDescriptor, + desc: ReadDesc, read_token: MemoryReadToken, hash: u64, } @@ -379,7 +379,7 @@ impl WaitingGet { let Self { engine, slot_waiter, - descriptor, + desc, read_token, hash, deadline, @@ -390,7 +390,7 @@ impl WaitingGet { Ok(ReservedGet { engine, slot, - descriptor, + desc, read_token, hash, }) @@ -412,7 +412,7 @@ impl HybridValueRead { #[derive(Clone)] pub struct RegionDataPlane { - core: Arc, + regions: Arc, data: DataSuperblock, runtime: RuntimeOptions, metrics: Arc, @@ -430,7 +430,7 @@ struct RunningOwner { } struct RunningShared { - core: Arc, + regions: Arc, read_engines: Box<[Arc]>, read_lane_cursor: AtomicUsize, read_waiters: Option>, @@ -442,9 +442,9 @@ struct RunningShared { managed_memory: Arc, metrics: Arc, memory: Arc, - staging: Arc, + staging: Arc, operations: Arc, - shards: Box<[Arc]>, + shards: Box<[Arc]>, write_flush_threshold_bytes: usize, align_reads_for_direct_io: bool, activity_counters: bool, @@ -521,7 +521,7 @@ impl RunningShared { fn try_queue_read( &self, route: u64, - descriptor: ReadDescriptor, + desc: ReadDesc, read_token: MemoryReadToken, timeout: Duration, ) -> io::Result { @@ -540,7 +540,7 @@ impl RunningShared { Ok(WaitingGet { engine, slot_waiter, - descriptor, + desc, read_token, hash: route, deadline, @@ -590,12 +590,12 @@ fn should_wake_write(previous_bytes: usize, current_bytes: usize, threshold: usi } #[derive(Clone)] -struct ShardFailure { +struct AppendWorkerFailure { kind: io::ErrorKind, message: Arc, } -impl ShardFailure { +impl AppendWorkerFailure { fn from_error(error: &io::Error) -> Self { Self { kind: error.kind(), @@ -609,25 +609,25 @@ impl ShardFailure { } #[derive(Default)] -struct ShardControlState { +struct AppendWorkerState { wake_flags: u8, drain_requested: u64, drain_completed: u64, stop: bool, - failure: Option, + failure: Option, } -struct ShardControl { - state: Mutex, +struct AppendWorkerControl { + state: Mutex, changed: Condvar, async_changed: watch::Sender<()>, } -impl ShardControl { +impl AppendWorkerControl { fn new() -> Self { let (async_changed, _) = watch::channel(()); Self { - state: Mutex::new(ShardControlState::default()), + state: Mutex::new(AppendWorkerState::default()), changed: Condvar::new(), async_changed, } @@ -729,13 +729,13 @@ impl ShardControl { .unwrap_or_else(|poisoned| poisoned.into_inner()); state .failure - .get_or_insert_with(|| ShardFailure::from_error(error)); + .get_or_insert_with(|| AppendWorkerFailure::from_error(error)); drop(state); self.changed.notify_all(); self.async_changed.send_replace(()); } - fn lock(&self) -> io::Result> { + fn lock(&self) -> io::Result> { self.state.lock().map_err(|_| poisoned_runtime_error()) } } @@ -746,7 +746,7 @@ impl RegionDataPlane { } pub fn new( - core: Arc, + regions: Arc, data: DataSuperblock, handles: DataFileHandles, config: CacheConfig, @@ -755,9 +755,9 @@ impl RegionDataPlane { // match the configuration selected for this open. let storage = config.storage(); if data.geometry != storage_geometry(storage) - || core.region_count()? != storage.region_count() as usize - || core.index_slot_count() != storage.index_slots() - || core.shard_count() != config.runtime().append_shards as usize + || regions.region_count()? != storage.region_count() as usize + || regions.index_slot_count() != storage.index_slots() + || regions.shard_count() != config.runtime().append_shards as usize { return Err(io::Error::new( io::ErrorKind::InvalidData, @@ -765,12 +765,14 @@ impl RegionDataPlane { )); } let runtime = config.runtime().clone(); - core.configure_reclaim_workers(IoPoolTopology::reclaim(runtime.io_engine).max_in_flight())?; - core.set_index_activity_counters_enabled(runtime.stats.activity_counters); - let metrics = Arc::new(RuntimeMetrics::new(core.shard_count(), runtime.stats)?); + regions.configure_reclaim_workers( + IoPoolTopology::reclaim(runtime.io_engine).max_in_flight(), + )?; + regions.set_index_activity_counters_enabled(runtime.stats.activity_counters); + let metrics = Arc::new(RuntimeMetrics::new(regions.shard_count(), runtime.stats)?); let operations = Arc::new(MutationGate::new()); let running = start_running( - Arc::clone(&core), + Arc::clone(®ions), data, handles, config, @@ -779,7 +781,7 @@ impl RegionDataPlane { )?; let shared = Arc::clone(&running.shared); Ok(Self { - core, + regions, data, runtime, metrics, @@ -825,7 +827,7 @@ impl RegionDataPlane { return Err(write_overload_error()); } let hash = hash_key(self.data.hash_seed, key); - let shard_id = self.core.append_shard(hash); + let shard_id = self.regions.append_shard(hash); let control = &running.shards[shard_id]; let activity = running .activity_counters @@ -839,7 +841,7 @@ impl RegionDataPlane { return Err(write_overload_error()); } }; - let staged = self.core.try_stage_value( + let staged = self.regions.try_stage_value( &running.staging, shard_id, hash, @@ -904,7 +906,7 @@ impl RegionDataPlane { return Err(write_overload_error()); } }; - let Some(seqno) = self.core.try_delete_value(hash)? else { + let Some(seqno) = self.regions.try_delete_value(hash)? else { drop(operation); if running.activity_counters { running.metrics.record_write_rejection(); @@ -970,15 +972,11 @@ impl RegionDataPlane { let ReservedGet { engine, slot, - descriptor, + desc, read_token, hash, } = reserved; - let Some(buffer) = self - .shared - .managed_memory - .try_read_buffer(descriptor.read_len) - else { + let Some(buffer) = self.shared.managed_memory.try_read_buffer(desc.read_len) else { if self.runtime.stats.activity_counters { self.metrics.record_read_overload(); } @@ -988,8 +986,8 @@ impl RegionDataPlane { )); }; match self - .core - .submit_value_read(engine.as_ref(), slot, buffer, descriptor) + .regions + .submit_value_read(engine.as_ref(), slot, buffer, desc) { Ok(read) => Ok(Some(PendingGet { engine, @@ -997,7 +995,7 @@ impl RegionDataPlane { read_token, hash, })), - Err(_) if !self.core.is_healthy() => { + Err(_) if !self.regions.is_healthy() => { if self.runtime.stats.activity_counters { RuntimeMetrics::increment(&self.metrics.io_failures); RuntimeMetrics::increment(&self.metrics.activity_for_hash(hash).l2_misses); @@ -1029,7 +1027,7 @@ impl RegionDataPlane { let activity = running .activity_counters .then(|| running.metrics.activity_for_hash(hash)); - if !self.core.is_healthy() { + if !self.regions.is_healthy() { if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l1_misses); RuntimeMetrics::increment(&activity.l2_misses); @@ -1057,21 +1055,21 @@ impl RegionDataPlane { token } }; - let Some(candidate) = self.core.begin_point_read(hash) else { + let Some(candidate) = self.regions.begin_point_read(hash) else { if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l2_misses); } return Ok(PreparedGet::Complete(None)); }; - let descriptor = match describe_read( + let desc = match describe_read( self.data.geometry, hash, candidate, running.align_reads_for_direct_io, ) { - Ok(descriptor) => descriptor, + Ok(desc) => desc, Err(error) => { - self.core + self.regions .enter_miss_only_with_error("record_read_descriptor_invalid", &error); if running.activity_counters { RuntimeMetrics::increment(&running.metrics.io_failures); @@ -1089,12 +1087,7 @@ impl RegionDataPlane { && !read_io_wait_timeout(&self.runtime).is_zero() => { let waiting = running - .try_queue_read( - hash, - descriptor, - read_token, - read_io_wait_timeout(&self.runtime), - ) + .try_queue_read(hash, desc, read_token, read_io_wait_timeout(&self.runtime)) .inspect_err(|_| { if running.activity_counters { running.metrics.record_read_overload(); @@ -1110,7 +1103,7 @@ impl RegionDataPlane { return Ok(PreparedGet::Complete(None)); } Err(error) => { - self.core + self.regions .enter_miss_only_with_error("read_engine_reservation_failed", &error); if running.activity_counters { RuntimeMetrics::increment(&running.metrics.io_failures); @@ -1121,7 +1114,7 @@ impl RegionDataPlane { return Ok(PreparedGet::Complete(None)); } }; - let Some(buffer) = running.managed_memory.try_read_buffer(descriptor.read_len) else { + let Some(buffer) = running.managed_memory.try_read_buffer(desc.read_len) else { if !read_io_wait_timeout(&self.runtime).is_zero() { if running.activity_counters { running.metrics.record_read_overload(); @@ -1138,8 +1131,8 @@ impl RegionDataPlane { return Ok(PreparedGet::Complete(None)); }; match self - .core - .submit_value_read(engine.as_ref(), slot, buffer, descriptor) + .regions + .submit_value_read(engine.as_ref(), slot, buffer, desc) { Ok(read) => Ok(PreparedGet::Pending(PendingGet { engine, @@ -1150,8 +1143,8 @@ impl RegionDataPlane { // MissOnly is a cache availability state, not an application data // error. The operation that trips the one-way health latch and all // later reads therefore fail open as cache misses. Resource - // overload remains explicit while the core is still healthy. - Err(_) if !self.core.is_healthy() => { + // overload remains explicit while the regions is still healthy. + Err(_) if !self.regions.is_healthy() => { if let Some(activity) = activity { RuntimeMetrics::increment(&running.metrics.io_failures); RuntimeMetrics::increment(&activity.l2_misses); @@ -1194,9 +1187,9 @@ impl RegionDataPlane { let activity = running .activity_counters .then(|| running.metrics.activity_for_hash(hash)); - let result = self.core.finish_value_read(read, key); + let result = self.regions.finish_value_read(read, key); match result { - Err(_) if !self.core.is_healthy() => { + Err(_) if !self.regions.is_healthy() => { if let Some(activity) = activity { RuntimeMetrics::increment(&running.metrics.io_failures); RuntimeMetrics::increment(&activity.l2_misses); @@ -1279,14 +1272,14 @@ impl RegionDataPlane { .write_buffer_rejections .load(Ordering::Relaxed), l1: running.memory.detailed_snapshot()?, - index: self.core.index_snapshot()?, - region: self.core.region_snapshot()?, + index: self.regions.index_snapshot()?, + region: self.regions.region_snapshot()?, }) } fn snapshot_running(&self, running: &RunningShared) -> CacheSnapshot { let mut snapshot = self.metrics.snapshot( - self.core.is_healthy(), + self.regions.is_healthy(), self.runtime.stats.activity_counters, running.managed_memory.snapshot(), running.memory.metrics_snapshot(), @@ -1399,14 +1392,14 @@ fn add_io_direction(aggregate: &mut CacheIoDirectionSnapshot, snapshot: CacheIoD } fn start_running( - core: Arc, + regions: Arc, data: DataSuperblock, handles: DataFileHandles, config: CacheConfig, metrics: Arc, operations: Arc, ) -> io::Result { - let shard_count = core.shard_count(); + let shard_count = regions.shard_count(); let runtime = config.runtime(); let l1_entry_capacity = l1_entry_capacity(&config); let memory_limit = runtime.managed_memory_limit_bytes; @@ -1421,7 +1414,7 @@ fn start_running( .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "Region size is too large"))?; let chunk_bytes = usable_region; let staging = Arc::new( - RegionStaging::try_new( + AppendStaging::try_new( shard_count, chunk_bytes, data.geometry.region_size, @@ -1496,9 +1489,9 @@ fn start_running( shards.try_reserve_exact(shard_count).map_err(|_| { io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate shard controls") })?; - shards.resize_with(shard_count, || Arc::new(ShardControl::new())); + shards.resize_with(shard_count, || Arc::new(AppendWorkerControl::new())); let shared = Arc::new(RunningShared { - core, + regions, read_engines, read_lane_cursor: AtomicUsize::new(0), read_waiters, @@ -1521,7 +1514,7 @@ fn start_running( }); // Inspect the recovered queue before workers can contend with foreground // mutations. Fresh caches have no sealed Regions and need no wakeup. - let reclaim_on_start = shared.core.reclaim_needed()?; + let reclaim_on_start = shared.regions.reclaim_needed()?; let mut reclaim_workers = Vec::new(); reclaim_workers .try_reserve_exact(reclaim_worker_count) @@ -1647,7 +1640,7 @@ fn shard_worker(shared: Arc, shard_id: usize) { "cache shard worker failed" ); } - shared.core.enter_miss_only(); + shared.regions.enter_miss_only(); shared.recovery.stop(); control.fail(&error); // Wake engine admission in case another shard is blocked behind work that @@ -1714,7 +1707,7 @@ fn reclaim_worker( .metrics .lifecycle .store(LIFECYCLE_FAILED, Ordering::Release); - shared.core.enter_miss_only(); + shared.regions.enter_miss_only(); shared.recovery.stop(); log::error!( target: "cache2::health", @@ -1753,7 +1746,7 @@ fn reclaim_worker_result( return Ok(()); } let mut recovery = shared.recovery.attempt(); - let Some(receipt) = shared.core.begin_reclaim()? else { + let Some(receipt) = shared.regions.begin_reclaim()? else { break; }; let used = usize::try_from(receipt.used_offset).map_err(|_| { @@ -1770,7 +1763,7 @@ fn reclaim_worker_result( Ok(buffer) => buffer, Err(error) => return Err(error.error), }; - let absolute = shared.core.reclaim_absolute(receipt)?; + let absolute = shared.regions.reclaim_absolute(receipt)?; // Reclaim owns a dedicated pool whose depth matches its worker // count. Use the bounded background wait so transient CAS // contention cannot turn a healthy cache miss-only; foreground @@ -1816,7 +1809,7 @@ fn reclaim_worker_result( // Keep one completion boundary per source Region while each // reclaimer rotates through a disjoint subset of append shards. let reinsert_shard = reinsert_shards.take(); - let preserve_hot = shared.core.reclaim_can_reinsert()?; + let preserve_hot = shared.regions.reclaim_can_reinsert()?; let reinsert_operation = if preserve_hot { shared.operations.try_enter() } else { @@ -1824,12 +1817,12 @@ fn reclaim_worker_result( }; let mut accepting_reinserts = reinsert_operation.is_some(); let mut staged_reinsert = false; - let stats = shared.core.scan_reclaim(receipt, bytes, |record| { + let stats = shared.regions.scan_reclaim(receipt, bytes, |record| { if !accepting_reinserts { return Ok(false); } match shared - .core + .regions .try_stage_reinsert(&shared.staging, reinsert_shard, record)? { RegionStageValue::Staged { .. } => { @@ -1846,7 +1839,7 @@ fn reclaim_worker_result( let generation = shared.shards[reinsert_shard].request_drain(false)?; shared.shards[reinsert_shard].wait_for_drain(generation)?; } - shared.core.complete_reclaim(receipt)?; + shared.regions.complete_reclaim(receipt)?; recovery.finish(); drop(reinsert_operation); if shared.activity_counters { @@ -1879,7 +1872,7 @@ fn reclaim_worker_result( fn shard_worker_result( shared: &RunningShared, shard_id: usize, - control: &ShardControl, + control: &AppendWorkerControl, ) -> io::Result<()> { let mut deadline = None; loop { @@ -1897,7 +1890,7 @@ fn shard_worker_result( } if force_flush || fill.bytes >= shared.write_flush_threshold_bytes { let engine = shared.write_engine_for(shard_id as u64); - shared.core.flush_staging_shard( + shared.regions.flush_staging_shard( &shared.staging, engine.as_ref(), shard_id, @@ -1909,7 +1902,7 @@ fn shard_worker_result( Ok(None) => { deadline = None; if rotate { - let rotated = shared.core.rotate_shard(shard_id)?; + let rotated = shared.regions.rotate_shard(shard_id)?; if rotated && shared.activity_counters { RuntimeMetrics::increment(&shared.metrics.region_rotations); } @@ -1932,7 +1925,7 @@ fn shard_worker_result( match shared.staging.shard_fill_snapshot(shard_id) { Ok(Some(_)) => { let engine = shared.write_engine_for(shard_id as u64); - shared.core.flush_staging_shard( + shared.regions.flush_staging_shard( &shared.staging, engine.as_ref(), shard_id, @@ -1955,7 +1948,7 @@ fn shard_worker_result( } fn wait_for_shard_work( - control: &ShardControl, + control: &AppendWorkerControl, deadline: Option, ) -> io::Result<(u8, u64, bool, bool)> { let mut state = control.lock()?; @@ -1999,7 +1992,7 @@ fn wait_for_shard_work( fn reject_staged_write( running: &RunningShared, - control: &ShardControl, + control: &AppendWorkerControl, flags: u8, operation: MutationGuard<'_>, ) -> io::Result { @@ -2012,7 +2005,7 @@ fn reject_staged_write( Err(write_overload_error()) } -fn complete_shard_drain(control: &ShardControl, generation: u64) -> io::Result<()> { +fn complete_shard_drain(control: &AppendWorkerControl, generation: u64) -> io::Result<()> { let mut state = control.lock()?; state.drain_completed = state.drain_completed.max(generation); control.changed.notify_all(); @@ -2118,7 +2111,7 @@ fn stop_running(mut owner: RunningOwner) -> io::Result { .any(|engine| engine.has_unfenced_writes()); // A request that missed its cancellation grace may still own a kernel // target and buffer. Joining that engine can wait forever. Retain only the - // engine Arc; the runtime/core can still be released normally. + // engine Arc; the runtime/regions can still be released normally. let skip_shutdown = in_flight != 0 || unfenced_before; let shutdown = if skip_shutdown { Ok(()) @@ -2232,7 +2225,7 @@ mod tests { static LANE_TEST_ID: AtomicU64 = AtomicU64::new(1); struct ShardStateLockProbe { - control: Arc, + control: Arc, observed_unlocked: AtomicBool, } @@ -2405,7 +2398,7 @@ mod tests { #[test] fn shard_failure_wakes_async_waiters_after_releasing_state_lock() { - let control = Arc::new(ShardControl::new()); + let control = Arc::new(AppendWorkerControl::new()); let probe = Arc::new(ShardStateLockProbe { control: Arc::clone(&control), observed_unlocked: AtomicBool::new(false), @@ -2422,7 +2415,7 @@ mod tests { #[test] fn urgent_empty_shard_wake_is_consumed() { - let control = ShardControl::new(); + let control = AppendWorkerControl::new(); control.notify(WAKE_URGENT).unwrap(); let (flags, drain_generation, stop, timed_out) = @@ -2442,7 +2435,7 @@ mod tests { #[test] fn reclaim_progress_does_not_fail_after_a_shard_stops() { - let control = ShardControl::new(); + let control = AppendWorkerControl::new(); control.request_drain(true).unwrap(); control.notify_if_running(WAKE_ROTATE).unwrap(); @@ -2454,7 +2447,7 @@ mod tests { #[tokio::test] async fn stopped_shard_rejects_new_drains_and_completes_accepted_drains() { - let control = ShardControl::new(); + let control = AppendWorkerControl::new(); let first = control.request_drain(false).unwrap(); let stop = control.request_drain(true).unwrap(); @@ -2509,6 +2502,7 @@ mod tests { #[test] fn completion_timeouts_follow_read_wait_mode() { + use crate::cache::session::CacheSession; use crate::config::runtime::IoEngineOptions; use crate::config::runtime::PosixIoOptions; use crate::region::file_backend::FileRegionBackend; @@ -2517,7 +2511,6 @@ mod tests { use crate::region::index::packed::PackedLocation; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::PersistentId; - use crate::region::store::RegionStore; let id = LANE_TEST_ID.fetch_add(1, Ordering::Relaxed); let path = env::temp_dir().join(format!( @@ -2564,7 +2557,7 @@ mod tests { }, ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut store = CacheSession::open( 8, FileRegionBackend::for_test_with_options(paths.clone(), data, 8, config), ) @@ -2576,7 +2569,7 @@ mod tests { let result = plane.finish_get( CompletedGet { read: ReadCompletion { - descriptor: ReadDescriptor { + desc: ReadDesc { hash: 7, entry: IndexEntry { location: PackedLocation::new(0, 0, 64).unwrap(), @@ -2595,7 +2588,7 @@ mod tests { b"key", ); let snapshot = plane.snapshot().unwrap(); - assert!(plane.core.is_healthy()); + assert!(plane.regions.is_healthy()); store.close_fast().unwrap(); if wait.is_zero() { assert!(matches!(result, Ok(None))); diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/region/runtime/shutdown_tests.rs index d0a3e87..eb34ab5 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/region/runtime/shutdown_tests.rs @@ -78,11 +78,11 @@ fn submitted_read_must_not_pin_close() { } fn assert_close_does_not_wait_for_read(submit_before_close: bool) { + use crate::cache::session::CacheSession; use crate::config::runtime::PosixIoOptions; use crate::region::file_backend::FileRegionBackend; use crate::region::file_backend::RegionPaths; use crate::region::recovery::PersistentId; - use crate::region::store::RegionStore; let root = env::temp_dir().join(format!( "cache2-close-race-{}-{submit_before_close}", std::process::id() @@ -111,7 +111,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { }), ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut store = CacheSession::open( 8, FileRegionBackend::for_test_with_options(paths, data, 8, config), ) @@ -181,10 +181,10 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { #[test] fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { + use crate::cache::session::CacheSession; use crate::region::file_backend::FileRegionBackend; use crate::region::file_backend::RegionPaths; use crate::region::recovery::PersistentId; - use crate::region::store::RegionStore; use crate::snapshot::CacheHealth; let root = env::temp_dir().join(format!("cache2-recovery-admission-{}", std::process::id())); std::fs::create_dir_all(&root).unwrap(); @@ -206,7 +206,7 @@ fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { l1_capacity_bytes: 0, ..RuntimeOptions::default() }; - let mut store = RegionStore::open( + let mut store = CacheSession::open( 8, FileRegionBackend::for_test_with_options(paths, data, 8, config), ) diff --git a/cache2/src/region/staging.rs b/cache2/src/region/staging.rs index 03d4493..9057531 100644 --- a/cache2/src/region/staging.rs +++ b/cache2/src/region/staging.rs @@ -122,8 +122,8 @@ impl StagedRecord { /// A zero-copy write job. `buffer` is the shard's former fill lease and is /// therefore 4 KiB aligned. The I/O completion must return this exact buffer -/// and the record vector to [`RegionStaging::finish_success`] or -/// [`RegionStaging::finish_failure`]. +/// and the record vector to [`AppendStaging::finish_success`] or +/// [`AppendStaging::finish_failure`]. pub struct StagedWrite { pub span: RegionWriteSpan, pub buffer: IoBuffer, @@ -249,14 +249,14 @@ impl Drop for EncodingBuffer<'_> { /// fill while the former is owned by the I/O engine. There is no resident-read /// copy and no staging-owned span sequence: the Region manager receipt is the /// sole identity accepted by sealing and completion. -pub struct RegionStaging { +pub struct AppendStaging { shards: Vec, chunk_bytes: usize, region_size: u64, _memory: RuntimeMemoryReservation, } -impl RegionStaging { +impl AppendStaging { pub fn reservation_bytes(shard_count: usize, chunk_bytes: usize) -> Option { let buffers_per_shard = chunk_bytes.checked_mul(2)?; let records_per_shard = MAX_STAGING_RECORDS @@ -978,7 +978,7 @@ mod tests { fn seal_moves_the_aligned_fill_lease_and_keeps_filling_the_second_buffer() { assert_eq!(size_of::(), 32); let managed_memory = managed_memory(4 * 1024 * 1024); - let staging = RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); + let staging = AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); assert_eq!(staging.chunk_bytes(), 4096); assert_eq!( managed_memory.snapshot().current_bytes, @@ -1048,7 +1048,7 @@ mod tests { #[test] fn fill_snapshot_distinguishes_empty_ready_submitted_and_terminal_shards() { let managed_memory = managed_memory(8 * 1024 * 1024); - let staging = RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); + let staging = AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); assert_eq!(staging.shard_fill_snapshot(0).unwrap(), None); assert_eq!( staging.shard_fill_snapshot(1), @@ -1084,7 +1084,7 @@ mod tests { staging.close(); assert_eq!(staging.shard_fill_snapshot(0), Err(StagingError::Closed)); - let failed = RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); + let failed = AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); failed .encode_reserved(receipt, |target| { target.fill(0x32); @@ -1104,7 +1104,7 @@ mod tests { fn fill_snapshot_never_observes_a_partially_encoded_record() { let managed_memory = managed_memory(4 * 1024 * 1024); let staging = - Arc::new(RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap()); + Arc::new(AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap()); let (receipt, record) = reservation(4096, 64, 11); let (entered_tx, entered_rx) = mpsc::sync_channel(0); let (release_tx, release_rx) = mpsc::sync_channel(0); @@ -1140,7 +1140,7 @@ mod tests { #[test] fn padding_receipt_expands_only_the_final_record_without_copying() { let managed_memory = managed_memory(4 * 1024 * 1024); - let staging = RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); + let staging = AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); let first_len = RecordHeader::aligned_len(0, 0).unwrap(); let (first, first_record) = reservation(4096, first_len, 11); let mut pointer = 0_usize; @@ -1219,7 +1219,7 @@ mod tests { fn fixed_record_and_byte_bounds_request_a_seal_without_running_encoder() { let managed_memory = managed_memory(8 * 1024 * 1024); let chunk_bytes = 256 * 1024; - let staging = RegionStaging::try_new(1, chunk_bytes, 512 * 1024, &managed_memory).unwrap(); + let staging = AppendStaging::try_new(1, chunk_bytes, 512 * 1024, &managed_memory).unwrap(); let mut offset = 0; for index in 0..MAX_STAGING_RECORDS { let (receipt, record) = reservation(offset, RECORD_ALIGNMENT, index as u64 + 1); @@ -1281,7 +1281,7 @@ mod tests { #[test] fn completion_fences_stale_receipts_and_write_failure_is_sticky() { let managed_memory = managed_memory(8 * 1024 * 1024); - let staging = RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); + let staging = AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); let (receipt, record) = reservation(4096, 64, 11); let entry = record.entry(); let mismatched = StagedRecord::new( @@ -1321,7 +1321,7 @@ mod tests { Err(StagingEncodeError::Staging(StagingError::Failed)) ); - let other = RegionStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); + let other = AppendStaging::try_new(1, 4096, 64 * 1024, &managed_memory).unwrap(); other .encode_reserved(receipt, |target| { target.fill(0x44); From b76b7954630f6428adf1549232ec30a194e1cd92 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:22:04 +0800 Subject: [PATCH 05/12] refactor: separate cache execution from region persistence Move L1/L2 request coordination and background workers into CacheRuntime. CacheSession now orders recovery, the durable RUNNING barrier, worker startup, shutdown, and CLEAN publication explicitly. Narrow RegionPersistence to file ownership and recovery image publication. Move live-state construction and freezing into RegionStore, remove the FileRegionRuntime forwarding wrapper, and pass validated configuration through the session instead of retaining it in the persistence layer. Preserve admission fences, write completion ordering, health checks, and file-lock retention when issued writes cannot be fenced. Update fixtures to exercise the same startup and shutdown path as production. --- cache2/src/cache.rs | 49 +- .../src/{region => cache}/runtime/metrics.rs | 6 +- cache2/src/{region => cache}/runtime/mod.rs | 569 +++++++++--------- .../runtime/shutdown_tests.rs | 78 ++- cache2/src/cache/session.rs | 192 ++++-- cache2/src/config/runtime.rs | 2 +- cache2/src/region/mod.rs | 222 ++++++- .../{file_backend => persistence}/mod.rs | 502 +++------------ .../{file_backend => persistence}/tests.rs | 557 ++++++----------- 9 files changed, 949 insertions(+), 1228 deletions(-) rename cache2/src/{region => cache}/runtime/metrics.rs (98%) rename cache2/src/{region => cache}/runtime/mod.rs (85%) rename cache2/src/{region => cache}/runtime/shutdown_tests.rs (77%) rename cache2/src/region/{file_backend => persistence}/mod.rs (70%) rename cache2/src/region/{file_backend => persistence}/tests.rs (79%) diff --git a/cache2/src/cache.rs b/cache2/src/cache.rs index 2541aa3..e294f56 100644 --- a/cache2/src/cache.rs +++ b/cache2/src/cache.rs @@ -37,6 +37,8 @@ use tokio::task::JoinError; use crate::LatencyMode; use crate::RequestOperation; +use crate::cache::runtime::CacheRead; +use crate::cache::runtime::CacheRuntime; use crate::cache::session::CacheSession; use crate::config::CacheConfig; use crate::config::storage::KEY_HASH_SEED; @@ -45,18 +47,16 @@ use crate::config::storage_geometry; use crate::error::Error; use crate::error::ErrorOperation; use crate::error::from_io; -use crate::region::file_backend::FileRegionBackend; -use crate::region::file_backend::RegionPaths; +use crate::region::persistence::RegionPaths; use crate::region::recovery::DataSuperblock; use crate::region::recovery::PersistentId; use crate::region::recovery::RECOVERY_IMAGE_INDEX_OFFSET; use crate::region::recovery::recovery_image_index_len; -use crate::region::runtime::HybridValueRead; -use crate::region::runtime::RegionDataPlane; use crate::snapshot::CacheSnapshot; use crate::snapshot::DetailedCacheSnapshot; use crate::snapshot::StartupMode; +pub mod runtime; pub mod session; /// Storage tier that served a lookup. @@ -77,7 +77,7 @@ pub enum CacheTier { /// The value dereferences to its bytes and keeps its L1 or transient L2 /// backing alive until it is dropped. pub struct Value { - inner: HybridValueRead, + inner: CacheRead, } impl Deref for Value { @@ -120,7 +120,7 @@ pub struct Cache { closed: AtomicBool, read_recording: bool, mutation_recording: bool, - data_plane: RegionDataPlane, + runtime: CacheRuntime, session: Arc>, startup: StartupMode, path: PathBuf, @@ -190,8 +190,8 @@ impl Cache { tokio_handle: tokio::runtime::Handle, started: Instant, ) -> io::Result { - let capacity_bytes = config.storage().capacity_bytes(); let index_slots = config.storage().index_slots(); + let capacity_bytes = config.storage().capacity_bytes(); let index_bytes = u64::try_from(index_slots) .ok() .and_then(recovery_image_index_len) @@ -247,7 +247,6 @@ impl Cache { sidecar_path(&path, ".state"), sidecar_path(&path, ".image"), ); - let index_slots = config.storage().index_slots(); let logical_disk_peak_bytes = config.storage().peak_disk_bytes(); let stats = config.runtime().stats; let read_recording = stats.request_counters @@ -255,15 +254,14 @@ impl Cache { || stats.l2_latency != LatencyMode::Off; let mutation_recording = stats.request_counters || stats.mutation_latency != LatencyMode::Off; - let backend = FileRegionBackend::new(paths, format_data, config); - let session = CacheSession::open(index_slots, backend)?; + let session = CacheSession::open(paths, format_data, config)?; let startup = session.startup(); - let data_plane = session.data_plane_handle()?; + let runtime = session.runtime()?.clone(); Ok(Cache { closed: AtomicBool::new(false), read_recording, mutation_recording, - data_plane, + runtime, session: Arc::new(Mutex::new(session)), startup, path, @@ -290,7 +288,7 @@ impl Cache { /// classifications. pub fn put(&self, key: impl AsRef<[u8]>, value: impl AsRef<[u8]>) -> Result { self.mutate(ErrorOperation::Put, RequestOperation::Put, || { - self.data_plane.put(key.as_ref(), value.as_ref()) + self.runtime.put(key.as_ref(), value.as_ref()) }) } @@ -308,7 +306,7 @@ impl Cache { /// [`ErrorOperation::PutL2`](ErrorOperation::PutL2) as its context. pub fn put_l2(&self, key: impl AsRef<[u8]>, value: impl AsRef<[u8]>) -> Result { self.mutate(ErrorOperation::PutL2, RequestOperation::PutL2, || { - self.data_plane.put_l2(key.as_ref(), value.as_ref()) + self.runtime.put_l2(key.as_ref(), value.as_ref()) }) } @@ -325,7 +323,7 @@ impl Cache { /// Runtime and device failures remain explicit. pub fn delete(&self, key: impl AsRef<[u8]>) -> Result { self.mutate(ErrorOperation::Delete, RequestOperation::Delete, || { - self.data_plane.delete(key.as_ref()) + self.runtime.delete(key.as_ref()) }) } @@ -353,22 +351,19 @@ impl Cache { } return public_result( ErrorOperation::Get, - self.data_plane + self.runtime .get_async(key.as_ref(), &self.tokio_handle, None) .await, ) .map(|value| value.map(|inner| Value { inner })); } - let mut guard = self - .data_plane - .stats_recorder() - .begin(RequestOperation::Get); + let mut guard = self.runtime.stats_recorder().begin(RequestOperation::Get); let result = if self.is_closed() { Ok(None) } else { public_result( ErrorOperation::Get, - self.data_plane + self.runtime .get_async(key.as_ref(), &self.tokio_handle, Some(&mut guard)) .await, ) @@ -394,7 +389,7 @@ impl Cache { /// Accepted work that cannot complete returns a structured runtime/device failure. pub async fn drain(&self) -> Result<(), Error> { self.ensure_open(ErrorOperation::Drain)?; - public_result(ErrorOperation::Drain, self.data_plane.drain_async().await) + public_result(ErrorOperation::Drain, self.runtime.drain_async().await) } /// Returns a lock-free operational snapshot. Activity and I/O counters are @@ -408,7 +403,7 @@ impl Cache { /// structured runtime failure if the snapshot cannot be read. pub fn snapshot(&self) -> Result { self.ensure_open(ErrorOperation::Snapshot)?; - let mut snapshot = public_result(ErrorOperation::Snapshot, self.data_plane.snapshot())?; + let mut snapshot = public_result(ErrorOperation::Snapshot, self.runtime.snapshot())?; snapshot.logical_disk_peak_bytes = self.logical_disk_peak_bytes; Ok(snapshot) } @@ -424,7 +419,7 @@ impl Cache { /// /// Uses the same availability and runtime failures as [`Self::snapshot`]. pub fn stats_snapshot(&self) -> Result { - Ok(self.data_plane.stats_recorder().snapshot(self.snapshot()?)) + Ok(self.runtime.stats_recorder().snapshot(self.snapshot()?)) } /// Samples L1, index, write-buffer rejection, I/O, and Region state in @@ -441,7 +436,7 @@ impl Cache { self.ensure_open(ErrorOperation::DetailedSnapshot)?; let mut snapshot = public_result( ErrorOperation::DetailedSnapshot, - self.data_plane.detailed_snapshot(), + self.runtime.detailed_snapshot(), )?; snapshot.summary.logical_disk_peak_bytes = self.logical_disk_peak_bytes; Ok(snapshot) @@ -486,7 +481,7 @@ impl Cache { ) -> Result { let guard = self .mutation_recording - .then(|| self.data_plane.stats_recorder().begin(stats_operation)); + .then(|| self.runtime.stats_recorder().begin(stats_operation)); let result = self .ensure_open(operation) .and_then(|()| public_result(operation, mutation())); @@ -517,7 +512,7 @@ impl Cache { }; let tokio_handle = self.tokio_handle.clone(); let close = (!self.closed.swap(true, Ordering::AcqRel)).then(|| { - self.data_plane.start_close(); + self.runtime.start_close(); let session = Arc::clone(&self.session); let path = self.path.clone(); let started = Instant::now(); diff --git a/cache2/src/region/runtime/metrics.rs b/cache2/src/cache/runtime/metrics.rs similarity index 98% rename from cache2/src/region/runtime/metrics.rs rename to cache2/src/cache/runtime/metrics.rs index cbe0cf2..5f8822f 100644 --- a/cache2/src/region/runtime/metrics.rs +++ b/cache2/src/cache/runtime/metrics.rs @@ -18,13 +18,13 @@ use std::sync::atomic::AtomicU64; use std::sync::atomic::Ordering; use std::time::Duration; +use crate::cache::runtime::LIFECYCLE_DRAINING; +use crate::cache::runtime::LIFECYCLE_FAILED; +use crate::cache::runtime::LIFECYCLE_RUNNING; use crate::hashing::route_hash; use crate::managed_memory::ManagedMemorySnapshot; use crate::memory::MemoryMetricsSnapshot; use crate::region::RegionReclaimStats; -use crate::region::runtime::LIFECYCLE_DRAINING; -use crate::region::runtime::LIFECYCLE_FAILED; -use crate::region::runtime::LIFECYCLE_RUNNING; use crate::snapshot::CacheHealth; use crate::snapshot::CacheIoSnapshot; use crate::snapshot::CacheReclaimSnapshot; diff --git a/cache2/src/region/runtime/mod.rs b/cache2/src/cache/runtime/mod.rs similarity index 85% rename from cache2/src/region/runtime/mod.rs rename to cache2/src/cache/runtime/mod.rs index e51e596..9a9a855 100644 --- a/cache2/src/region/runtime/mod.rs +++ b/cache2/src/cache/runtime/mod.rs @@ -12,12 +12,11 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Steady-state runtime that owns the CacheSession shard and reclaim workers. +//! Cache request execution, L1/L2 coordination, and bounded background workers. //! //! Foreground writers encode directly into the fixed per-shard write //! buffers. Shard workers carry only coalesced control state, so queueing cannot -//! duplicate payload memory or let a benchmark generator inflate the measured -//! device path. A fixed age deadline publishes partial batches without adding +//! duplicate payload memory. A fixed age deadline publishes partial batches without adding //! a durability sync; CLEAN remains the only steady-state durability boundary. use std::io; @@ -76,7 +75,7 @@ use crate::memory::MemoryStore; use crate::memory::MemoryValue; use crate::region::RegionStageValue; use crate::region::RegionStore; -use crate::region::RegionValueRead; +use crate::region::RegionValue; #[cfg(test)] use crate::region::index::packed::IndexEntry; #[cfg(test)] @@ -297,9 +296,9 @@ const WAKE_DATA: u8 = 1; const WAKE_URGENT: u8 = 2; const WAKE_ROTATE: u8 = 4; -pub enum HybridValueRead { +pub enum CacheRead { L1(MemoryValue), - L2(RegionValueRead), + L2(RegionValue), /// An L2 hit copied into the bounded L1 tier. The public tier remains /// L2 because that is where this lookup was served, but the transient /// aligned read allocation can be released before `get` returns. @@ -307,7 +306,7 @@ pub enum HybridValueRead { } enum PreparedGet { - Complete(Option), + Complete(Option), Pending(PendingGet), Waiting(WaitingGet), } @@ -397,7 +396,7 @@ impl WaitingGet { } } -impl HybridValueRead { +impl CacheRead { pub fn value(&self) -> &[u8] { match self { Self::L1(value) | Self::PromotedL2(value) => value.as_ref(), @@ -411,25 +410,25 @@ impl HybridValueRead { } #[derive(Clone)] -pub struct RegionDataPlane { +pub struct CacheRuntime { regions: Arc, data: DataSuperblock, - runtime: RuntimeOptions, + options: RuntimeOptions, metrics: Arc, - shared: Arc, - owner: Arc>>, + state: Arc, + workers: Arc>>, // Fences write admission for drain, flush, and shutdown. Reads do not // participate because they cannot extend the set of records being fenced. operations: Arc, } -struct RunningOwner { - shared: Arc, - shard_workers: Vec>, +struct RuntimeWorkers { + state: Arc, + append_workers: Vec>, reclaim_workers: Vec>, } -struct RunningShared { +struct RuntimeState { regions: Arc, read_engines: Box<[Arc]>, read_lane_cursor: AtomicUsize, @@ -444,7 +443,7 @@ struct RunningShared { memory: Arc, staging: Arc, operations: Arc, - shards: Box<[Arc]>, + append_controls: Box<[Arc]>, write_flush_threshold_bytes: usize, align_reads_for_direct_io: bool, activity_counters: bool, @@ -509,7 +508,7 @@ impl ReclaimControl { } } -impl RunningShared { +impl RuntimeState { fn write_engine_for(&self, route: u64) -> &Arc { &self.write_engines[route_hash(route, self.write_engines.len())] } @@ -740,12 +739,16 @@ impl AppendWorkerControl { } } -impl RegionDataPlane { +impl CacheRuntime { + pub fn regions(&self) -> &Arc { + &self.regions + } + pub fn stats_recorder(&self) -> &crate::stats::recording::Recorder { &self.metrics.stats } - pub fn new( + pub fn start( regions: Arc, data: DataSuperblock, handles: DataFileHandles, @@ -764,14 +767,14 @@ impl RegionDataPlane { "recovered layout does not match the cache configuration", )); } - let runtime = config.runtime().clone(); + let options = config.runtime().clone(); regions.configure_reclaim_workers( - IoPoolTopology::reclaim(runtime.io_engine).max_in_flight(), + IoPoolTopology::reclaim(options.io_engine).max_in_flight(), )?; - regions.set_index_activity_counters_enabled(runtime.stats.activity_counters); - let metrics = Arc::new(RuntimeMetrics::new(regions.shard_count(), runtime.stats)?); + regions.set_index_activity_counters_enabled(options.stats.activity_counters); + let metrics = Arc::new(RuntimeMetrics::new(regions.shard_count(), options.stats)?); let operations = Arc::new(MutationGate::new()); - let running = start_running( + let workers = start_workers( Arc::clone(®ions), data, handles, @@ -779,20 +782,20 @@ impl RegionDataPlane { Arc::clone(&metrics), Arc::clone(&operations), )?; - let shared = Arc::clone(&running.shared); + let state = Arc::clone(&workers.state); Ok(Self { regions, data, - runtime, + options, metrics, - shared, - owner: Arc::new(Mutex::new(Some(running))), + state, + workers: Arc::new(Mutex::new(Some(workers))), operations, }) } pub fn start_close(&self) { - self.shared.recovery.stop(); + self.state.recovery.stop(); self.operations.start_close(); } @@ -819,30 +822,30 @@ impl RegionDataPlane { "encoded file-chunk entry exceeds one Region", )); } - let running = &self.shared; - if running.recovery.is_recovering() { - if running.activity_counters { - running.metrics.record_write_rejection(); + let state = &self.state; + if state.recovery.is_recovering() { + if state.activity_counters { + state.metrics.record_write_rejection(); } return Err(write_overload_error()); } let hash = hash_key(self.data.hash_seed, key); let shard_id = self.regions.append_shard(hash); - let control = &running.shards[shard_id]; - let activity = running + let control = &state.append_controls[shard_id]; + let activity = state .activity_counters - .then(|| running.metrics.activity_for_hash(hash)); + .then(|| state.metrics.activity_for_hash(hash)); let operation = match self.operations.try_enter() { Some(operation) => operation, None => { - if running.activity_counters { - running.metrics.record_write_rejection(); + if state.activity_counters { + state.metrics.record_write_rejection(); } return Err(write_overload_error()); } }; let staged = self.regions.try_stage_value( - &running.staging, + &state.staging, shard_id, hash, record_bytes, @@ -856,17 +859,17 @@ impl RegionDataPlane { current_bytes, } => { if ADMIT_L1 { - let _published = running.memory.publish(hash, key, value, seqno); + let _published = state.memory.publish(hash, key, value, seqno); } else { // Prevent an older exact-key L1 value from indefinitely // shadowing the prefetched L2 record. Contention remains a // valid best-effort stale outcome. - let _removed = running.memory.delete(hash, key, seqno); + let _removed = state.memory.delete(hash, key, seqno); } if should_wake_write( previous_bytes, current_bytes, - running.write_flush_threshold_bytes, + state.write_flush_threshold_bytes, ) { control.notify(WAKE_DATA)?; } @@ -877,10 +880,10 @@ impl RegionDataPlane { Ok(seqno) } RegionStageValue::NeedsProgress => { - reject_staged_write(running, control, WAKE_URGENT, operation) + reject_staged_write(state, control, WAKE_URGENT, operation) } RegionStageValue::NeedsRotation => { - reject_staged_write(running, control, WAKE_ROTATE | WAKE_URGENT, operation) + reject_staged_write(state, control, WAKE_ROTATE | WAKE_URGENT, operation) } } } @@ -892,28 +895,28 @@ impl RegionDataPlane { "file-chunk key exceeds the 4 KiB limit", )); } - let running = &self.shared; + let state = &self.state; let hash = hash_key(self.data.hash_seed, key); - let activity = running + let activity = state .activity_counters - .then(|| running.metrics.activity_for_hash(hash)); + .then(|| state.metrics.activity_for_hash(hash)); let operation = match self.operations.try_enter() { Some(operation) => operation, None => { - if running.activity_counters { - running.metrics.record_write_rejection(); + if state.activity_counters { + state.metrics.record_write_rejection(); } return Err(write_overload_error()); } }; let Some(seqno) = self.regions.try_delete_value(hash)? else { drop(operation); - if running.activity_counters { - running.metrics.record_write_rejection(); + if state.activity_counters { + state.metrics.record_write_rejection(); } return Err(write_overload_error()); }; - let _removed = running.memory.delete(hash, key, seqno); + let _removed = state.memory.delete(hash, key, seqno); if let Some(activity) = activity { RuntimeMetrics::increment(&activity.deletes); } @@ -921,7 +924,7 @@ impl RegionDataPlane { } #[cfg(test)] - pub fn get(&self, key: &[u8]) -> io::Result> { + pub fn get(&self, key: &[u8]) -> io::Result> { match self.prepare_get(key, None)? { PreparedGet::Complete(value) => Ok(value), PreparedGet::Pending(pending) => self.finish_get(pending.wait(), key), @@ -936,14 +939,14 @@ impl RegionDataPlane { key: &[u8], tokio_handle: &tokio::runtime::Handle, guard: Option<&mut crate::stats::recording::RequestGuard<'_>>, - ) -> io::Result> { + ) -> io::Result> { match self.prepare_get(key, guard)? { PreparedGet::Complete(value) => Ok(value), PreparedGet::Pending(pending) => { self.finish_get(pending.wait_async(tokio_handle).await, key) } PreparedGet::Waiting(waiting) => { - let wait_started = self.runtime.stats.activity_counters.then(Instant::now); + let wait_started = self.options.stats.activity_counters.then(Instant::now); let reserved = waiting.reserve_async(tokio_handle).await; if let Some(wait_started) = wait_started { self.metrics.record_read_wait(wait_started.elapsed()); @@ -958,7 +961,7 @@ impl RegionDataPlane { } fn record_read_wait_error(&self, error: &io::Error) { - if !self.runtime.stats.activity_counters { + if !self.options.stats.activity_counters { return; } if is_read_pressure(error.kind()) { @@ -976,8 +979,8 @@ impl RegionDataPlane { read_token, hash, } = reserved; - let Some(buffer) = self.shared.managed_memory.try_read_buffer(desc.read_len) else { - if self.runtime.stats.activity_counters { + let Some(buffer) = self.state.managed_memory.try_read_buffer(desc.read_len) else { + if self.options.stats.activity_counters { self.metrics.record_read_overload(); } return Err(io::Error::new( @@ -996,7 +999,7 @@ impl RegionDataPlane { hash, })), Err(_) if !self.regions.is_healthy() => { - if self.runtime.stats.activity_counters { + if self.options.stats.activity_counters { RuntimeMetrics::increment(&self.metrics.io_failures); RuntimeMetrics::increment(&self.metrics.activity_for_hash(hash).l2_misses); } @@ -1015,18 +1018,18 @@ impl RegionDataPlane { guard: Option<&mut crate::stats::recording::RequestGuard<'_>>, ) -> io::Result { if key.len() > MAX_KEY_SIZE { - if self.runtime.stats.activity_counters { + if self.options.stats.activity_counters { let activity = self.metrics.activity(0); RuntimeMetrics::increment(&activity.l1_misses); RuntimeMetrics::increment(&activity.l2_misses); } return Ok(PreparedGet::Complete(None)); } - let running = &self.shared; + let state = &self.state; let hash = hash_key(self.data.hash_seed, key); - let activity = running + let activity = state .activity_counters - .then(|| running.metrics.activity_for_hash(hash)); + .then(|| state.metrics.activity_for_hash(hash)); if !self.regions.is_healthy() { if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l1_misses); @@ -1037,13 +1040,13 @@ impl RegionDataPlane { // This health observation is the read's availability linearization // point. A later one-way transition to miss-only does not invalidate a // value that was already resident here. - let read_token = match running.memory.lookup(hash, key) { + let read_token = match state.memory.lookup(hash, key) { MemoryLookup::Hit(value) => { if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l1_hits); RuntimeMetrics::add(&activity.served_bytes, value.len()); } - return Ok(PreparedGet::Complete(Some(HybridValueRead::L1(value)))); + return Ok(PreparedGet::Complete(Some(CacheRead::L1(value)))); } MemoryLookup::Miss(token) => { if let Some(guard) = guard { @@ -1065,14 +1068,14 @@ impl RegionDataPlane { self.data.geometry, hash, candidate, - running.align_reads_for_direct_io, + state.align_reads_for_direct_io, ) { Ok(desc) => desc, Err(error) => { self.regions .enter_miss_only_with_error("record_read_descriptor_invalid", &error); - if running.activity_counters { - RuntimeMetrics::increment(&running.metrics.io_failures); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.io_failures); } if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l2_misses); @@ -1080,17 +1083,17 @@ impl RegionDataPlane { return Ok(PreparedGet::Complete(None)); } }; - let (engine, slot) = match running.try_reserve_read(hash) { + let (engine, slot) = match state.try_reserve_read(hash) { Ok(reservation) => reservation, Err(error) if error.kind() == io::ErrorKind::WouldBlock - && !read_io_wait_timeout(&self.runtime).is_zero() => + && !read_io_wait_timeout(&self.options).is_zero() => { - let waiting = running - .try_queue_read(hash, desc, read_token, read_io_wait_timeout(&self.runtime)) + let waiting = state + .try_queue_read(hash, desc, read_token, read_io_wait_timeout(&self.options)) .inspect_err(|_| { - if running.activity_counters { - running.metrics.record_read_overload(); + if state.activity_counters { + state.metrics.record_read_overload(); } })?; return Ok(PreparedGet::Waiting(waiting)); @@ -1105,8 +1108,8 @@ impl RegionDataPlane { Err(error) => { self.regions .enter_miss_only_with_error("read_engine_reservation_failed", &error); - if running.activity_counters { - RuntimeMetrics::increment(&running.metrics.io_failures); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.io_failures); } if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l2_misses); @@ -1114,10 +1117,10 @@ impl RegionDataPlane { return Ok(PreparedGet::Complete(None)); } }; - let Some(buffer) = running.managed_memory.try_read_buffer(desc.read_len) else { - if !read_io_wait_timeout(&self.runtime).is_zero() { - if running.activity_counters { - running.metrics.record_read_overload(); + let Some(buffer) = state.managed_memory.try_read_buffer(desc.read_len) else { + if !read_io_wait_timeout(&self.options).is_zero() { + if state.activity_counters { + state.metrics.record_read_overload(); } return Err(io::Error::new( io::ErrorKind::OutOfMemory, @@ -1146,15 +1149,15 @@ impl RegionDataPlane { // overload remains explicit while the regions is still healthy. Err(_) if !self.regions.is_healthy() => { if let Some(activity) = activity { - RuntimeMetrics::increment(&running.metrics.io_failures); + RuntimeMetrics::increment(&state.metrics.io_failures); RuntimeMetrics::increment(&activity.l2_misses); } Ok(PreparedGet::Complete(None)) } Err(error) if is_read_pressure(error.kind()) => { - if !read_io_wait_timeout(&self.runtime).is_zero() { - if running.activity_counters { - running.metrics.record_read_overload(); + if !read_io_wait_timeout(&self.options).is_zero() { + if state.activity_counters { + state.metrics.record_read_overload(); } return Err(error); } @@ -1165,33 +1168,29 @@ impl RegionDataPlane { Ok(PreparedGet::Complete(None)) } Err(error) => { - if running.activity_counters { - RuntimeMetrics::increment(&running.metrics.io_failures); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.io_failures); } Err(error) } } } - fn finish_get( - &self, - completed: CompletedGet, - key: &[u8], - ) -> io::Result> { - let running = &self.shared; + fn finish_get(&self, completed: CompletedGet, key: &[u8]) -> io::Result> { + let state = &self.state; let CompletedGet { read, read_token, hash, } = completed; - let activity = running + let activity = state .activity_counters - .then(|| running.metrics.activity_for_hash(hash)); + .then(|| state.metrics.activity_for_hash(hash)); let result = self.regions.finish_value_read(read, key); match result { Err(_) if !self.regions.is_healthy() => { if let Some(activity) = activity { - RuntimeMetrics::increment(&running.metrics.io_failures); + RuntimeMetrics::increment(&state.metrics.io_failures); RuntimeMetrics::increment(&activity.l2_misses); } Ok(None) @@ -1202,16 +1201,16 @@ impl RegionDataPlane { RuntimeMetrics::add(&activity.served_bytes, value.value().len()); } let promoted = - running + state .memory .promote(read_token, hash, key, value.value(), value.seqno()); if let Some(promoted) = promoted { if let Some(activity) = activity { RuntimeMetrics::increment(&activity.l1_promotions); } - return Ok(Some(HybridValueRead::PromotedL2(promoted))); + return Ok(Some(CacheRead::PromotedL2(promoted))); } - Ok(Some(HybridValueRead::L2(value))) + Ok(Some(CacheRead::L2(value))) } Ok(None) => { if let Some(activity) = activity { @@ -1220,7 +1219,7 @@ impl RegionDataPlane { Ok(None) } Err(error) if is_read_pressure(error.kind()) => { - if !read_io_wait_timeout(&self.runtime).is_zero() { + if !read_io_wait_timeout(&self.options).is_zero() { self.record_read_wait_error(&error); return Err(error); } @@ -1231,8 +1230,8 @@ impl RegionDataPlane { Ok(None) } Err(error) => { - if running.activity_counters { - RuntimeMetrics::increment(&running.metrics.io_failures); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.io_failures); } Err(error) } @@ -1246,59 +1245,59 @@ impl RegionDataPlane { let operations = self.operations.begin_drain()?; operations.wait()?; let _draining = LifecycleDrainingGuard::enter(&self.metrics.lifecycle, &self.operations); - let running = &self.shared; - drain_shards(running, false) + let state = &self.state; + drain_shards(state, false) } pub async fn drain_async(&self) -> io::Result<()> { let operations = self.operations.begin_drain()?; operations.wait_async().await; let _draining = LifecycleDrainingGuard::enter(&self.metrics.lifecycle, &self.operations); - let running = &self.shared; - drain_shards_async(running, false).await + let state = &self.state; + drain_shards_async(state, false).await } pub fn snapshot(&self) -> io::Result { - let running = &self.shared; - Ok(self.snapshot_running(running)) + let state = &self.state; + Ok(self.snapshot_state(state)) } pub fn detailed_snapshot(&self) -> io::Result { - let running = &self.shared; + let state = &self.state; Ok(DetailedCacheSnapshot { - summary: self.snapshot_running(running), - write_buffer_rejections: running + summary: self.snapshot_state(state), + write_buffer_rejections: state .metrics .write_buffer_rejections .load(Ordering::Relaxed), - l1: running.memory.detailed_snapshot()?, + l1: state.memory.detailed_snapshot()?, index: self.regions.index_snapshot()?, region: self.regions.region_snapshot()?, }) } - fn snapshot_running(&self, running: &RunningShared) -> CacheSnapshot { + fn snapshot_state(&self, state: &RuntimeState) -> CacheSnapshot { let mut snapshot = self.metrics.snapshot( self.regions.is_healthy(), - self.runtime.stats.activity_counters, - running.managed_memory.snapshot(), - running.memory.metrics_snapshot(), + self.options.stats.activity_counters, + state.managed_memory.snapshot(), + state.memory.metrics_snapshot(), ); if snapshot.health == crate::snapshot::CacheHealth::Running - && running.recovery.is_recovering() + && state.recovery.is_recovering() { snapshot.health = crate::snapshot::CacheHealth::Recovering; } snapshot.io = aggregate_io_stats( - &running.read_engines, - &running.write_engines, - &running.reclaim_engines, + &state.read_engines, + &state.write_engines, + &state.reclaim_engines, ); snapshot } /// Fences admission, drains all workers, and shuts down the I/O engine. - /// The return value asks the io to retain flock for process lifetime + /// A true return value requires the session to retain flock for process lifetime /// because an issued write or flush could not be fenced. pub fn shutdown(&self) -> io::Result { self.start_close(); @@ -1309,27 +1308,31 @@ impl RegionDataPlane { Ordering::AcqRel, Ordering::Acquire, ); - let running = self - .owner + let workers = self + .workers .lock() .map_err(|_| poisoned_runtime_error())? .take() .ok_or_else(closed_runtime_error)?; - let retain_lock = stop_running(running)?; + let retain_lock = stop_workers(workers)?; Ok(retain_lock) } #[cfg(test)] pub fn reserve_read_slot_for_test(&self) -> ReadSlot { - self.shared + self.state .try_reserve_read(0) .map(|(_, slot)| slot) .expect("test read slot is available") } #[cfg(test)] - pub fn poison_shard_for_test(&self, shard_id: usize) { - let shard = self.shared.shards.get(shard_id).expect("test shard exists"); + pub fn poison_append_worker_for_test(&self, shard_id: usize) { + let shard = self + .state + .append_controls + .get(shard_id) + .expect("test shard exists"); let result = panic::catch_unwind(AssertUnwindSafe(|| { let _state = shard.state.lock().unwrap(); panic!("poison shard gate"); @@ -1391,18 +1394,18 @@ fn add_io_direction(aggregate: &mut CacheIoDirectionSnapshot, snapshot: CacheIoD .saturating_add(snapshot.request_time_ns); } -fn start_running( +fn start_workers( regions: Arc, data: DataSuperblock, handles: DataFileHandles, config: CacheConfig, metrics: Arc, operations: Arc, -) -> io::Result { +) -> io::Result { let shard_count = regions.shard_count(); - let runtime = config.runtime(); + let options = config.runtime(); let l1_entry_capacity = l1_entry_capacity(&config); - let memory_limit = runtime.managed_memory_limit_bytes; + let memory_limit = options.managed_memory_limit_bytes; let managed_memory = Arc::new( ManagedMemory::try_new(ManagedMemoryLimits { memory_limit_bytes: memory_limit, @@ -1423,13 +1426,13 @@ fn start_running( .map_err(managed_memory_io_error)?, ); let memory = Arc::new(MemoryStore::new( - runtime.l1_capacity_bytes, + options.l1_capacity_bytes, l1_entry_capacity, - runtime.l1_shards, - runtime.l1_eviction_policy, - runtime.stats.activity_counters, + options.l1_shards, + options.l1_eviction_policy, + options.stats.activity_counters, )?); - let reclaim_worker_count = IoPoolTopology::reclaim(runtime.io_engine).max_in_flight(); + let reclaim_worker_count = IoPoolTopology::reclaim(options.io_engine).max_in_flight(); let mut reclaim_buffers = Vec::new(); reclaim_buffers .try_reserve_exact(reclaim_worker_count) @@ -1453,25 +1456,25 @@ fn start_running( } let reclaim_handles = handles.try_clone()?; let write_handles = handles.try_clone()?; - let read_wait_enabled = !read_io_wait_timeout(runtime).is_zero(); + let read_wait_enabled = !read_io_wait_timeout(options).is_zero(); let read_engines = build_engine_pool( handles, - runtime, - IoPoolTopology::read(runtime.io_engine), + options, + IoPoolTopology::read(options.io_engine), read_wait_enabled, )?; let read_waiters = - read_wait_enabled.then(|| Arc::new(Semaphore::new(read_io_wait_capacity(runtime)))); + read_wait_enabled.then(|| Arc::new(Semaphore::new(read_io_wait_capacity(options)))); let write_engines = build_engine_pool( write_handles, - runtime, - IoPoolTopology::write(runtime.io_engine), + options, + IoPoolTopology::write(options.io_engine), false, )?; let reclaim_engines = build_engine_pool( reclaim_handles, - runtime, - IoPoolTopology::reclaim(runtime.io_engine), + options, + IoPoolTopology::reclaim(options.io_engine), false, )?; for (engines, role) in [ @@ -1485,12 +1488,14 @@ fn start_running( } } } - let mut shards = Vec::new(); - shards.try_reserve_exact(shard_count).map_err(|_| { - io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate shard controls") - })?; - shards.resize_with(shard_count, || Arc::new(AppendWorkerControl::new())); - let shared = Arc::new(RunningShared { + let mut append_controls = Vec::new(); + append_controls + .try_reserve_exact(shard_count) + .map_err(|_| { + io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate shard controls") + })?; + append_controls.resize_with(shard_count, || Arc::new(AppendWorkerControl::new())); + let state = Arc::new(RuntimeState { regions, read_engines, read_lane_cursor: AtomicUsize::new(0), @@ -1498,50 +1503,50 @@ fn start_running( write_engines, reclaim_engines, reclaim_control: ReclaimControl::new(), - reclaim_io_timeout: runtime.reclaim_io_timeout, - recovery: BackgroundRecovery::new(runtime.io_recovery_timeout), + reclaim_io_timeout: options.reclaim_io_timeout, + recovery: BackgroundRecovery::new(options.io_recovery_timeout), managed_memory, metrics, memory, staging, operations, - shards: shards.into_boxed_slice(), - write_flush_threshold_bytes: runtime.write_flush_threshold_bytes, - align_reads_for_direct_io: runtime.io_mode == IoMode::Direct, - activity_counters: runtime.stats.activity_counters, + append_controls: append_controls.into_boxed_slice(), + write_flush_threshold_bytes: options.write_flush_threshold_bytes, + align_reads_for_direct_io: options.io_mode == IoMode::Direct, + activity_counters: options.stats.activity_counters, #[cfg(test)] after_io_snapshot: Mutex::new(None), }); // Inspect the recovered queue before workers can contend with foreground // mutations. Fresh caches have no sealed Regions and need no wakeup. - let reclaim_on_start = shared.regions.reclaim_needed()?; + let reclaim_on_start = state.regions.reclaim_needed()?; let mut reclaim_workers = Vec::new(); reclaim_workers .try_reserve_exact(reclaim_worker_count) .map_err(|_| { io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate worker handles") })?; - let mut shard_workers = Vec::new(); - shard_workers.try_reserve_exact(shard_count).map_err(|_| { + let mut append_workers = Vec::new(); + append_workers.try_reserve_exact(shard_count).map_err(|_| { io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate worker handles") })?; for shard_id in 0..shard_count { - let worker_shared = Arc::clone(&shared); + let worker_state = Arc::clone(&state); match std::thread::Builder::new() .name(format!("cache2-shard-{shard_id}")) .stack_size(CACHE_THREAD_STACK_BYTES) - .spawn(move || shard_worker(worker_shared, shard_id)) + .spawn(move || append_worker(worker_state, shard_id)) { - Ok(worker) => shard_workers.push(worker), + Ok(worker) => append_workers.push(worker), Err(error) => { - for shard in &shared.shards { + for shard in &state.append_controls { let _ = shard.request_drain(true); } - for worker in shard_workers { + for worker in append_workers { let _ = worker.join(); } - shared.staging.close(); - for engine in shared.engines() { + state.staging.close(); + for engine in state.engines() { let _ = engine.shutdown(); } return Err(error); @@ -1549,7 +1554,7 @@ fn start_running( } } for (worker_id, buffer) in reclaim_buffers.into_iter().enumerate() { - let reclaim_shared = Arc::clone(&shared); + let reclaim_shared = Arc::clone(&state); match std::thread::Builder::new() .name(format!("cache2-reclaim-{worker_id}")) .stack_size(CACHE_THREAD_STACK_BYTES) @@ -1557,18 +1562,18 @@ fn start_running( { Ok(worker) => reclaim_workers.push(worker), Err(error) => { - let _ = shared.reclaim_control.stop(); + let _ = state.reclaim_control.stop(); for worker in reclaim_workers { let _ = worker.join(); } - for shard in &shared.shards { + for shard in &state.append_controls { let _ = shard.request_drain(true); } - for worker in shard_workers { + for worker in append_workers { let _ = worker.join(); } - shared.staging.close(); - for engine in shared.engines() { + state.staging.close(); + for engine in state.engines() { let _ = engine.shutdown(); } return Err(error); @@ -1576,18 +1581,18 @@ fn start_running( } } if reclaim_on_start { - shared.reclaim_control.notify()?; + state.reclaim_control.notify()?; } - Ok(RunningOwner { - shared, - shard_workers, + Ok(RuntimeWorkers { + state, + append_workers, reclaim_workers, }) } fn build_engine_pool( handles: DataFileHandles, - runtime: &RuntimeOptions, + options: &RuntimeOptions, topology: IoPoolTopology, read_wait_enabled: bool, ) -> io::Result]>> { @@ -1598,35 +1603,35 @@ fn build_engine_pool( .try_reserve_exact(engine_count) .map_err(|_| io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate I/O workers"))?; for engine in 0..engine_count { - let worker_files = if engine + 1 == engine_count { + let worker_handles = if engine + 1 == engine_count { source.take().expect("last I/O worker owns file set") } else { source.as_ref().expect("I/O file set exists").try_clone()? }; engines.push(build_file_engine( - worker_files, + worker_handles, topology.engine_config(engine), - runtime.stats.activity_counters, + options.stats.activity_counters, read_wait_enabled, )?); } Ok(engines.into_boxed_slice()) } -fn shard_worker(shared: Arc, shard_id: usize) { - let control = Arc::clone(&shared.shards[shard_id]); +fn append_worker(state: Arc, shard_id: usize) { + let control = Arc::clone(&state.append_controls[shard_id]); let result = panic::catch_unwind(AssertUnwindSafe(|| { - shard_worker_result(&shared, shard_id, &control) + append_worker_result(&state, shard_id, &control) })); let error = match result { Ok(Ok(())) => return, Ok(Err(error)) => error, Err(_) => io::Error::other("shard worker panicked"), }; - if shared.activity_counters { - RuntimeMetrics::increment(&shared.metrics.io_failures); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.io_failures); } - let first_failure = shared + let first_failure = state .metrics .lifecycle .swap(LIFECYCLE_FAILED, Ordering::AcqRel) @@ -1634,21 +1639,21 @@ fn shard_worker(shared: Arc, shard_id: usize) { if first_failure { log::error!( target: "cache2::health", - event = "cache_shard_worker_failed", + event = "cache_append_worker_failed", shard_id, error:% = error; "cache shard worker failed" ); } - shared.regions.enter_miss_only(); - shared.recovery.stop(); + state.regions.enter_miss_only(); + state.recovery.stop(); control.fail(&error); // Wake engine admission in case another shard is blocked behind work that // can no longer make progress after this runtime entered miss-only. - for engine in shared.engines() { + for engine in state.engines() { engine.wake_slot_waiters(); } - for shard in &shared.shards { + for shard in &state.append_controls { if !Arc::ptr_eq(shard, &control) { shard.fail(&error); } @@ -1686,29 +1691,29 @@ impl ReinsertShardCursor { } fn reclaim_worker( - shared: Arc, + state: Arc, buffer: BufferLease, worker_id: usize, worker_count: usize, ) { let mut buffer = Some(buffer); let result = panic::catch_unwind(AssertUnwindSafe(|| { - reclaim_worker_result(&shared, &mut buffer, worker_id, worker_count) + reclaim_worker_result(&state, &mut buffer, worker_id, worker_count) })); let error = match result { Ok(Ok(())) => return, Ok(Err(error)) => error, Err(_) => io::Error::other("Region reclaim worker panicked"), }; - if shared.activity_counters { - RuntimeMetrics::increment(&shared.metrics.io_failures); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.io_failures); } - shared + state .metrics .lifecycle .store(LIFECYCLE_FAILED, Ordering::Release); - shared.regions.enter_miss_only(); - shared.recovery.stop(); + state.regions.enter_miss_only(); + state.recovery.stop(); log::error!( target: "cache2::health", event = "cache_reclaim_worker_failed", @@ -1716,19 +1721,19 @@ fn reclaim_worker( error:% = error; "cache Region reclaim worker failed" ); - for shard in &shared.shards { + for shard in &state.append_controls { shard.fail(&error); } } fn reclaim_worker_result( - shared: &RunningShared, + state: &RuntimeState, buffer: &mut Option, worker_id: usize, worker_count: usize, ) -> io::Result<()> { - let engine_index = route_hash(worker_id as u64, shared.reclaim_engines.len()); - let engine = shared.reclaim_engines.get(engine_index).ok_or_else(|| { + let engine_index = route_hash(worker_id as u64, state.reclaim_engines.len()); + let engine = state.reclaim_engines.get(engine_index).ok_or_else(|| { io::Error::new( io::ErrorKind::InvalidData, "reclaim worker has no I/O engine", @@ -1736,17 +1741,17 @@ fn reclaim_worker_result( })?; let mut observed_generation = 0_u64; let mut reinsert_shards = - ReinsertShardCursor::new(worker_id, worker_count, shared.shards.len()); - while shared.reclaim_control.wait(&mut observed_generation)? { + ReinsertShardCursor::new(worker_id, worker_count, state.append_controls.len()); + while state.reclaim_control.wait(&mut observed_generation)? { loop { // Finish an already-started victim, but do not begin another once // shutdown has asked the worker to stop. A large clean-reserve // deficit must not turn close into a multi-Region reclaim pass. - if shared.reclaim_control.is_stopped()? { + if state.reclaim_control.is_stopped()? { return Ok(()); } - let mut recovery = shared.recovery.attempt(); - let Some(receipt) = shared.regions.begin_reclaim()? else { + let mut recovery = state.recovery.attempt(); + let Some(receipt) = state.regions.begin_reclaim()? else { break; }; let used = usize::try_from(receipt.used_offset).map_err(|_| { @@ -1763,7 +1768,7 @@ fn reclaim_worker_result( Ok(buffer) => buffer, Err(error) => return Err(error.error), }; - let absolute = shared.regions.reclaim_absolute(receipt)?; + let absolute = state.regions.reclaim_absolute(receipt)?; // Reclaim owns a dedicated pool whose depth matches its worker // count. Use the bounded background wait so transient CAS // contention cannot turn a healthy cache miss-only; foreground @@ -1771,7 +1776,7 @@ fn reclaim_worker_result( let request = submit_background_io( engine.as_ref(), IoOperation::read(io_buffer, absolute), - shared.reclaim_io_timeout, + state.reclaim_io_timeout, &mut recovery, ) .map_err(|error| error.into_lease().0)?; @@ -1809,21 +1814,21 @@ fn reclaim_worker_result( // Keep one completion boundary per source Region while each // reclaimer rotates through a disjoint subset of append shards. let reinsert_shard = reinsert_shards.take(); - let preserve_hot = shared.regions.reclaim_can_reinsert()?; + let preserve_hot = state.regions.reclaim_can_reinsert()?; let reinsert_operation = if preserve_hot { - shared.operations.try_enter() + state.operations.try_enter() } else { None }; let mut accepting_reinserts = reinsert_operation.is_some(); let mut staged_reinsert = false; - let stats = shared.regions.scan_reclaim(receipt, bytes, |record| { + let stats = state.regions.scan_reclaim(receipt, bytes, |record| { if !accepting_reinserts { return Ok(false); } - match shared + match state .regions - .try_stage_reinsert(&shared.staging, reinsert_shard, record)? + .try_stage_reinsert(&state.staging, reinsert_shard, record)? { RegionStageValue::Staged { .. } => { staged_reinsert = true; @@ -1836,14 +1841,14 @@ fn reclaim_worker_result( } })?; if staged_reinsert { - let generation = shared.shards[reinsert_shard].request_drain(false)?; - shared.shards[reinsert_shard].wait_for_drain(generation)?; + let generation = state.append_controls[reinsert_shard].request_drain(false)?; + state.append_controls[reinsert_shard].wait_for_drain(generation)?; } - shared.regions.complete_reclaim(receipt)?; + state.regions.complete_reclaim(receipt)?; recovery.finish(); drop(reinsert_operation); - if shared.activity_counters { - shared.metrics.record_reclaim(stats); + if state.activity_counters { + state.metrics.record_reclaim(stats); } log::debug!( target: "cache2::reclaim", @@ -1861,7 +1866,7 @@ fn reclaim_worker_result( reinsert_budget_skipped = stats.reinsert_budget_skipped; "cache Region reclaimed" ); - for shard in &shared.shards { + for shard in &state.append_controls { shard.notify_if_running(WAKE_ROTATE)?; } } @@ -1869,8 +1874,8 @@ fn reclaim_worker_result( Ok(()) } -fn shard_worker_result( - shared: &RunningShared, +fn append_worker_result( + state: &RuntimeState, shard_id: usize, control: &AppendWorkerControl, ) -> io::Result<()> { @@ -1881,20 +1886,20 @@ fn shard_worker_result( let force_flush = flags & WAKE_URGENT != 0 || timed_out || draining; let rotate = flags & WAKE_ROTATE != 0; - match shared.staging.shard_fill_snapshot(shard_id) { + match state.staging.shard_fill_snapshot(shard_id) { Ok(Some(fill)) => { if deadline.is_none() { deadline = Some(Instant::now().checked_add(WRITE_FLUSH_DELAY).ok_or_else( || invalid_runtime_config("partial flush deadline overflow"), )?); } - if force_flush || fill.bytes >= shared.write_flush_threshold_bytes { - let engine = shared.write_engine_for(shard_id as u64); - shared.regions.flush_staging_shard( - &shared.staging, + if force_flush || fill.bytes >= state.write_flush_threshold_bytes { + let engine = state.write_engine_for(shard_id as u64); + state.regions.flush_staging_shard( + &state.staging, engine.as_ref(), shard_id, - &shared.recovery, + &state.recovery, )?; deadline = None; } @@ -1902,12 +1907,12 @@ fn shard_worker_result( Ok(None) => { deadline = None; if rotate { - let rotated = shared.regions.rotate_shard(shard_id)?; - if rotated && shared.activity_counters { - RuntimeMetrics::increment(&shared.metrics.region_rotations); + let rotated = state.regions.rotate_shard(shard_id)?; + if rotated && state.activity_counters { + RuntimeMetrics::increment(&state.metrics.region_rotations); } if rotated { - shared.reclaim_control.notify()?; + state.reclaim_control.notify()?; } } } @@ -1922,14 +1927,14 @@ fn shard_worker_result( // completion boundary without fencing foreground mutations, so a // short in-progress encode must be retried rather than treated as // structural staging failure. - match shared.staging.shard_fill_snapshot(shard_id) { + match state.staging.shard_fill_snapshot(shard_id) { Ok(Some(_)) => { - let engine = shared.write_engine_for(shard_id as u64); - shared.regions.flush_staging_shard( - &shared.staging, + let engine = state.write_engine_for(shard_id as u64); + state.regions.flush_staging_shard( + &state.staging, engine.as_ref(), shard_id, - &shared.recovery, + &state.recovery, )?; } Ok(None) => {} @@ -1991,16 +1996,16 @@ fn wait_for_shard_work( } fn reject_staged_write( - running: &RunningShared, + state: &RuntimeState, control: &AppendWorkerControl, flags: u8, operation: MutationGuard<'_>, ) -> io::Result { control.notify(flags)?; drop(operation); - if running.activity_counters { - RuntimeMetrics::increment(&running.metrics.write_buffer_rejections); - running.metrics.record_write_rejection(); + if state.activity_counters { + RuntimeMetrics::increment(&state.metrics.write_buffer_rejections); + state.metrics.record_write_rejection(); } Err(write_overload_error()) } @@ -2014,13 +2019,13 @@ fn complete_shard_drain(control: &AppendWorkerControl, generation: u64) -> io::R Ok(()) } -fn drain_shards(shared: &RunningShared, stop: bool) -> io::Result<()> { +fn drain_shards(state: &RuntimeState, stop: bool) -> io::Result<()> { let mut generations = Vec::new(); generations - .try_reserve_exact(shared.shards.len()) + .try_reserve_exact(state.append_controls.len()) .map_err(|_| io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate drain fence"))?; let mut first_error = None; - for shard in &shared.shards { + for shard in &state.append_controls { match shard.request_drain(stop) { Ok(generation) => generations.push(Some(generation)), Err(error) => { @@ -2029,7 +2034,7 @@ fn drain_shards(shared: &RunningShared, stop: bool) -> io::Result<()> { } } } - for (shard, generation) in shared.shards.iter().zip(generations) { + for (shard, generation) in state.append_controls.iter().zip(generations) { if let Some(generation) = generation && let Err(error) = shard.wait_for_drain(generation) { @@ -2039,13 +2044,13 @@ fn drain_shards(shared: &RunningShared, stop: bool) -> io::Result<()> { first_error.map_or(Ok(()), Err) } -async fn drain_shards_async(shared: &RunningShared, stop: bool) -> io::Result<()> { +async fn drain_shards_async(state: &RuntimeState, stop: bool) -> io::Result<()> { let mut generations = Vec::new(); generations - .try_reserve_exact(shared.shards.len()) + .try_reserve_exact(state.append_controls.len()) .map_err(|_| io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate drain fence"))?; let mut first_error = None; - for shard in &shared.shards { + for shard in &state.append_controls { match shard.request_drain(stop) { Ok(generation) => generations.push(Some(generation)), Err(error) => { @@ -2054,7 +2059,7 @@ async fn drain_shards_async(shared: &RunningShared, stop: bool) -> io::Result<() } } } - for (shard, generation) in shared.shards.iter().zip(generations) { + for (shard, generation) in state.append_controls.iter().zip(generations) { if let Some(generation) = generation && let Err(error) = shard.wait_for_drain_async(generation).await { @@ -2064,49 +2069,49 @@ async fn drain_shards_async(shared: &RunningShared, stop: bool) -> io::Result<() first_error.map_or(Ok(()), Err) } -fn stop_running(mut owner: RunningOwner) -> io::Result { - owner.shared.recovery.stop(); - let drain = drain_shards(&owner.shared, true); +fn stop_workers(mut workers: RuntimeWorkers) -> io::Result { + workers.state.recovery.stop(); + let drain = drain_shards(&workers.state, true); let mut join_error = None; - for worker in owner.shard_workers.drain(..) { + for worker in workers.append_workers.drain(..) { if worker.join().is_err() { join_error.get_or_insert_with(|| io::Error::other("shard worker panicked")); } } - if let Err(error) = owner.shared.reclaim_control.stop() { + if let Err(error) = workers.state.reclaim_control.stop() { join_error.get_or_insert(error); } - for worker in owner.reclaim_workers.drain(..) { + for worker in workers.reclaim_workers.drain(..) { if worker.join().is_err() { join_error.get_or_insert_with(|| io::Error::other("Region reclaim worker panicked")); } } - owner.shared.staging.close(); + workers.state.staging.close(); // Fence submission before observing idle engines. A read that already // passed the public close check must not appear between this snapshot // and a synchronous engine shutdown. - for engine in owner.shared.engines() { + for engine in workers.state.engines() { engine.stop_accepting_requests(); } - let in_flight = owner - .shared + let in_flight = workers + .state .engines() .map(|engine| engine.in_flight()) .sum::(); #[cfg(test)] { - let after_snapshot = owner.shared.after_io_snapshot.lock().unwrap().take(); + let after_snapshot = workers.state.after_io_snapshot.lock().unwrap().take(); if let Some(after_snapshot) = after_snapshot { after_snapshot(); } } - let writes_in_flight = owner - .shared + let writes_in_flight = workers + .state .engines() .map(|engine| engine.writes_in_flight()) .sum::(); - let unfenced_before = owner - .shared + let unfenced_before = workers + .state .engines() .any(|engine| engine.has_unfenced_writes()); // A request that missed its cancellation grace may still own a kernel @@ -2117,7 +2122,7 @@ fn stop_running(mut owner: RunningOwner) -> io::Result { Ok(()) } else { let mut result = Ok(()); - for engine in owner.shared.engines() { + for engine in workers.state.engines() { if let Err(error) = engine.shutdown() && result.is_ok() { @@ -2127,8 +2132,8 @@ fn stop_running(mut owner: RunningOwner) -> io::Result { result }; let unfenced = unfenced_before - || owner - .shared + || workers + .state .engines() .any(|engine| engine.has_unfenced_writes()); let result = drain @@ -2140,11 +2145,11 @@ fn stop_running(mut owner: RunningOwner) -> io::Result { // reclaims its fd/thread/buffer set. A sticky fatal unfenced write // has no trustworthy future fence and remains process-lifetime state. if unfenced { - for engine in owner.shared.engines() { + for engine in workers.state.engines() { mem::forget(Arc::clone(engine)); } } else { - for engine in owner.shared.engines() { + for engine in workers.state.engines() { if engine.in_flight() != 0 { reap_engine_after_target_fence(engine); } else { @@ -2170,7 +2175,7 @@ fn reap_engine_after_target_fence(engine: &Arc) { let _ = reaper_engine.shutdown(); }); if spawn.is_err() { - // The original owner is still alive while this fallback clone is + // The original workers is still alive while this fallback clone is // created, so a failed thread spawn cannot synchronously run the // engine's blocking Drop path. mem::forget(Arc::clone(engine)); @@ -2201,7 +2206,7 @@ fn poisoned_runtime_error() -> io::Error { } fn closed_runtime_error() -> io::Error { - io::Error::new(io::ErrorKind::NotConnected, "data plane is closed") + io::Error::new(io::ErrorKind::NotConnected, "cache runtime is closed") } fn invalid_runtime_config(message: &'static str) -> io::Error { @@ -2505,10 +2510,9 @@ mod tests { use crate::cache::session::CacheSession; use crate::config::runtime::IoEngineOptions; use crate::config::runtime::PosixIoOptions; - use crate::region::file_backend::FileRegionBackend; - use crate::region::file_backend::RegionPaths; use crate::region::index::packed::IndexEntry; use crate::region::index::packed::PackedLocation; + use crate::region::persistence::RegionPaths; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::PersistentId; @@ -2557,16 +2561,13 @@ mod tests { }, ..RuntimeOptions::default() }; - let mut store = CacheSession::open( - 8, - FileRegionBackend::for_test_with_options(paths.clone(), data, 8, config), - ) - .unwrap(); - let plane = store.data_plane_handle().unwrap(); - let MemoryLookup::Miss(read_token) = plane.shared.memory.lookup(7, b"key") else { + let mut session = + CacheSession::for_test_with_options(paths.clone(), data, 8, config).unwrap(); + let runtime = session.runtime().unwrap().clone(); + let MemoryLookup::Miss(read_token) = runtime.state.memory.lookup(7, b"key") else { panic!("empty L1 must miss"); }; - let result = plane.finish_get( + let result = runtime.finish_get( CompletedGet { read: ReadCompletion { desc: ReadDesc { @@ -2587,9 +2588,9 @@ mod tests { }, b"key", ); - let snapshot = plane.snapshot().unwrap(); - assert!(plane.regions.is_healthy()); - store.close_fast().unwrap(); + let snapshot = runtime.snapshot().unwrap(); + assert!(runtime.regions.is_healthy()); + session.close_fast().unwrap(); if wait.is_zero() { assert!(matches!(result, Ok(None))); assert_eq!(snapshot.l2_read_busy_misses, 1); diff --git a/cache2/src/region/runtime/shutdown_tests.rs b/cache2/src/cache/runtime/shutdown_tests.rs similarity index 77% rename from cache2/src/region/runtime/shutdown_tests.rs rename to cache2/src/cache/runtime/shutdown_tests.rs index eb34ab5..33aa90b 100644 --- a/cache2/src/region/runtime/shutdown_tests.rs +++ b/cache2/src/cache/runtime/shutdown_tests.rs @@ -80,8 +80,7 @@ fn submitted_read_must_not_pin_close() { fn assert_close_does_not_wait_for_read(submit_before_close: bool) { use crate::cache::session::CacheSession; use crate::config::runtime::PosixIoOptions; - use crate::region::file_backend::FileRegionBackend; - use crate::region::file_backend::RegionPaths; + use crate::region::persistence::RegionPaths; use crate::region::recovery::PersistentId; let root = env::temp_dir().join(format!( "cache2-close-race-{}-{submit_before_close}", @@ -111,20 +110,16 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { }), ..RuntimeOptions::default() }; - let mut store = CacheSession::open( - 8, - FileRegionBackend::for_test_with_options(paths, data, 8, config), - ) - .unwrap(); - let mut plane = store.data_plane_handle().unwrap(); - store.close_fast().unwrap(); + let mut session = CacheSession::for_test_with_options(paths, data, 8, config).unwrap(); + let mut runtime = session.runtime().unwrap().clone(); + session.close_fast().unwrap(); // Reuse a stopped runtime's fixed resources without unrelated workers. - let shared = Arc::get_mut(&mut plane.shared).unwrap(); + let state = Arc::get_mut(&mut runtime.state).unwrap(); let io = Arc::new(BlockedRead::default()); let engine = Arc::new(IoEngine::for_test(io.clone(), 1).unwrap()); let read_engine = Arc::clone(&engine); let read_io = Arc::clone(&io); - let managed_memory = Arc::clone(&shared.managed_memory); + let managed_memory = Arc::clone(&state.managed_memory); let submit_read = move || -> io::Result { let slot = read_engine.try_reserve_read()?; let buffer = @@ -141,20 +136,20 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { assert_eq!(engine.in_flight(), 1); } else { // Exercise the interval between the idle snapshot and synchronous shutdown. - *shared.after_io_snapshot.get_mut().unwrap() = Some(Box::new(move || { + *state.after_io_snapshot.get_mut().unwrap() = Some(Box::new(move || { submitted.send(submit_read()).unwrap(); })); } - shared.read_engines = vec![engine.clone()].into_boxed_slice(); - shared.write_engines = Box::new([]); - shared.reclaim_engines = Box::new([]); - shared.shards = Box::new([]); - let shared = Arc::clone(&plane.shared); + state.read_engines = vec![engine.clone()].into_boxed_slice(); + state.write_engines = Box::new([]); + state.reclaim_engines = Box::new([]); + state.append_controls = Box::new([]); + let state = Arc::clone(&runtime.state); let (tx, rx) = mpsc::channel(); let thread = std::thread::spawn(move || { - let result = stop_running(RunningOwner { - shared, - shard_workers: vec![], + let result = stop_workers(RuntimeWorkers { + state, + append_workers: vec![], reclaim_workers: vec![], }); tx.send(result).unwrap(); @@ -182,8 +177,7 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { #[test] fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { use crate::cache::session::CacheSession; - use crate::region::file_backend::FileRegionBackend; - use crate::region::file_backend::RegionPaths; + use crate::region::persistence::RegionPaths; use crate::region::recovery::PersistentId; use crate::snapshot::CacheHealth; let root = env::temp_dir().join(format!("cache2-recovery-admission-{}", std::process::id())); @@ -206,37 +200,33 @@ fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { l1_capacity_bytes: 0, ..RuntimeOptions::default() }; - let mut store = CacheSession::open( - 8, - FileRegionBackend::for_test_with_options(paths, data, 8, config), - ) - .unwrap(); - let plane = store.data_plane_handle().unwrap(); - plane.put(b"existing", b"value").unwrap(); - plane.drain().unwrap(); - let mut first = plane.shared.recovery.attempt(); - let mut second = plane.shared.recovery.attempt(); + let mut session = CacheSession::for_test_with_options(paths, data, 8, config).unwrap(); + let runtime = session.runtime().unwrap().clone(); + runtime.put(b"existing", b"value").unwrap(); + runtime.drain().unwrap(); + let mut first = runtime.state.recovery.attempt(); + let mut second = runtime.state.recovery.attempt(); assert!(first.next_deadline(Instant::now()).is_some()); assert!(second.next_deadline(Instant::now()).is_some()); - assert_eq!(plane.snapshot().unwrap().health, CacheHealth::Recovering); + assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Recovering); assert_eq!( - plane.put(b"new", b"value").unwrap_err().kind(), + runtime.put(b"new", b"value").unwrap_err().kind(), io::ErrorKind::WouldBlock ); assert_eq!( - plane.put_l2(b"new", b"value").unwrap_err().kind(), + runtime.put_l2(b"new", b"value").unwrap_err().kind(), io::ErrorKind::WouldBlock ); - assert_eq!(plane.get(b"existing").unwrap().unwrap().value(), b"value"); - plane.delete(b"existing").unwrap(); - assert!(plane.get(b"existing").unwrap().is_none()); + assert_eq!(runtime.get(b"existing").unwrap().unwrap().value(), b"value"); + runtime.delete(b"existing").unwrap(); + assert!(runtime.get(b"existing").unwrap().is_none()); first.finish(); - assert_eq!(plane.snapshot().unwrap().health, CacheHealth::Recovering); + assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Recovering); second.finish(); - assert_eq!(plane.snapshot().unwrap().health, CacheHealth::Running); + assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Running); let deadline = Instant::now() + Duration::from_secs(2); loop { - match plane.put(b"new", b"value") { + match runtime.put(b"new", b"value") { Ok(_) => break, Err(error) => { assert_eq!(error.kind(), io::ErrorKind::WouldBlock); @@ -245,8 +235,8 @@ fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { } } } - plane.drain().unwrap(); - assert_eq!(plane.get(b"new").unwrap().unwrap().value(), b"value"); - store.close_fast().unwrap(); + runtime.drain().unwrap(); + assert_eq!(runtime.get(b"new").unwrap().unwrap().value(), b"value"); + session.close_fast().unwrap(); std::fs::remove_dir_all(root).unwrap(); } diff --git a/cache2/src/cache/session.rs b/cache2/src/cache/session.rs index 3c03e0d..cc6b788 100644 --- a/cache2/src/cache/session.rs +++ b/cache2/src/cache/session.rs @@ -12,90 +12,144 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Recovery and shutdown state machine for the Region-backed cache. +//! Owns one open cache session and orders restart recovery, execution, and close. //! -//! The coordinator owns no file-format or data-plane logic. A -//! [`FileRegionBackend`] supplies those operations, while this module enforces the -//! order that makes warm recovery safe: -//! -//! - inspect recovery before constructing an index; -//! - publish `RUNNING` before exposing a runtime; -//! - publish `CLEAN` only after freezing and persisting a complete image; -//! - release exclusive ownership on every terminal path. +//! RUNNING is durable before workers start. CLEAN is published only after +//! mutation quiescence and a durable recovery image. Every terminal path +//! releases file ownership unless an issued write still requires the lock. use std::io; - +use std::sync::Arc; + +use crate::cache::runtime::CacheRuntime; +use crate::config::CacheConfig; +#[cfg(test)] +use crate::config::runtime::RuntimeOptions; +#[cfg(test)] +use crate::config::storage::cache_config; use crate::io::fs::FileSystem; use crate::io::fs::OsFileSystem; -use crate::region::file_backend::FileRegionBackend; -use crate::region::file_backend::FileRegionRuntime; -use crate::region::index::storage::validated_index_partition_ranges; -use crate::region::runtime::RegionDataPlane; +#[cfg(test)] +use crate::region::RegionStore; +use crate::region::persistence::RegionPaths; +use crate::region::persistence::RegionPersistence; +use crate::region::recovery::DataSuperblock; use crate::snapshot::StartupMode; -/// Owns the files and runtime for one Region-backed cache. pub struct CacheSession { - backend: FileRegionBackend, - runtime: Option, + persistence: RegionPersistence, + runtime: Option, startup: StartupMode, closed: bool, } +impl CacheSession { + pub fn open( + paths: RegionPaths, + format_data: DataSuperblock, + config: CacheConfig, + ) -> io::Result { + Self::open_with_file_system(paths, format_data, config, OsFileSystem) + } + + #[cfg(test)] + pub fn for_test( + paths: RegionPaths, + data: DataSuperblock, + index_slots: usize, + ) -> io::Result { + Self::for_test_with_options(paths, data, index_slots, RuntimeOptions::default()) + } + + #[cfg(test)] + pub fn for_test_with_options( + paths: RegionPaths, + data: DataSuperblock, + index_slots: usize, + options: RuntimeOptions, + ) -> io::Result { + Self::open( + paths, + data, + cache_config(data.geometry, index_slots, options), + ) + } +} + impl CacheSession { - pub fn open(index_slots: usize, mut backend: FileRegionBackend) -> io::Result { - validate_index_slots(index_slots)?; - backend.acquire_exclusive()?; + fn open_with_file_system( + paths: RegionPaths, + format_data: DataSuperblock, + config: CacheConfig, + file_system: F, + ) -> io::Result { + let index_slots = config.storage().index_slots(); + let append_shards = config.runtime().append_shards; + let mut persistence = RegionPersistence::new(paths, format_data, file_system); + persistence.acquire_exclusive(config.runtime().io_mode)?; let opened = (|| { - let runtime = match backend.inspect_recovery(index_slots)? { - Some(clean) => backend.map_clean_runtime(clean, index_slots)?, + let regions = match persistence.inspect_recovery(index_slots)? { + Some(image) => persistence.recover_regions(image, index_slots, append_shards)?, None => None, }; - let (runtime, startup) = match runtime { - Some(runtime) => (runtime, StartupMode::Warm), - None => (backend.anonymous_runtime(index_slots)?, StartupMode::Cold), + let (regions, startup) = match regions { + Some(regions) => (regions, StartupMode::Warm), + None => ( + persistence.cold_regions(index_slots, append_shards)?, + StartupMode::Cold, + ), }; - - backend.publish_running()?; - let runtime = backend.start_runtime(runtime)?; + persistence.publish_running()?; + let runtime = CacheRuntime::start( + Arc::new(regions), + persistence.data_superblock()?, + persistence.clone_data_handles()?, + config, + )?; Ok((runtime, startup)) })(); match opened { Ok((runtime, startup)) => Ok(Self { - backend, + persistence, runtime: Some(runtime), startup, closed: false, }), Err(error) => { - let _ = backend.release_exclusive(); + let _ = persistence.release_exclusive(); Err(error) } } } - pub const fn startup(&self) -> StartupMode { - self.startup + #[cfg(test)] + pub fn for_test_with_file_system( + paths: RegionPaths, + data: DataSuperblock, + index_slots: usize, + file_system: F, + ) -> io::Result { + Self::open_with_file_system( + paths, + data, + cache_config(data.geometry, index_slots, RuntimeOptions::default()), + file_system, + ) } - pub fn data_plane_handle(&self) -> io::Result { - Ok(self.runtime()?.data_plane()?.clone()) + pub const fn startup(&self) -> StartupMode { + self.startup } - pub fn runtime(&self) -> io::Result<&FileRegionRuntime> { - if self.closed { - return Err(closed_error()); - } + pub fn runtime(&self) -> io::Result<&CacheRuntime> { self.runtime.as_ref().ok_or_else(closed_error) } #[cfg(test)] - pub fn runtime_mut(&mut self) -> io::Result<&mut FileRegionRuntime> { - if self.closed { - return Err(closed_error()); - } - self.runtime.as_mut().ok_or_else(closed_error) + pub fn regions(&self) -> io::Result<&RegionStore> { + Ok(self.runtime()?.regions()) } /// Stop without producing a recovery image. The next open starts empty. @@ -112,18 +166,38 @@ impl CacheSession { if self.closed { return Ok(()); } - let result = match self.runtime.take() { - Some(runtime) if warm => self - .backend - .freeze_warm(runtime) - .and_then(|frozen| self.backend.persist_frozen(&frozen)) - .and_then(|prepared| self.backend.publish_clean(prepared)), - Some(runtime) => self.backend.stop_fast(runtime), + Some(runtime) => (|| { + let regions = Arc::clone(runtime.regions()); + let stopped = runtime.shutdown(); + drop(runtime); + match stopped { + Ok(false) => {} + Ok(true) => { + self.persistence.retain_data_lock(); + let message = if warm { + regions.enter_miss_only(); + "I/O engine could not fence an issued write; CLEAN rejected" + } else { + "I/O engine could not fence an issued write; lock retained" + }; + return Err(io::Error::other(message)); + } + Err(error) => { + regions.enter_miss_only(); + return Err(error); + } + } + if warm { + let frozen = regions.freeze()?; + let prepared = self.persistence.persist_frozen(&frozen)?; + self.persistence.publish_clean(prepared)?; + } + Ok(()) + })(), None => Err(closed_error()), }; - - let unlock = self.backend.release_exclusive(); + let unlock = self.persistence.release_exclusive(); self.closed = true; result.and(unlock) } @@ -138,17 +212,5 @@ impl Drop for CacheSession { } fn closed_error() -> io::Error { - io::Error::new(io::ErrorKind::BrokenPipe, "CacheSession is closed") -} - -fn validate_index_slots(index_slots: usize) -> io::Result<()> { - if index_slots < 8 { - return Err(io::Error::new( - io::ErrorKind::InvalidInput, - "CacheSession requires at least 8 index slots", - )); - } - validated_index_partition_ranges(index_slots) - .map(|_| ()) - .map_err(|error| io::Error::new(io::ErrorKind::InvalidInput, error)) + io::Error::new(io::ErrorKind::BrokenPipe, "cache session is closed") } diff --git a/cache2/src/config/runtime.rs b/cache2/src/config/runtime.rs index cab9dde..454b4a4 100644 --- a/cache2/src/config/runtime.rs +++ b/cache2/src/config/runtime.rs @@ -17,6 +17,7 @@ use std::time::Duration; use std::time::Instant; use crate::StatsOptions; +use crate::cache::runtime::metrics::ActivityMetrics; use crate::config::CacheConfig; use crate::config::StorageLayout; use crate::error::Error; @@ -30,7 +31,6 @@ use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::MAX_CONFIG_COUNT; use crate::memory::MemoryStore; use crate::region::recovery::DataGeometry; -use crate::region::runtime::metrics::ActivityMetrics; use crate::region::runtime_fixed_memory_bytes; use crate::region::staging::AppendStaging; use crate::stats::recording::Recorder; diff --git a/cache2/src/region/mod.rs b/cache2/src/region/mod.rs index 28e6593..0265b13 100644 --- a/cache2/src/region/mod.rs +++ b/cache2/src/region/mod.rs @@ -63,6 +63,9 @@ use crate::region::appender::RegionSpanCompletion; #[cfg(test)] use crate::region::index::packed::IndexEntry; use crate::region::index::packed::PackedLocation; +use crate::region::index::storage::IndexPartitionRange; +use crate::region::index::storage::IndexPhysicalStats; +use crate::region::index::storage::PartitionedIndexStorage; use crate::region::index::storage::page_format::INDEX_IMAGE_PAGE_SIZE; use crate::region::manager::RegionWriteSpan; use crate::region::record::codec::RecordEncodeError; @@ -72,24 +75,25 @@ use crate::region::record::codec::encode_value_into_hashed; #[cfg(test)] use crate::region::record::codec::hash_key; use crate::region::recovery::DataGeometry; +use crate::region::recovery::metadata::PartitionMetadataRecord; use crate::region::recovery::metadata::REGION_METADATA_PAGE_SIZE; use crate::region::recovery::metadata::REGION_METADATA_PARTITIONS_PER_PAGE; use crate::region::recovery::metadata::REGION_METADATA_REGIONS_PER_PAGE; +use crate::region::recovery::metadata::RegionMetadata; use crate::region::recovery::metadata::RegionMetadataError; use crate::region::staging::AppendStaging; use crate::snapshot::CacheIndexSnapshot; use crate::snapshot::RegionSnapshot; -pub mod file_backend; pub mod index; pub mod manager; +pub mod persistence; pub mod record; pub mod recovery; -pub mod runtime; pub mod staging; -mod appender; -mod reader; +pub mod appender; +pub mod reader; const REGION_HEALTHY: u8 = 0; const REGION_MISS_ONLY: u8 = 1; @@ -210,6 +214,12 @@ impl RegionManagerLock { } } +pub struct FrozenRegionStore { + regions: Arc, + metadata: RegionMetadata, +} + +/// Live L2 index, Region allocation, and append/read/reclaim coordination. pub struct RegionStore { index: RegionIndex, manager: RegionManagerLock, @@ -258,7 +268,7 @@ pub enum RegionStageValue { NeedsRotation, } -pub struct RegionValueRead { +pub struct RegionValue { buffer: BufferLease, buffer_len: usize, value_range: Range, @@ -268,9 +278,9 @@ pub struct RegionValueRead { // SAFETY: this type is constructed only after the owned record read reaches // terminal completion. From then until drop, it exposes initialized bytes only // through shared slices and never returns the allocation to a mutable pool. -unsafe impl Sync for RegionValueRead {} +unsafe impl Sync for RegionValue {} -impl RegionValueRead { +impl RegionValue { pub fn value(&self) -> &[u8] { &self .buffer @@ -284,6 +294,86 @@ impl RegionValueRead { } impl RegionStore { + /// Consume the index and Region metadata into one live store. + pub fn from_recovery( + index: PartitionedIndexStorage, + metadata: RegionMetadata, + ) -> io::Result { + let physical_stats = index.partition_stats().map_err(index_storage_io_error)?; + let slot_count = u64::try_from(index.slot_count()).map_err(|_| { + io::Error::new(io::ErrorKind::InvalidData, "index capacity is too large") + })?; + if metadata.root.index_slots != slot_count + || metadata.root.partition_count as usize != index.partition_count() + || !metadata_partition_stats_match(&metadata, &physical_stats) + { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "index and Region metadata do not describe one authority", + )); + } + let manager = RegionManager::from_metadata(metadata).map_err(region_metadata_io_error)?; + let mut append_gates = Vec::new(); + append_gates + .try_reserve_exact(manager.active_regions().len()) + .map_err(|_| { + io::Error::new( + io::ErrorKind::OutOfMemory, + "cannot allocate append shard gates", + ) + })?; + append_gates.resize_with(manager.active_regions().len(), AppendShardGate::default); + let mut region_generations = Vec::new(); + region_generations + .try_reserve_exact(manager.regions().len()) + .map_err(|_| { + io::Error::new( + io::ErrorKind::OutOfMemory, + "cannot allocate Region generations", + ) + })?; + for region in manager.regions() { + region_generations.push(AtomicU64::new(region.created_seqno)); + } + let health = RegionHealthLatch::healthy(); + let index = RegionIndex::from_storage(index).map_err(index_storage_io_error)?; + Ok(Self { + index, + manager: RegionManagerLock::new(manager, health.clone()), + append_gates: append_gates.into_boxed_slice(), + region_generations: region_generations.into_boxed_slice(), + rotation: Mutex::new(()), + health, + }) + } + + /// Freeze the quiescent index and Region metadata for a recovery image. + /// The caller must stop mutation admission and finish all accepted writes first. + pub fn freeze(self: Arc) -> io::Result { + self.health.require_healthy()?; + let regions = self; + if regions + .append_gates + .iter() + .any(|shard| shard.mutation.is_poisoned()) + || regions.rotation.is_poisoned() + { + regions.health.enter_miss_only(); + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "append shard gate is poisoned", + )); + } + let partitions = index_partition_metadata(regions.index.storage(), ®ions.health)?; + let metadata = regions + .manager + .lock()? + .freeze_metadata(partitions) + .map_err(region_metadata_io_error)?; + regions.health.require_healthy()?; + Ok(FrozenRegionStore { regions, metadata }) + } + pub const fn shard_count(&self) -> usize { self.append_gates.len() } @@ -548,7 +638,7 @@ impl RegionStore { self.health.require_healthy()?; gate.lock().map_err(|_| { self.health.enter_miss_only(); - io::Error::new(io::ErrorKind::InvalidData, "data shard gate is poisoned") + io::Error::new(io::ErrorKind::InvalidData, "append shard gate is poisoned") }) } @@ -608,7 +698,7 @@ impl RegionStore { buffer: BufferLease, hash_seed: u64, key: &[u8], - ) -> io::Result> { + ) -> io::Result> { let hash = hash_key(hash_seed, key); let Some(candidate) = self.begin_point_read(hash) else { return Ok(None); @@ -626,7 +716,7 @@ impl RegionStore { buffer: BufferLease, desc: ReadDesc, key: &[u8], - ) -> io::Result> { + ) -> io::Result> { let pending = self.submit_value_read(engine, slot, buffer, desc)?; let completion = pending.wait(engine); self.finish_value_read(completion, key) @@ -655,7 +745,7 @@ impl RegionStore { &self, completion: ReadCompletion, key: &[u8], - ) -> io::Result> { + ) -> io::Result> { let hash = completion.desc.hash; if let Err(error) = completion.result { if !is_read_pressure(error.kind()) { @@ -729,7 +819,7 @@ impl RegionStore { .enter_miss_only_with_error("record_read_completion_invalid", &error); return Err(error); }; - Ok(Some(RegionValueRead { + Ok(Some(RegionValue { buffer, buffer_len: completion.desc.read_len, value_range: value_start..value_end, @@ -797,7 +887,7 @@ impl RegionStore { self.health.enter_miss_only(); return Err(io::Error::new( io::ErrorKind::InvalidData, - "data shard gate is poisoned", + "append shard gate is poisoned", )); } }; @@ -1305,3 +1395,109 @@ fn enter_miss_only_for_index_error(health: &RegionHealthLatch, error: &IndexStor }; health.enter_miss_only_with_error(reason, error); } + +fn empty_partition_metadata( + ranges: &[IndexPartitionRange], +) -> io::Result> { + let mut stats = Vec::new(); + stats.try_reserve_exact(ranges.len()).map_err(|_| { + io::Error::new( + io::ErrorKind::OutOfMemory, + "cannot allocate index partition statistics", + ) + })?; + stats.resize(ranges.len(), IndexPhysicalStats::default()); + partition_metadata_from_stats(ranges, &stats) +} + +fn index_partition_metadata( + index: &PartitionedIndexStorage, + health: &RegionHealthLatch, +) -> io::Result> { + let stats = guarded_index_result(health, index.partition_stats())?; + partition_metadata_from_stats(index.partition_ranges(), &stats) +} + +fn partition_metadata_from_stats( + ranges: &[IndexPartitionRange], + stats: &[IndexPhysicalStats], +) -> io::Result> { + if ranges.len() != stats.len() { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "index partition ranges and statistics disagree", + )); + } + let mut partitions = Vec::new(); + partitions.try_reserve_exact(ranges.len()).map_err(|_| { + io::Error::new( + io::ErrorKind::OutOfMemory, + "cannot allocate index partition directory", + ) + })?; + for (range, stats) in ranges.iter().zip(stats) { + partitions.push(PartitionMetadataRecord { + partition_id: u32::try_from(range.partition_id).map_err(|_| { + io::Error::new(io::ErrorKind::InvalidInput, "partition id is too large") + })?, + first_index_page: u64::try_from(range.first_page).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidInput, + "partition page offset is too large", + ) + })?, + index_page_count: u64::try_from(range.page_count).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidInput, + "partition page count is too large", + ) + })?, + first_slot: u64::try_from(range.first_slot).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidInput, + "partition slot offset is too large", + ) + })?, + slot_count: u64::try_from(range.slot_count).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidInput, + "partition slot count is too large", + ) + })?, + physical_value_slots: stats.value, + physical_deleted_slots: stats.deleted, + }); + } + Ok(partitions.into_boxed_slice()) +} + +fn metadata_partition_stats(metadata: &RegionMetadata) -> io::Result> { + let mut stats = Vec::new(); + stats + .try_reserve_exact(metadata.partitions.len()) + .map_err(|_| { + io::Error::new( + io::ErrorKind::OutOfMemory, + "cannot allocate index partition statistics", + ) + })?; + for partition in &metadata.partitions { + stats.push(IndexPhysicalStats { + value: partition.physical_value_slots, + deleted: partition.physical_deleted_slots, + }); + } + Ok(stats.into_boxed_slice()) +} + +fn metadata_partition_stats_match(metadata: &RegionMetadata, stats: &[IndexPhysicalStats]) -> bool { + metadata.partitions.len() == stats.len() + && metadata + .partitions + .iter() + .zip(stats) + .all(|(metadata, actual)| { + metadata.physical_value_slots == actual.value + && metadata.physical_deleted_slots == actual.deleted + }) +} diff --git a/cache2/src/region/file_backend/mod.rs b/cache2/src/region/persistence/mod.rs similarity index 70% rename from cache2/src/region/file_backend/mod.rs rename to cache2/src/region/persistence/mod.rs index d30b74f..77289b9 100644 --- a/cache2/src/region/file_backend/mod.rs +++ b/cache2/src/region/persistence/mod.rs @@ -12,25 +12,16 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! File ownership, recovery, and lifecycle adapter for the Region regions. +//! File ownership, restart recovery, and durable Region image publication. use std::fmt; use std::fs::File; use std::io::Write; use std::io::{self}; -use std::ops::Deref; use std::path::Path; use std::path::PathBuf; -use std::sync::Arc; -use std::sync::Mutex; -use std::sync::atomic::AtomicU64; -use crate::config::CacheConfig; use crate::config::runtime::IoMode; -#[cfg(test)] -use crate::config::runtime::RuntimeOptions; -#[cfg(test)] -use crate::config::storage::cache_config; use crate::io::file::DataFileHandles; use crate::io::file::PositionedIo; use crate::io::file::StorageFile; @@ -43,20 +34,18 @@ use crate::io::file::write_all_at; use crate::io::fs::FileSystem; use crate::io::fs::OsFileSystem; use crate::io::fs::parent_directory; -use crate::region::AppendShardGate; +use crate::region::FrozenRegionStore; use crate::region::RegionHealthLatch; -use crate::region::RegionManagerLock; use crate::region::RegionStore; +use crate::region::empty_partition_metadata; use crate::region::guarded_index_result; -use crate::region::index::RegionIndex; use crate::region::index::packed::MAX_INDEX_PARTITIONS; use crate::region::index::storage::IndexImageBinding; -use crate::region::index::storage::IndexPartitionRange; -use crate::region::index::storage::IndexPhysicalStats; use crate::region::index::storage::PartitionedIndexStorage; use crate::region::index::storage::canonical_index_partition_ranges; use crate::region::index_storage_io_error; -use crate::region::manager::RegionManager; +use crate::region::metadata_partition_stats; +use crate::region::metadata_partition_stats_match; use crate::region::recovery::DataSuperblock; use crate::region::recovery::DataSuperblockProbe; use crate::region::recovery::PersistentId; @@ -74,7 +63,6 @@ use crate::region::recovery::StateRecord; use crate::region::recovery::StateSelectionError; use crate::region::recovery::clean_image_matches; use crate::region::recovery::latest_state; -use crate::region::recovery::metadata::PartitionMetadataRecord; use crate::region::recovery::metadata::REGION_METADATA_PAGE_SIZE; use crate::region::recovery::metadata::REGION_METADATA_PARTITIONS_PER_PAGE; use crate::region::recovery::metadata::REGION_METADATA_REGIONS_PER_PAGE; @@ -87,9 +75,8 @@ use crate::region::recovery::prepare_next_state; use crate::region::recovery::prepare_running_barrier; use crate::region::recovery::recovery_image_index_len; use crate::region::region_metadata_io_error; -use crate::region::runtime::RegionDataPlane; -/// Shared shard count for compact concrete-backend fixtures. +/// Shared shard count for compact concrete-file fixtures. #[cfg(test)] const REGION_SHARDS: u32 = 4; @@ -114,132 +101,19 @@ impl RegionPaths { } } } -pub struct FileRegionRuntime { - regions: Arc, - data_plane: Option, -} - -impl Deref for FileRegionRuntime { - type Target = RegionStore; - - fn deref(&self) -> &Self::Target { - &self.regions - } -} -pub struct FrozenFileRegionView { - regions: Arc, - metadata: RegionMetadata, -} - -pub struct CleanFileRegionImage { +pub struct RecoveryImage { file: File, header: RecoveryImageHeader, metadata: RegionMetadata, } -pub struct PreparedFileRegionClean { +pub struct PreparedClean { state: StatePageWrite, health: RegionHealthLatch, } -impl FileRegionRuntime { - /// Installs one complete authority. Recovery metadata is consumed here so - /// the live runtime cannot retain a stale second copy beside the manager. - fn install(index: PartitionedIndexStorage, metadata: RegionMetadata) -> io::Result { - let physical_stats = index.partition_stats().map_err(index_storage_io_error)?; - let slot_count = u64::try_from(index.slot_count()).map_err(|_| { - io::Error::new(io::ErrorKind::InvalidData, "index capacity is too large") - })?; - if metadata.root.index_slots != slot_count - || metadata.root.partition_count as usize != index.partition_count() - || !metadata_partition_stats_match(&metadata, &physical_stats) - { - return Err(io::Error::new( - io::ErrorKind::InvalidData, - "index and Region metadata do not describe one authority", - )); - } - let manager = RegionManager::from_metadata(metadata).map_err(region_metadata_io_error)?; - let mut append_gates = Vec::new(); - append_gates - .try_reserve_exact(manager.active_regions().len()) - .map_err(|_| { - io::Error::new( - io::ErrorKind::OutOfMemory, - "cannot allocate data shard gates", - ) - })?; - append_gates.resize_with(manager.active_regions().len(), AppendShardGate::default); - let mut region_generations = Vec::new(); - region_generations - .try_reserve_exact(manager.regions().len()) - .map_err(|_| { - io::Error::new( - io::ErrorKind::OutOfMemory, - "cannot allocate Region access state", - ) - })?; - for region in manager.regions() { - region_generations.push(AtomicU64::new(region.created_seqno)); - } - let health = RegionHealthLatch::healthy(); - let index = RegionIndex::from_storage(index).map_err(index_storage_io_error)?; - Ok(Self { - regions: Arc::new(RegionStore { - index, - manager: RegionManagerLock::new(manager, health.clone()), - append_gates: append_gates.into_boxed_slice(), - region_generations: region_generations.into_boxed_slice(), - rotation: Mutex::new(()), - health, - }), - data_plane: None, - }) - } - - fn attach_data_plane( - &mut self, - data: DataSuperblock, - handles: DataFileHandles, - config: CacheConfig, - ) -> io::Result<()> { - if self.data_plane.is_some() { - return Err(io::Error::new( - io::ErrorKind::AlreadyExists, - "data plane is already attached", - )); - } - self.data_plane = Some(RegionDataPlane::new( - Arc::clone(&self.regions), - data, - handles, - config, - )?); - Ok(()) - } - - fn shutdown_data_plane(&mut self) -> io::Result { - self.data_plane - .take() - .map(|plane| plane.shutdown()) - .unwrap_or(Ok(false)) - } - - pub fn data_plane(&self) -> io::Result<&RegionDataPlane> { - self.data_plane.as_ref().ok_or_else(|| { - io::Error::new( - io::ErrorKind::Unsupported, - "backend does not provide a native runtime data path", - ) - }) - } -} - -/// Concrete state/index lifecycle backed by one data file and two sidecars. -/// -/// It owns the append/read runtime, persists one complete index plus the -/// Region/FIFO physical view, and never scans records during open. -pub struct FileRegionBackend +/// Owns cache files and publishes the persistent state used by restart recovery. +pub struct RegionPersistence where F: FileSystem, { @@ -248,7 +122,6 @@ where /// files retain their on-disk identities but must match this geometry and /// storage-layout fingerprint. format_data: DataSuperblock, - config: CacheConfig, file_system: F, data_file: Option, state_file: Option, @@ -260,53 +133,11 @@ where retain_lock: bool, } -impl FileRegionBackend { - #[cfg(test)] - fn for_test(paths: RegionPaths, format_data: DataSuperblock, index_slots: usize) -> Self { - Self::for_test_with_options(paths, format_data, index_slots, RuntimeOptions::default()) - } - - #[cfg(test)] - pub fn for_test_with_options( - paths: RegionPaths, - format_data: DataSuperblock, - index_slots: usize, - runtime_options: RuntimeOptions, - ) -> Self { - let config = cache_config(format_data.geometry, index_slots, runtime_options); - Self::new(paths, format_data, config) - } - - pub fn new(paths: RegionPaths, format_data: DataSuperblock, config: CacheConfig) -> Self { - Self::new_with_file_system(paths, format_data, OsFileSystem, config) - } -} - -impl FileRegionBackend -where - F: FileSystem, -{ - #[cfg(test)] - fn for_test_with_file_system( - paths: RegionPaths, - format_data: DataSuperblock, - index_slots: usize, - file_system: F, - ) -> Self { - let config = cache_config(format_data.geometry, index_slots, RuntimeOptions::default()); - Self::new_with_file_system(paths, format_data, file_system, config) - } - - fn new_with_file_system( - paths: RegionPaths, - format_data: DataSuperblock, - file_system: F, - config: CacheConfig, - ) -> Self { +impl RegionPersistence { + pub fn new(paths: RegionPaths, format_data: DataSuperblock, file_system: F) -> Self { Self { paths, format_data, - config, file_system, data_file: None, state_file: None, @@ -325,7 +156,7 @@ where .ok_or_else(|| io::Error::new(io::ErrorKind::NotConnected, "state file is not open")) } - fn data_superblock(&self) -> io::Result { + pub fn data_superblock(&self) -> io::Result { self.data.ok_or_else(|| { io::Error::new( io::ErrorKind::InvalidData, @@ -380,17 +211,17 @@ where ); } - fn cold_recovery(&self, reason: &'static str) -> io::Result> { + fn cold_recovery(&self, reason: &'static str) -> io::Result> { self.log_cold_recovery(reason); Ok(None) } /// Acquire exclusive ownership of the data and state files before inspection. - pub fn acquire_exclusive(&mut self) -> io::Result<()> { + pub fn acquire_exclusive(&mut self, io_mode: IoMode) -> io::Result<()> { if self.locked { return Err(io::Error::new( io::ErrorKind::AlreadyExists, - "CacheSession backend is already locked", + "cache files are already locked", )); } if self.paths.data == self.paths.state @@ -399,7 +230,7 @@ where { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CacheSession data/state/image paths must be distinct", + "cache data/state/image paths must be distinct", )); } if parent_directory(&self.paths.data) != parent_directory(&self.paths.state) @@ -407,7 +238,7 @@ where { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CacheSession data/state/image files must share one directory", + "cache data/state/image files must share one directory", )); } let temporary = recovery_temporary_path(&self.paths.image); @@ -417,12 +248,10 @@ where { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CacheSession recovery temporary path collides with a cache file", + "cache recovery temporary path collides with a cache file", )); } - let data = self - .file_system - .open(&self.paths.data, true, self.config.runtime().io_mode)?; + let data = self.file_system.open(&self.paths.data, true, io_mode)?; data.try_lock_exclusive()?; let state = match self .file_system @@ -445,7 +274,7 @@ where let _ = data.unlock(); return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CacheSession data and state paths resolve to the same file", + "cache data and state paths resolve to the same file", )); } if let Err(error) = state.try_lock_exclusive() { @@ -460,10 +289,7 @@ where /// Return an eligible clean image, or `None` to select a cold start. /// This must not allocate or scan the full index or Region data extents. - pub fn inspect_recovery( - &mut self, - index_slots: usize, - ) -> io::Result> { + pub fn inspect_recovery(&mut self, index_slots: usize) -> io::Result> { self.file_system .remove_file(&recovery_temporary_path(&self.paths.image))?; let format_data = self.format_data; @@ -527,7 +353,7 @@ where if image.is_same_file(data_file)? || image.is_same_file(state_file)? { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CacheSession image aliases the data or state file", + "cache image aliases the data or state file", )); } @@ -617,15 +443,19 @@ where } let file = image.try_clone_mapping_file()?; self.cold_reset_needed = false; - Ok(Some(CleanFileRegionImage { + Ok(Some(RecoveryImage { file, header, metadata, })) } - /// Construct a provisional empty runtime without starting workers. - pub fn anonymous_runtime(&mut self, index_slots: usize) -> io::Result { + /// Discard stale recovery artifacts and construct an empty Region store. + pub fn cold_regions( + &mut self, + index_slots: usize, + append_shards: u32, + ) -> io::Result { self.file_system.remove_file(&self.paths.image)?; self.file_system .remove_file(&recovery_temporary_path(&self.paths.image))?; @@ -641,8 +471,7 @@ where self.current_state = None; self.cold_reset_needed = false; } - let metadata = - empty_region_metadata(data, index_slots, self.config.runtime().append_shards)?; + let metadata = empty_region_metadata(data, index_slots, append_shards)?; let index = match PartitionedIndexStorage::anonymous(index_slots) { Ok(index) => index, Err(error) => { @@ -663,16 +492,16 @@ where return Err(index_storage_io_error(error)); } }; - let runtime = FileRegionRuntime::install(index, metadata)?; - Ok(runtime) + RegionStore::from_recovery(index, metadata) } /// `Ok(None)` rejects the complete image and selects a cold start. - pub fn map_clean_runtime( + pub fn recover_regions( &mut self, - mut clean: CleanFileRegionImage, + mut image: RecoveryImage, index_slots: usize, - ) -> io::Result> { + append_shards: u32, + ) -> io::Result> { let data = self.data_superblock()?; let expected_slots = u64::try_from(index_slots).map_err(|_| { io::Error::new( @@ -683,45 +512,44 @@ where let expected_index_len = recovery_image_index_len(expected_slots).ok_or_else(|| { io::Error::new(io::ErrorKind::InvalidInput, "invalid index image length") })?; - let actual_file_len = clean.file.metadata()?.len(); + let actual_file_len = image.file.metadata()?.len(); let eligible = self.current_state.is_some_and(|selected| { clean_image_matches( selected.record, data, - clean.header, + image.header, actual_file_len, expected_slots, expected_index_len, ) - }) && clean.metadata.matches_image(data, clean.header) - && clean.metadata.validate().is_ok(); + }) && image.metadata.matches_image(data, image.header) + && image.metadata.validate().is_ok(); if !eligible { self.cold_reset_needed = true; self.log_cold_recovery("image_became_ineligible"); return Ok(None); } - let previous_append_shards = clean.metadata.root.shard_count; - let shard_count = self.config.runtime().append_shards; - if previous_append_shards != shard_count { - let added_shards = shard_count.saturating_sub(previous_append_shards); - if added_shards > clean.metadata.root.free_region_count { + let previous_append_shards = image.metadata.root.shard_count; + if previous_append_shards != append_shards { + let added_shards = append_shards.saturating_sub(previous_append_shards); + if added_shards > image.metadata.root.free_region_count { self.cold_reset_needed = true; self.log_cold_recovery("append_shards_rebind_insufficient_free_regions"); return Ok(None); } - if clean.metadata.rebind_append_shards(shard_count).is_err() { + if image.metadata.rebind_append_shards(append_shards).is_err() { self.cold_reset_needed = true; self.log_cold_recovery("append_shards_rebind_invalid"); return Ok(None); } - self.log_append_shards_rebind_planned(previous_append_shards, shard_count); + self.log_append_shards_rebind_planned(previous_append_shards, append_shards); } - let partition_stats = metadata_partition_stats(&clean.metadata)?; - let binding = index_image_binding(clean.header); - let index_mapping_bytes = clean.header.index_offset.saturating_add(expected_index_len); + let partition_stats = metadata_partition_stats(&image.metadata)?; + let binding = index_image_binding(image.header); + let index_mapping_bytes = image.header.index_offset.saturating_add(expected_index_len); let index = match PartitionedIndexStorage::map_private( - &clean.file, - clean.header.index_offset, + &image.file, + image.header.index_offset, index_slots, binding, &partition_stats, @@ -738,8 +566,8 @@ where return Ok(None); } }; - let runtime = FileRegionRuntime::install(index, clean.metadata)?; - Ok(Some(runtime)) + let regions = RegionStore::from_recovery(index, image.metadata)?; + Ok(Some(regions)) } /// Replace both state slots with durable `RUNNING` generations so a torn @@ -762,93 +590,27 @@ where Ok(()) } - /// Start workers only after `RUNNING`; an error must tear them down. - pub fn start_runtime( - &mut self, - mut runtime: FileRegionRuntime, - ) -> io::Result { - let data = self.data_superblock()?; - let data_file = self - .data_file + pub fn clone_data_handles(&self) -> io::Result { + self.data_file .as_ref() - .ok_or_else(|| io::Error::new(io::ErrorKind::NotConnected, "data file is not open"))?; - let handles = data_file.try_clone_data_handles()?; - runtime.attach_data_plane(data, handles, self.config.clone())?; - Ok(runtime) + .ok_or_else(|| io::Error::new(io::ErrorKind::NotConnected, "data file is not open"))? + .try_clone_data_handles() } - /// Quiesce all mutation sources without constructing recovery metadata. - pub fn stop_fast(&mut self, mut runtime: FileRegionRuntime) -> io::Result<()> { - match runtime.shutdown_data_plane() { - Ok(false) => Ok(()), - Ok(true) => { - self.retain_lock = true; - Err(io::Error::other( - "I/O engine could not fence an issued write; lock retained", - )) - } - Err(error) => { - runtime.health.enter_miss_only(); - Err(error) - } - } - } - - /// Quiesce the runtime and return its immutable recovery authority. - pub fn freeze_warm( - &mut self, - mut runtime: FileRegionRuntime, - ) -> io::Result { - match runtime.shutdown_data_plane() { - Ok(false) => {} - Ok(true) => { - self.retain_lock = true; - runtime.health.enter_miss_only(); - return Err(io::Error::other( - "I/O engine could not fence an issued write; CLEAN rejected", - )); - } - Err(error) => { - runtime.health.enter_miss_only(); - return Err(error); - } - } - runtime.health.require_healthy()?; - let regions = runtime.regions; - if regions - .append_gates - .iter() - .any(|shard| shard.mutation.is_poisoned()) - || regions.rotation.is_poisoned() - { - regions.health.enter_miss_only(); - return Err(io::Error::new( - io::ErrorKind::InvalidData, - "data shard gate is poisoned", - )); - } - let partitions = index_partition_metadata(regions.index.storage(), ®ions.health)?; - let metadata = regions - .manager - .lock()? - .freeze_metadata(partitions) - .map_err(region_metadata_io_error)?; - regions.health.require_healthy()?; - Ok(FrozenFileRegionView { regions, metadata }) + /// Keep ownership with issued writes whose completion cannot be fenced. + pub fn retain_data_lock(&mut self) { + self.retain_lock = true; } /// Make completed data and one complete image durable. - pub fn persist_frozen( - &mut self, - view: &FrozenFileRegionView, - ) -> io::Result { - let health = &view.regions.health; + pub fn persist_frozen(&mut self, frozen: &FrozenRegionStore) -> io::Result { + let health = &frozen.regions.health; health.require_healthy()?; - let source_metadata = &view.metadata; + let source_metadata = &frozen.metadata; source_metadata .validate() .map_err(region_metadata_io_error)?; - let storage = view.regions.index.storage(); + let storage = frozen.regions.index.storage(); let physical_stats = guarded_index_result(health, storage.physical_stats())?; let partition_stats = guarded_index_result(health, storage.partition_stats())?; if source_metadata.root.index_slots @@ -961,19 +723,19 @@ where } persisted?; self.prepared_clean = Some((clean_state.slot, clean_state.record)); - Ok(PreparedFileRegionClean { + Ok(PreparedClean { state: clean_state, health: health.clone(), }) } /// Publish `CLEAN` durably using the token returned after persistence. - pub fn publish_clean(&mut self, prepared: PreparedFileRegionClean) -> io::Result<()> { + pub fn publish_clean(&mut self, prepared: PreparedClean) -> io::Result<()> { prepared.health.require_healthy()?; if self.prepared_clean.take() != Some((prepared.state.slot, prepared.state.record)) { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CLEAN token does not belong to this backend session", + "CLEAN token does not belong to this persistence session", )); } let data = self.data_superblock()?; @@ -1160,121 +922,15 @@ fn maximum_region_metadata_len(region_count: u32) -> io::Result { .ok_or_else(|| io::Error::other("Region metadata length overflow")) } -fn empty_partition_metadata( - ranges: &[IndexPartitionRange], -) -> io::Result> { - let mut stats = Vec::new(); - stats.try_reserve_exact(ranges.len()).map_err(|_| { - io::Error::new( - io::ErrorKind::OutOfMemory, - "cannot allocate index partition statistics", - ) - })?; - stats.resize(ranges.len(), IndexPhysicalStats::default()); - partition_metadata_from_stats(ranges, &stats) -} - -fn index_partition_metadata( - index: &PartitionedIndexStorage, - health: &RegionHealthLatch, -) -> io::Result> { - let stats = guarded_index_result(health, index.partition_stats())?; - partition_metadata_from_stats(index.partition_ranges(), &stats) -} - -fn partition_metadata_from_stats( - ranges: &[IndexPartitionRange], - stats: &[IndexPhysicalStats], -) -> io::Result> { - if ranges.len() != stats.len() { - return Err(io::Error::new( - io::ErrorKind::InvalidData, - "index partition ranges and statistics disagree", - )); - } - let mut partitions = Vec::new(); - partitions.try_reserve_exact(ranges.len()).map_err(|_| { - io::Error::new( - io::ErrorKind::OutOfMemory, - "cannot allocate index partition directory", - ) - })?; - for (range, stats) in ranges.iter().zip(stats) { - partitions.push(PartitionMetadataRecord { - partition_id: u32::try_from(range.partition_id).map_err(|_| { - io::Error::new(io::ErrorKind::InvalidInput, "partition id is too large") - })?, - first_index_page: u64::try_from(range.first_page).map_err(|_| { - io::Error::new( - io::ErrorKind::InvalidInput, - "partition page offset is too large", - ) - })?, - index_page_count: u64::try_from(range.page_count).map_err(|_| { - io::Error::new( - io::ErrorKind::InvalidInput, - "partition page count is too large", - ) - })?, - first_slot: u64::try_from(range.first_slot).map_err(|_| { - io::Error::new( - io::ErrorKind::InvalidInput, - "partition slot offset is too large", - ) - })?, - slot_count: u64::try_from(range.slot_count).map_err(|_| { - io::Error::new( - io::ErrorKind::InvalidInput, - "partition slot count is too large", - ) - })?, - physical_value_slots: stats.value, - physical_deleted_slots: stats.deleted, - }); - } - Ok(partitions.into_boxed_slice()) -} - -fn metadata_partition_stats(metadata: &RegionMetadata) -> io::Result> { - let mut stats = Vec::new(); - stats - .try_reserve_exact(metadata.partitions.len()) - .map_err(|_| { - io::Error::new( - io::ErrorKind::OutOfMemory, - "cannot allocate index partition statistics", - ) - })?; - for partition in &metadata.partitions { - stats.push(IndexPhysicalStats { - value: partition.physical_value_slots, - deleted: partition.physical_deleted_slots, - }); - } - Ok(stats.into_boxed_slice()) -} - -fn metadata_partition_stats_match(metadata: &RegionMetadata, stats: &[IndexPhysicalStats]) -> bool { - metadata.partitions.len() == stats.len() - && metadata - .partitions - .iter() - .zip(stats) - .all(|(metadata, actual)| { - metadata.physical_value_slots == actual.value - && metadata.physical_deleted_slots == actual.deleted - }) -} - fn empty_region_metadata( data: DataSuperblock, index_slots: usize, - append_gates: u32, + append_shards: u32, ) -> io::Result { - if append_gates == 0 || data.geometry.region_count <= append_gates { + if append_shards == 0 || data.geometry.region_count <= append_shards { return Err(io::Error::new( io::ErrorKind::InvalidInput, - "CacheSession requires one Active Region per shard plus one spare", + "cache requires one Active Region per shard plus one spare", )); } let partition_ranges = @@ -1292,7 +948,7 @@ fn empty_region_metadata( .map_err(|_| io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate Region table"))?; let mut free_ordinal = 0_u32; for region_id in 0..data.geometry.region_count { - let active = region_id < append_gates; + let active = region_id < append_shards; let queue_ordinal = if active { region_id } else { @@ -1330,10 +986,10 @@ fn empty_region_metadata( partition_count: u32::try_from(partition_ranges.len()).map_err(|_| { io::Error::new(io::ErrorKind::InvalidInput, "too many index partitions") })?, - shard_count: append_gates, - max_seqno: u64::from(append_gates), - free_region_count: data.geometry.region_count - append_gates, - active_region_count: append_gates, + shard_count: append_shards, + max_seqno: u64::from(append_shards), + free_region_count: data.geometry.region_count - append_shards, + active_region_count: append_shards, sealed_region_count: 0, }, regions: regions.into_boxed_slice(), @@ -1394,15 +1050,15 @@ const fn anonymous_index_backing_name() -> &'static str { } struct PositionedIoWriter<'a, B: PositionedIo + ?Sized> { - backend: &'a B, + file: &'a B, point: WritePoint, offset: u64, } impl<'a, B: PositionedIo + ?Sized> PositionedIoWriter<'a, B> { - const fn new(backend: &'a B, point: WritePoint, offset: u64) -> Self { + const fn new(file: &'a B, point: WritePoint, offset: u64) -> Self { Self { - backend, + file, point, offset, } @@ -1415,7 +1071,7 @@ impl<'a, B: PositionedIo + ?Sized> PositionedIoWriter<'a, B> { impl Write for PositionedIoWriter<'_, B> { fn write(&mut self, buffer: &[u8]) -> io::Result { - let written = self.backend.write_at(self.point, buffer, self.offset)?; + let written = self.file.write_at(self.point, buffer, self.offset)?; self.offset = self .offset .checked_add(written as u64) diff --git a/cache2/src/region/file_backend/tests.rs b/cache2/src/region/persistence/tests.rs similarity index 79% rename from cache2/src/region/file_backend/tests.rs rename to cache2/src/region/persistence/tests.rs index b49fa61..57dc0db 100644 --- a/cache2/src/region/file_backend/tests.rs +++ b/cache2/src/region/persistence/tests.rs @@ -35,10 +35,12 @@ use std::time::Instant; use super::*; use crate::IoEngineOptions; use crate::StatsOptions; +use crate::cache::runtime::CacheRead; use crate::cache::session::CacheSession; use crate::config::runtime::MAX_WRITE_FLUSH_THRESHOLD_BYTES; use crate::config::runtime::PosixIoOptions; use crate::config::runtime::ReadAdmission; +use crate::config::runtime::RuntimeOptions; use crate::io::engine::IoEngine; use crate::io::file::MAX_INTERRUPTED_RETRIES; use crate::io::file::testing::FaultAction; @@ -54,6 +56,7 @@ use crate::region::index::packed::PackedLocation; use crate::region::index::storage::IndexSlot; use crate::region::index::storage::IndexSlotState; use crate::region::index::storage::page_format::INDEX_IMAGE_SLOTS_PER_PAGE; +use crate::region::manager::RegionManager; use crate::region::reader::ReadCandidate; use crate::region::reader::ReadCompletion; use crate::region::reader::ReadDesc; @@ -64,7 +67,6 @@ use crate::region::record::codec::required_record_bytes; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; use crate::region::recovery::PersistentId; -use crate::region::runtime::HybridValueRead; use crate::region::staging::AppendStaging; use crate::region::staging::StagedRecord; use crate::snapshot::CacheSnapshot; @@ -73,34 +75,31 @@ use crate::snapshot::StartupMode; impl CacheSession { fn put_value(&self, key: &[u8], value: &[u8]) -> io::Result { - self.runtime()?.data_plane()?.put(key, value) + self.runtime()?.put(key, value) } - fn get_value(&self, key: &[u8]) -> io::Result> { - self.runtime()?.data_plane()?.get(key) + fn get_value(&self, key: &[u8]) -> io::Result> { + self.runtime()?.get(key) } async fn get_value_async( &self, key: &[u8], tokio_handle: &tokio::runtime::Handle, - ) -> io::Result> { - self.runtime()? - .data_plane()? - .get_async(key, tokio_handle, None) - .await + ) -> io::Result> { + self.runtime()?.get_async(key, tokio_handle, None).await } pub fn drain(&self) -> io::Result<()> { - self.runtime()?.data_plane()?.drain() + self.runtime()?.drain() } pub fn snapshot(&self) -> io::Result { - self.runtime()?.data_plane()?.snapshot() + self.runtime()?.snapshot() } pub fn detailed_snapshot(&self) -> io::Result { - self.runtime()?.data_plane()?.detailed_snapshot() + self.runtime()?.detailed_snapshot() } } @@ -327,18 +326,14 @@ fn external_process_kill_recovery_contract() { ] { let directory = TestDirectory::new(); let data = data_path_superblock(); - let mut initial = CacheSession::open( - 4096, - FileRegionBackend::for_test(directory.paths.clone(), data, 4096), - ) - .unwrap(); + let mut initial = CacheSession::for_test(directory.paths.clone(), data, 4096).unwrap(); eventually_admitted(|| initial.put_value(b"survivor", b"old")); initial.drain().unwrap(); initial.close_warm().unwrap(); let status = Command::new(env::current_exe().unwrap()) .arg("--exact") - .arg("region::file_backend::tests::external_process_kill_recovery_contract") + .arg("region::persistence::tests::external_process_kill_recovery_contract") .arg("--ignored") .arg("--nocapture") .env(CHILD_CASE, case) @@ -353,11 +348,7 @@ fn external_process_kill_recovery_contract() { "crash case {case} did not SIGKILL" ); - let mut reopened = CacheSession::open( - 4096, - FileRegionBackend::for_test(directory.paths.clone(), data, 4096), - ) - .unwrap(); + let mut reopened = CacheSession::for_test(directory.paths.clone(), data, 4096).unwrap(); if expect_clean { assert_eq!(reopened.startup(), StartupMode::Warm, "{case}"); assert_eq!( @@ -378,13 +369,11 @@ fn run_crash_child(case: &str, paths: RegionPaths) -> ! { let data = data_path_superblock(); match case { "open" => { - let _store = - CacheSession::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); + let _store = CacheSession::for_test(paths, data, 4096).unwrap(); kill_process(); } "write" | "drain" => { - let session = - CacheSession::open(4096, FileRegionBackend::for_test(paths, data, 4096)).unwrap(); + let session = CacheSession::for_test(paths, data, 4096).unwrap(); eventually_admitted(|| session.put_value(b"replacement", b"new")); if case == "drain" { session.drain().unwrap(); @@ -393,11 +382,8 @@ fn run_crash_child(case: &str, paths: RegionPaths) -> ! { } "warm-data" | "warm-image" | "clean-state" => { let (file_system, faults, _) = FaultFileSystem::new(); - let mut session = CacheSession::open( - 4096, - FileRegionBackend::for_test_with_file_system(paths, data, 4096, file_system), - ) - .unwrap(); + let mut session = + CacheSession::for_test_with_file_system(paths, data, 4096, file_system).unwrap(); let point = match case { "warm-data" => SyncPoint::WarmData, "warm-image" => SyncPoint::RecoveryImage, @@ -459,16 +445,9 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { }, ..RuntimeOptions::default() }; - let mut session = CacheSession::open( - 4096, - FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - 4096, - runtime_options, - ), - ) - .unwrap(); + let mut session = + CacheSession::for_test_with_options(directory.paths.clone(), data, 4096, runtime_options) + .unwrap(); let tokio_runtime = tokio::runtime::Builder::new_multi_thread() .worker_threads(2) .enable_time() @@ -476,8 +455,10 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { .unwrap(); eventually_admitted(|| session.put_value(b"queued-read", b"local-value")); session.drain().unwrap(); - let plane = session.data_plane_handle().unwrap(); - let mut slots: Vec<_> = (0..2).map(|_| plane.reserve_read_slot_for_test()).collect(); + let runtime = session.runtime().unwrap().clone(); + let mut slots: Vec<_> = (0..2) + .map(|_| runtime.reserve_read_slot_for_test()) + .collect(); tokio_runtime.block_on(async { let mut cancelled = Box::pin(session.get_value_async(b"queued-read", tokio_runtime.handle())); @@ -501,7 +482,9 @@ fn configured_read_wait_is_bounded_and_cancel_safe() { drop(slots); assert_eq!(value.value(), b"local-value"); drop(value); - let blocked: Vec<_> = (0..2).map(|_| plane.reserve_read_slot_for_test()).collect(); + let blocked: Vec<_> = (0..2) + .map(|_| runtime.reserve_read_slot_for_test()) + .collect(); tokio_runtime.block_on(async { let mut waiting = Box::pin(session.get_value_async(b"queued-read", tokio_runtime.handle())); assert_pending( @@ -547,14 +530,11 @@ fn queued_l2_read_does_not_pin_warm_close() { }, ..RuntimeOptions::default() }; - let mut session = CacheSession::open( + let mut session = CacheSession::for_test_with_options( + directory.paths.clone(), + data, 4096, - FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - 4096, - runtime_options.clone(), - ), + runtime_options.clone(), ) .unwrap(); let tokio_runtime = tokio::runtime::Builder::new_current_thread() @@ -563,11 +543,12 @@ fn queued_l2_read_does_not_pin_warm_close() { .unwrap(); eventually_admitted(|| session.put_value(b"queued-close", b"value")); session.drain().unwrap(); - let plane = session.data_plane_handle().unwrap(); - let slot = plane.reserve_read_slot_for_test(); + let runtime = session.runtime().unwrap().clone(); + let slot = runtime.reserve_read_slot_for_test(); tokio_runtime.block_on(async { - let mut waiting = Box::pin(plane.get_async(b"queued-close", tokio_runtime.handle(), None)); + let mut waiting = + Box::pin(runtime.get_async(b"queued-close", tokio_runtime.handle(), None)); assert_pending(waiting.as_mut(), "saturated read must enter the wait queue").await; session.close_warm().unwrap(); @@ -579,22 +560,15 @@ fn queued_l2_read_does_not_pin_warm_close() { assert_eq!(error.kind(), io::ErrorKind::BrokenPipe); }); - let mut reopened = CacheSession::open( - 4096, - FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - 4096, - runtime_options, - ), - ) - .unwrap(); + let mut reopened = + CacheSession::for_test_with_options(directory.paths.clone(), data, 4096, runtime_options) + .unwrap(); assert_eq!(reopened.startup(), StartupMode::Warm); reopened.close_fast().unwrap(); } #[test] -fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { +fn cache_runtime_reads_mixed_chunks_rotates_and_warm_recovers() { let directory = TestDirectory::new(); let data = production_data_superblock(512 * 1024); let runtime_options = RuntimeOptions { @@ -605,16 +579,9 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { }, ..RuntimeOptions::default() }; - let mut session = CacheSession::open( - 4096, - FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - 4096, - runtime_options, - ), - ) - .unwrap(); + let mut session = + CacheSession::for_test_with_options(directory.paths.clone(), data, 4096, runtime_options) + .unwrap(); let mixed = [16 * 1024, 64 * 1024, 128 * 1024, 256 * 1024]; let mut expected = Vec::new(); @@ -684,11 +651,7 @@ fn production_data_plane_reads_mixed_chunks_rotates_and_warm_recovers() { session.close_warm().unwrap(); assert_eq!(retained_hits[0].value(), rotation_value); drop(retained_hits); - let mut recovered = CacheSession::open( - 4096, - FileRegionBackend::for_test(directory.paths.clone(), data, 4096), - ) - .unwrap(); + let mut recovered = CacheSession::for_test(directory.paths.clone(), data, 4096).unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); assert_eq!( recovered @@ -717,38 +680,28 @@ fn poisoned_runtime_gates_stop_workers_and_reject_warm_close() { write_flush_threshold_bytes: 128 * 1024, ..RuntimeOptions::default() }; - let mut session = CacheSession::open( + let mut session = CacheSession::for_test_with_options( + directory.paths.clone(), + data, 4096, - FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - 4096, - runtime_options.clone(), - ), + runtime_options.clone(), ) .unwrap(); - let runtime = session.runtime().unwrap(); + let runtime = session.regions().unwrap(); match case { - "shard" => runtime.data_plane().unwrap().poison_shard_for_test(0), - "index" => runtime - .regions - .index - .storage() - .poison_hash_partition_for_test(0), + "shard" => session.runtime().unwrap().poison_append_worker_for_test(0), + "index" => runtime.index.storage().poison_hash_partition_for_test(0), _ => unreachable!(), } let error = session.close_warm().unwrap_err(); assert_eq!(error.kind(), io::ErrorKind::InvalidData, "{case}"); - let mut reopened = CacheSession::open( + let mut reopened = CacheSession::for_test_with_options( + directory.paths.clone(), + data, 4096, - FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - 4096, - runtime_options, - ), + runtime_options, ) .unwrap(); assert_eq!(reopened.startup(), StartupMode::Cold, "{case}"); @@ -810,9 +763,9 @@ fn four_tib_region_metadata_round_trips_into_runtime_authority() { assert_eq!(manager.next_seqno(), u64::from(REGION_SHARDS) + 1); } -fn foreground_stage_fixture() -> (DataSuperblock, FileRegionRuntime, AppendStaging) { +fn foreground_stage_fixture() -> (DataSuperblock, Arc, AppendStaging) { let data = data_path_superblock(); - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(64).unwrap(), empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) @@ -825,19 +778,19 @@ fn foreground_stage_fixture() -> (DataSuperblock, FileRegionRuntime, AppendStagi &managed_memory, ) .unwrap(); - (data, runtime, staging) + (data, Arc::new(regions), staging) } #[test] fn foreground_stage_bypasses_busy_manager_without_consuming_a_sequence() { - let (data, runtime, staging) = foreground_stage_fixture(); - let manager = runtime.manager.state.lock().unwrap(); + let (data, regions, staging) = foreground_stage_fixture(); + let manager = regions.manager.state.lock().unwrap(); let next_seqno = manager.next_seqno(); let hash = hash_key(data.hash_seed, b"key"); let record_bytes = required_record_bytes(b"key".len(), b"value".len()).unwrap(); assert_eq!( - runtime + regions .try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value") .unwrap(), RegionStageValue::NeedsProgress @@ -847,15 +800,16 @@ fn foreground_stage_bypasses_busy_manager_without_consuming_a_sequence() { #[test] fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { - let (data, runtime, staging) = foreground_stage_fixture(); - let next_seqno = runtime.manager.state.lock().unwrap().next_seqno(); - let mutation = runtime.regions.append_gates[0].mutation.lock().unwrap(); + let (data, regions, staging) = foreground_stage_fixture(); + let next_seqno = regions.manager.state.lock().unwrap().next_seqno(); + let mutation = regions.append_gates[0].mutation.lock().unwrap(); let hash = hash_key(data.hash_seed, b"key"); let record_bytes = required_record_bytes(b"key".len(), b"value".len()).unwrap(); let (sender, receiver) = mpsc::sync_channel(1); - let regions = Arc::clone(&runtime.regions); + let writer_regions = Arc::clone(®ions); let writer = std::thread::spawn(move || { - let result = regions.try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value"); + let result = + writer_regions.try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value"); sender.send((result, staging)).unwrap(); }); @@ -865,13 +819,12 @@ fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { let (result, staging) = early.expect("a busy shard must reject without waiting for its lock"); assert!(matches!(result.unwrap(), RegionStageValue::NeedsProgress)); assert_eq!( - runtime.manager.state.lock().unwrap().next_seqno(), + regions.manager.state.lock().unwrap().next_seqno(), next_seqno ); assert!(staging.shard_fill_snapshot(0).unwrap().is_none()); - let staged = runtime - .regions + let staged = regions .try_stage_value(&staging, 0, hash, record_bytes, b"key", b"value") .unwrap(); assert!(matches!( @@ -879,7 +832,7 @@ fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { RegionStageValue::Staged { seqno, .. } if seqno == next_seqno )); assert_eq!( - runtime.manager.state.lock().unwrap().next_seqno(), + regions.manager.state.lock().unwrap().next_seqno(), next_seqno + 1 ); } @@ -887,12 +840,12 @@ fn foreground_stage_rejects_busy_shard_without_reserving_then_stages_once() { #[test] fn completed_record_publication_does_not_enter_region_manager() { let data = data_path_superblock(); - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(64).unwrap(), empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) .unwrap(); - let regions = Arc::clone(&runtime.regions); + let regions = Arc::new(regions); let manager = regions.manager.state.lock().unwrap(); let record = StagedRecord::new( 7, @@ -902,10 +855,10 @@ fn completed_record_publication_does_not_enter_region_manager() { 1, ); let (sender, receiver) = mpsc::sync_channel(1); - let publisher_core = Arc::clone(®ions); + let publisher_regions = Arc::clone(®ions); let publisher = std::thread::spawn(move || { sender - .send(publisher_core.publish_completed_records(&[record])) + .send(publisher_regions.publish_completed_records(&[record])) .unwrap(); }); @@ -913,14 +866,14 @@ fn completed_record_publication_does_not_enter_region_manager() { drop(manager); publisher.join().unwrap(); published.unwrap().unwrap(); - assert_eq!(runtime.lookup_snapshot(7).unwrap(), Some(record.entry())); + assert_eq!(regions.lookup_snapshot(7).unwrap(), Some(record.entry())); } #[test] fn completed_owned_span_publishes_index_without_a_steady_state_sync() { let data = data_path_superblock(); let index_slots = INDEX_IMAGE_SLOTS_PER_PAGE * 4; - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(index_slots).unwrap(), empty_region_metadata(data, index_slots, REGION_SHARDS).unwrap(), ) @@ -934,9 +887,9 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, faults) = FaultFile::open(&directory.paths.data).unwrap(); - backend.set_len(data.geometry.data_file_len).unwrap(); - let engine = IoEngine::for_test(Arc::new(backend), 2).unwrap(); + let (file, faults) = FaultFile::open(&directory.paths.data).unwrap(); + file.set_len(data.geometry.data_file_len).unwrap(); + let engine = IoEngine::for_test(Arc::new(file), 2).unwrap(); let value = vec![0x5a; 16 * 1024]; let mut first = None; let mut last = None; @@ -945,7 +898,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { let key = format!("file/chunk/{staged_records:04}"); let hash = hash_key(data.hash_seed, key.as_bytes()); let record_bytes = required_record_bytes(key.len(), value.len()).unwrap(); - match runtime + match regions .try_stage_value(&staging, 0, hash, record_bytes, key.as_bytes(), &value) .unwrap() { @@ -964,9 +917,9 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { first.expect("4 MiB span must contain target-size records"); let (last_key, last_hash, last_seqno) = last.expect("4 MiB span must retain its final record"); assert!(staged_records > 240); - assert_eq!(runtime.lookup_snapshot(first_hash).unwrap(), None); + assert_eq!(regions.lookup_snapshot(first_hash).unwrap(), None); - let published = runtime + let published = regions .flush_staging_shard( &staging, &engine, @@ -979,7 +932,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { (published.end_offset - published.start_offset) % RECOVERY_PAGE_SIZE as u64, 0 ); - let Some(entry) = runtime.lookup_snapshot(first_hash).unwrap() else { + let Some(entry) = regions.lookup_snapshot(first_hash).unwrap() else { panic!("completed first record must be published"); }; let first_exact = PackedLocation::new( @@ -990,7 +943,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { .unwrap(); assert!(entry.location.index_equivalent(first_exact)); assert_ne!(entry.location.record_len() % RECOVERY_PAGE_SIZE as u32, 0); - let Some(last_entry) = runtime.lookup_snapshot(last_hash).unwrap() else { + let Some(last_entry) = regions.lookup_snapshot(last_hash).unwrap() else { panic!("completed final record must be published"); }; assert!( @@ -1006,19 +959,19 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { ) .unwrap(); assert!(last_entry.location.index_equivalent(last_exact)); - let read = runtime + let read = regions .begin_point_read(first_hash) .expect("completed entry must yield a Region read candidate"); assert_eq!(read.entry, entry); - let last_read = runtime + let last_read = regions .begin_point_read(last_hash) .expect("completed final entry must yield a Region read candidate"); let read_buffer_bytes = describe_read(data.geometry, last_hash, last_read, true) .unwrap() .read_len; let memory_before_read = managed_memory.snapshot().current_bytes; - let hit = runtime + let hit = regions .read_value( &engine, data.geometry, @@ -1045,7 +998,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { #[test] fn read_availability_errors_do_not_latch_miss_only() { let data = data_path_superblock(); - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(64).unwrap(), empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) @@ -1073,18 +1026,18 @@ fn read_availability_errors_do_not_latch_miss_only() { result: Err(io::Error::new(kind, "injected read availability error")), buffer: None, }; - let Err(error) = runtime.finish_value_read(completion, b"key") else { + let Err(error) = regions.finish_value_read(completion, b"key") else { panic!("injected read availability error unexpectedly succeeded"); }; assert_eq!(error.kind(), kind); - assert!(runtime.is_healthy(), "{kind:?}"); + assert!(regions.is_healthy(), "{kind:?}"); } } #[test] fn same_hash_candidate_requires_full_key() { let data = data_path_superblock(); - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(64).unwrap(), empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) @@ -1098,15 +1051,15 @@ fn same_hash_candidate_requires_full_key() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, _) = FaultFile::open(&directory.paths.data).unwrap(); - backend.set_len(data.geometry.data_file_len).unwrap(); - let engine = IoEngine::for_test(Arc::new(backend), 1).unwrap(); + let (file, _) = FaultFile::open(&directory.paths.data).unwrap(); + file.set_len(data.geometry.data_file_len).unwrap(); + let engine = IoEngine::for_test(Arc::new(file), 1).unwrap(); let owner_key = b"collision-owner"; let foreign_key = b"collision-foreign"; let value = b"owner-value-must-not-leak"; let owner_hash = hash_key(data.hash_seed, owner_key); let owner_record_bytes = required_record_bytes(owner_key.len(), value.len()).unwrap(); - let RegionStageValue::Staged { .. } = runtime + let RegionStageValue::Staged { .. } = regions .try_stage_value( &staging, 0, @@ -1119,7 +1072,7 @@ fn same_hash_candidate_requires_full_key() { else { panic!("collision owner must stage"); }; - runtime + regions .flush_staging_shard( &staging, &engine, @@ -1129,14 +1082,14 @@ fn same_hash_candidate_requires_full_key() { .unwrap() .expect("collision owner must publish"); - let Some(entry) = runtime.lookup_snapshot(owner_hash).unwrap() else { + let Some(entry) = regions.lookup_snapshot(owner_hash).unwrap() else { panic!("collision owner must be indexed after publication"); }; let read_buffer_bytes = (entry.location.record_len() as usize).div_ceil(RECOVERY_PAGE_SIZE) * RECOVERY_PAGE_SIZE; let memory_before_read = managed_memory.snapshot().current_bytes; let read_buffer = managed_memory.try_read_buffer(read_buffer_bytes).unwrap(); - let entry = runtime + let entry = regions .begin_point_read(owner_hash) .expect("hash lookup must return the collision candidate"); let desc = describe_read(data.geometry, owner_hash, entry, true).unwrap(); @@ -1144,7 +1097,7 @@ fn same_hash_candidate_requires_full_key() { // Supplying a different key after the hash lookup precisely models a // 64-bit collision at the L2 record-validation boundary. assert!( - runtime + regions .read_value_from_desc( &engine, engine.try_reserve_read().unwrap(), @@ -1155,10 +1108,10 @@ fn same_hash_candidate_requires_full_key() { .unwrap() .is_none() ); - assert!(runtime.health.is_healthy()); + assert!(regions.health.is_healthy()); assert_eq!(managed_memory.snapshot().current_bytes, memory_before_read); - let current = runtime + let current = regions .begin_point_read(owner_hash) .expect("owner remains indexed"); let stale_generation = ReadCandidate { @@ -1167,7 +1120,7 @@ fn same_hash_candidate_requires_full_key() { }; let stale_desc = describe_read(data.geometry, owner_hash, stale_generation, true).unwrap(); assert!( - runtime + regions .read_value_from_desc( &engine, engine.try_reserve_read().unwrap(), @@ -1194,7 +1147,7 @@ fn same_hash_candidate_requires_full_key() { let wrong_length_desc = describe_read(data.geometry, owner_hash, wrong_length, true).unwrap(); let wrong_length_read_bytes = wrong_length_desc.read_len; assert!( - runtime + regions .read_value_from_desc( &engine, engine.try_reserve_read().unwrap(), @@ -1208,7 +1161,7 @@ fn same_hash_candidate_requires_full_key() { .is_none() ); - let hit = runtime + let hit = regions .read_value( &engine, data.geometry, @@ -1227,7 +1180,7 @@ fn same_hash_candidate_requires_full_key() { #[test] fn failed_span_write_never_publishes_and_latches_miss_only() { let data = data_path_superblock(); - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(64).unwrap(), empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) @@ -1241,12 +1194,12 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { ) .unwrap(); let directory = TestDirectory::new(); - let (backend, faults) = FaultFile::open(&directory.paths.data).unwrap(); - backend.set_len(data.geometry.data_file_len).unwrap(); - let engine = IoEngine::for_test(Arc::new(backend), 1).unwrap(); + let (file, faults) = FaultFile::open(&directory.paths.data).unwrap(); + file.set_len(data.geometry.data_file_len).unwrap(); + let engine = IoEngine::for_test(Arc::new(file), 1).unwrap(); let hash = hash_key(data.hash_seed, b"key"); let record_bytes = required_record_bytes(b"key".len(), 16 * 1024).unwrap(); - let RegionStageValue::Staged { .. } = runtime + let RegionStageValue::Staged { .. } = regions .try_stage_value(&staging, 0, hash, record_bytes, b"key", &[7; 16 * 1024]) .unwrap() else { @@ -1259,7 +1212,7 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { ); assert_eq!( - runtime + regions .flush_staging_shard( &staging, &engine, @@ -1272,11 +1225,11 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { .raw_os_error(), Some(5) ); - assert!(!runtime.health.is_healthy()); - assert_eq!(runtime.lookup_snapshot(hash).unwrap(), None); - assert_eq!(runtime.index.lookup_raw(hash).unwrap(), None); + assert!(!regions.health.is_healthy()); + assert_eq!(regions.lookup_snapshot(hash).unwrap(), None); + assert_eq!(regions.index.lookup_raw(hash).unwrap(), None); assert_eq!( - runtime.manager.state.lock().unwrap().regions()[0].completed_used, + regions.manager.state.lock().unwrap().regions()[0].completed_used, 0 ); engine.shutdown().unwrap(); @@ -1285,12 +1238,12 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { #[test] fn rotation_is_committed_without_a_metadata_io_boundary() { let data = data_path_superblock(); - let runtime = FileRegionRuntime::install( + let regions = RegionStore::from_recovery( PartitionedIndexStorage::anonymous(64).unwrap(), empty_region_metadata(data, 64, REGION_SHARDS).unwrap(), ) .unwrap(); - runtime + regions .manager .state .lock() @@ -1298,9 +1251,9 @@ fn rotation_is_committed_without_a_metadata_io_boundary() { .request_rotation_for_test(0) .unwrap(); - assert!(runtime.rotate_shard(0).unwrap()); - assert!(runtime.health.is_healthy()); - let manager = runtime.manager.lock().unwrap(); + assert!(regions.rotate_shard(0).unwrap()); + assert!(regions.health.is_healthy()); + let manager = regions.manager.lock().unwrap(); assert_eq!(manager.active_regions()[0], REGION_SHARDS); assert_eq!(manager.sealed_regions().back(), Some(&0)); } @@ -1325,14 +1278,11 @@ fn fresh_and_dirty_startup_do_not_write_runtime_region_metadata() { let data = test_data_superblock(); let (fresh_file_system, fresh_io, _) = FaultFileSystem::new(); - let mut fresh = CacheSession::open( + let mut fresh = CacheSession::for_test_with_file_system( + directory.paths.clone(), + data, config, - FileRegionBackend::for_test_with_file_system( - directory.paths.clone(), - data, - config, - fresh_file_system, - ), + fresh_file_system, ) .unwrap(); assert_eq!(fresh.startup(), StartupMode::Cold); @@ -1340,14 +1290,11 @@ fn fresh_and_dirty_startup_do_not_write_runtime_region_metadata() { fresh.close_fast().unwrap(); let (dirty_file_system, dirty_io, _) = FaultFileSystem::new(); - let mut dirty = CacheSession::open( + let mut dirty = CacheSession::for_test_with_file_system( + directory.paths.clone(), + data, config, - FileRegionBackend::for_test_with_file_system( - directory.paths.clone(), - data, - config, - dirty_file_system, - ), + dirty_file_system, ) .unwrap(); assert_eq!(dirty.startup(), StartupMode::Cold); @@ -1361,30 +1308,21 @@ fn publish_custom_clean_image( data: DataSuperblock, metadata: RegionMetadata, ) { - let shard_count = metadata.root.shard_count; - let runtime_options = RuntimeOptions { - append_shards: shard_count, - ..RuntimeOptions::default() - }; - let mut backend = FileRegionBackend::for_test_with_options( - directory.paths.clone(), - data, - index_slots, - runtime_options, + let mut persistence = RegionPersistence::new(directory.paths.clone(), data, OsFileSystem); + persistence.acquire_exclusive(IoMode::Buffered).unwrap(); + assert!(persistence.inspect_recovery(index_slots).unwrap().is_none()); + let regions = Arc::new( + RegionStore::from_recovery( + PartitionedIndexStorage::anonymous(index_slots).unwrap(), + metadata, + ) + .unwrap(), ); - backend.acquire_exclusive().unwrap(); - assert!(backend.inspect_recovery(index_slots).unwrap().is_none()); - let runtime = FileRegionRuntime::install( - PartitionedIndexStorage::anonymous(index_slots).unwrap(), - metadata, - ) - .unwrap(); - backend.publish_running().unwrap(); - let runtime = backend.start_runtime(runtime).unwrap(); - let frozen = backend.freeze_warm(runtime).unwrap(); - let prepared = backend.persist_frozen(&frozen).unwrap(); - backend.publish_clean(prepared).unwrap(); - backend.release_exclusive().unwrap(); + persistence.publish_running().unwrap(); + let frozen = regions.freeze().unwrap(); + let prepared = persistence.persist_frozen(&frozen).unwrap(); + persistence.publish_clean(prepared).unwrap(); + persistence.release_exclusive().unwrap(); } #[test] @@ -1396,27 +1334,19 @@ fn clean_image_rebinds_a_different_append_shard_topology() { let metadata = empty_region_metadata(data, config, 1).unwrap(); publish_custom_clean_image(&directory, config, data, metadata); - let mut reopened = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut reopened = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(reopened.startup(), StartupMode::Warm); - let manager = reopened.runtime().unwrap().manager.lock().unwrap(); + let manager = reopened.regions().unwrap().manager.lock().unwrap(); assert_eq!(manager.active_regions(), &[0, 2, 3, 4]); assert_eq!(manager.free_regions().len(), 1); drop(manager); reopened.close_warm().unwrap(); - let mut stable = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut stable = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(stable.startup(), StartupMode::Warm); assert_eq!( stable - .runtime() + .regions() .unwrap() .manager .lock() @@ -1445,13 +1375,9 @@ fn append_shard_growth_without_free_regions_cold_starts_safely() { metadata.validate().unwrap(); publish_custom_clean_image(&directory, config, data, metadata); - let mut reopened = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut reopened = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(reopened.startup(), StartupMode::Cold); - let manager = reopened.runtime().unwrap().manager.lock().unwrap(); + let manager = reopened.regions().unwrap().manager.lock().unwrap(); assert_eq!(manager.active_regions(), &[0, 1, 2, 3]); assert_eq!(manager.free_regions().len(), 1); drop(manager); @@ -1465,11 +1391,7 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let mut first = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut first = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); first.close_fast().unwrap(); let stale_offset = 2 * RECOVERY_PAGE_SIZE as u64; @@ -1481,11 +1403,7 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { file.write_all_at(b"stale-record", stale_offset).unwrap(); file.sync_data().unwrap(); - let mut cold = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut cold = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); let mut observed = [0xff_u8; 12]; File::open(&directory.paths.data) @@ -1496,43 +1414,18 @@ fn dirty_cold_start_discards_stale_region_bytes_without_scanning() { cold.close_fast().unwrap(); } -#[test] -fn invalid_capacity_is_rejected_before_creating_files() { - for index_slots in [0, 1, 7, usize::MAX] { - let directory = TestDirectory::new(); - let opened = CacheSession::open( - index_slots, - FileRegionBackend::for_test(directory.paths.clone(), test_data_superblock(), 8), - ); - assert_eq!(opened.err().unwrap().kind(), io::ErrorKind::InvalidInput); - assert_eq!(fs::read_dir(&directory.root).unwrap().count(), 0); - } -} - #[test] fn dropping_a_warm_runtime_releases_ownership_and_reopens_cold() { let directory = TestDirectory::new(); let data = test_data_superblock(); - let mut initial = CacheSession::open( - 8, - FileRegionBackend::for_test(directory.paths.clone(), data, 8), - ) - .unwrap(); + let mut initial = CacheSession::for_test(directory.paths.clone(), data, 8).unwrap(); initial.close_warm().unwrap(); - let recovered = CacheSession::open( - 8, - FileRegionBackend::for_test(directory.paths.clone(), data, 8), - ) - .unwrap(); + let recovered = CacheSession::for_test(directory.paths.clone(), data, 8).unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); drop(recovered); - let mut cold = CacheSession::open( - 8, - FileRegionBackend::for_test(directory.paths.clone(), data, 8), - ) - .unwrap(); + let mut cold = CacheSession::for_test(directory.paths.clone(), data, 8).unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); cold.close_fast().unwrap(); } @@ -1550,12 +1443,8 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { }, }); - let mut first = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); - let runtime = first.runtime_mut().unwrap(); + let mut first = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); + let runtime = first.regions().unwrap(); assert_eq!(runtime.index.storage().partition_count(), 2); runtime .index @@ -1567,18 +1456,14 @@ fn complete_warm_image_maps_without_rebuilding_index_slots() { first.close_warm().unwrap(); first.close_fast().unwrap(); assert_eq!( - first.runtime().err().unwrap().kind(), + first.regions().err().unwrap().kind(), io::ErrorKind::BrokenPipe ); drop(first); - let mut recovered = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut recovered = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); - let recovered_runtime = recovered.runtime().unwrap(); + let recovered_runtime = recovered.regions().unwrap(); assert_eq!(recovered_runtime.index.storage().partition_count(), 2); assert_eq!( recovered_runtime @@ -1598,11 +1483,7 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { let directory = TestDirectory::new(); let config = 130; let data = test_data_superblock_with_regions(REGION_SHARDS + 1); - let mut first = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut first = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); first.close_warm().unwrap(); let image = File::options() @@ -1620,14 +1501,10 @@ fn corrupt_region_metadata_rejects_the_complete_clean_image() { .unwrap(); image.sync_data().unwrap(); - let mut rejected = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut rejected = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(rejected.startup(), StartupMode::Cold); assert_eq!( - rejected.runtime().unwrap().lookup_snapshot(0).unwrap(), + rejected.regions().unwrap().lookup_snapshot(0).unwrap(), None ); rejected.close_fast().unwrap(); @@ -1640,11 +1517,7 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { let directory = TestDirectory::new(); let config = INDEX_IMAGE_SLOTS_PER_PAGE + 8; let data = test_data_superblock_with_regions(REGION_SHARDS + 1); - let mut first = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut first = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); first.close_warm().unwrap(); let image = File::options() @@ -1660,13 +1533,9 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { .unwrap(); image.sync_data().unwrap(); - let mut recovered = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut recovered = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(recovered.startup(), StartupMode::Warm); - let runtime = recovered.runtime().unwrap(); + let runtime = recovered.regions().unwrap(); assert_eq!(runtime.lookup_snapshot(0).unwrap(), None); assert!(runtime.health.is_healthy()); assert_eq!(runtime.lookup_snapshot(1).unwrap(), None); @@ -1675,13 +1544,9 @@ fn one_corrupt_lazy_index_page_rejects_all_pages() { assert_eq!(runtime.lookup_snapshot(0).unwrap(), None); assert!(recovered.close_warm().is_err()); - let mut cold = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut cold = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); - assert_eq!(cold.runtime().unwrap().lookup_snapshot(1).unwrap(), None); + assert_eq!(cold.regions().unwrap().lookup_snapshot(1).unwrap(), None); cold.close_fast().unwrap(); } @@ -1746,13 +1611,13 @@ fn every_prepublication_failure_leaves_no_selectable_clean_state() { let config = 8; let data = test_data_superblock_with_regions(REGION_SHARDS + 1); let (file_system, io_faults, file_system_faults) = FaultFileSystem::new(); - let backend = FileRegionBackend::for_test_with_file_system( + let mut session = CacheSession::for_test_with_file_system( directory.paths.clone(), data, config, file_system, - ); - let mut session = CacheSession::open(config, backend).unwrap(); + ) + .unwrap(); if let Some((event, action)) = io_fault { io_faults.arm(event, 1, action); } @@ -1761,11 +1626,7 @@ fn every_prepublication_failure_leaves_no_selectable_clean_state() { } assert!(session.close_warm().is_err(), "failure case {case}"); - let mut reopened = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut reopened = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert_eq!(reopened.startup(), StartupMode::Cold, "failure case {case}"); reopened.close_fast().unwrap(); } @@ -1800,14 +1661,11 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { let data = test_data_superblock(); let (file_system, faults, _) = FaultFileSystem::new(); faults.arm(event, occurrence, action); - let opened = CacheSession::open( + let opened = CacheSession::for_test_with_file_system( + directory.paths.clone(), + data, config, - FileRegionBackend::for_test_with_file_system( - directory.paths.clone(), - data, - config, - file_system, - ), + file_system, ); assert!(opened.is_err(), "RUNNING barrier case {case}"); assert!( @@ -1815,11 +1673,7 @@ fn concrete_running_barrier_failures_abort_before_runtime_start() { "runtime paths must not be requested before RUNNING is durable" ); - let mut cold = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut cold = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert!(matches!(cold.startup(), StartupMode::Cold)); cold.close_fast().unwrap(); } @@ -1831,10 +1685,8 @@ fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { let data = test_data_superblock(); let (file_system, faults, _) = FaultFileSystem::new(); faults.arm(FaultEvent::CloneDataHandles, 1, FaultAction::Error(5)); - let opened = CacheSession::open( - 8, - FileRegionBackend::for_test_with_file_system(directory.paths.clone(), data, 8, file_system), - ); + let opened = + CacheSession::for_test_with_file_system(directory.paths.clone(), data, 8, file_system); assert_eq!(opened.err().unwrap().raw_os_error(), Some(5)); assert!(faults.events().contains(&FaultEvent::CloneDataHandles)); assert!( @@ -1843,11 +1695,7 @@ fn runtime_file_failure_after_running_releases_ownership_and_reopens_cold() { .contains(&FaultEvent::Sync(SyncPoint::RunningState)) ); - let mut cold = CacheSession::open( - 8, - FileRegionBackend::for_test(directory.paths.clone(), data, 8), - ) - .unwrap(); + let mut cold = CacheSession::for_test(directory.paths.clone(), data, 8).unwrap(); assert_eq!(cold.startup(), StartupMode::Cold); cold.close_fast().unwrap(); } @@ -1858,16 +1706,9 @@ fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { let config = 8; let data = test_data_superblock(); let (file_system, faults, _) = FaultFileSystem::new(); - let mut session = CacheSession::open( - config, - FileRegionBackend::for_test_with_file_system( - directory.paths.clone(), - data, - config, - file_system, - ), - ) - .unwrap(); + let mut session = + CacheSession::for_test_with_file_system(directory.paths.clone(), data, config, file_system) + .unwrap(); faults.arm( FaultEvent::Sync(SyncPoint::CleanState), 1, @@ -1875,17 +1716,13 @@ fn final_clean_sync_failure_reopens_as_safe_clean_or_empty() { ); assert!(session.close_warm().is_err()); - let mut reopened = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut reopened = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); assert!(matches!( reopened.startup(), StartupMode::Warm | StartupMode::Cold )); assert_eq!( - reopened.runtime().unwrap().lookup_snapshot(0).unwrap(), + reopened.regions().unwrap().lookup_snapshot(0).unwrap(), None ); reopened.close_fast().unwrap(); @@ -1900,10 +1737,7 @@ fn data_and_state_inode_alias_is_rejected_without_truncation() { fs::write(&directory.paths.data, marker).unwrap(); fs::hard_link(&directory.paths.data, &directory.paths.state).unwrap(); - let opened = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ); + let opened = CacheSession::for_test(directory.paths.clone(), data, config); assert!(matches!( opened, Err(error) if error.kind() == io::ErrorKind::InvalidInput @@ -1920,13 +1754,10 @@ fn recovery_temporary_path_cannot_name_the_data_or_state_file() { fs::write(&data_path, marker).unwrap(); let paths = RegionPaths::new(&data_path, directory.root.join("state"), image); - let opened = CacheSession::open( + let opened = CacheSession::for_test( + paths, + test_data_superblock_with_regions(REGION_SHARDS + 1), 8, - FileRegionBackend::for_test( - paths, - test_data_superblock_with_regions(REGION_SHARDS + 1), - 8, - ), ); assert!(matches!( opened, @@ -1945,10 +1776,7 @@ fn recovery_sidecars_must_share_one_directory() { other.join("state"), directory.root.join("image"), ); - let opened = CacheSession::open( - 8, - FileRegionBackend::for_test(paths, test_data_superblock(), 8), - ); + let opened = CacheSession::for_test(paths, test_data_superblock(), 8); assert!(matches!( opened, Err(error) if error.kind() == io::ErrorKind::InvalidInput @@ -1962,21 +1790,14 @@ fn state_sidecar_lock_prevents_cross_data_file_races() { let directory = TestDirectory::new(); let config = 8; let data = test_data_superblock(); - let mut first = CacheSession::open( - config, - FileRegionBackend::for_test(directory.paths.clone(), data, config), - ) - .unwrap(); + let mut first = CacheSession::for_test(directory.paths.clone(), data, config).unwrap(); let conflicting_files = RegionPaths::new( directory.root.join("other-data"), directory.paths.state.clone(), directory.root.join("other-image"), ); - let opened = CacheSession::open( - config, - FileRegionBackend::for_test(conflicting_files, data, config), - ); + let opened = CacheSession::for_test(conflicting_files, data, config); assert!(opened.is_err()); first.close_fast().unwrap(); } From dac663bec96153a3506165a78039de55258a8dfd Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:23:30 +0800 Subject: [PATCH 06/12] refactor: distinguish I/O recovery from persistent session state Name the shared timeout controller IoRecovery and its per-operation fence IoRecoveryAttempt. Place the protocol under io so engines and Region workers share it without implying that it belongs to one engine. Name the persisted EMPTY/RUNNING/CLEAN enum SessionState and update the associated calls and locals. Keep state encodings, recovery deadlines, admission fences, public health values, and diagnostic event names intact. --- cache2/src/cache/runtime/mod.rs | 30 +++---- cache2/src/cache/runtime/shutdown_tests.rs | 4 +- cache2/src/io/engine/mod.rs | 13 ++-- cache2/src/io/engine/tests.rs | 40 +++++----- cache2/src/io/mod.rs | 1 + cache2/src/io/{engine => }/recovery.rs | 27 +++---- cache2/src/region/appender.rs | 26 +++---- cache2/src/region/mod.rs | 10 +-- cache2/src/region/persistence/mod.rs | 8 +- cache2/src/region/persistence/tests.rs | 8 +- cache2/src/region/recovery/mod.rs | 91 +++++++++++----------- 11 files changed, 127 insertions(+), 131 deletions(-) rename cache2/src/io/{engine => }/recovery.rs (79%) diff --git a/cache2/src/cache/runtime/mod.rs b/cache2/src/cache/runtime/mod.rs index 9a9a855..54d46b9 100644 --- a/cache2/src/cache/runtime/mod.rs +++ b/cache2/src/cache/runtime/mod.rs @@ -57,9 +57,9 @@ use crate::io::engine::IoOperation; use crate::io::engine::ReadSlot; use crate::io::engine::ReadSlotWaiter; use crate::io::engine::build_file_engine; -use crate::io::engine::recovery::BackgroundRecovery; use crate::io::engine::submit_background_io; use crate::io::file::DataFileHandles; +use crate::io::recovery::IoRecovery; use crate::managed_memory::BufferLease; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::ManagedMemory; @@ -437,7 +437,7 @@ struct RuntimeState { reclaim_engines: Box<[Arc]>, reclaim_control: ReclaimControl, reclaim_io_timeout: Duration, - recovery: BackgroundRecovery, + io_recovery: IoRecovery, managed_memory: Arc, metrics: Arc, memory: Arc, @@ -795,7 +795,7 @@ impl CacheRuntime { } pub fn start_close(&self) { - self.state.recovery.stop(); + self.state.io_recovery.stop(); self.operations.start_close(); } @@ -823,7 +823,7 @@ impl CacheRuntime { )); } let state = &self.state; - if state.recovery.is_recovering() { + if state.io_recovery.is_recovering() { if state.activity_counters { state.metrics.record_write_rejection(); } @@ -1284,7 +1284,7 @@ impl CacheRuntime { state.memory.metrics_snapshot(), ); if snapshot.health == crate::snapshot::CacheHealth::Running - && state.recovery.is_recovering() + && state.io_recovery.is_recovering() { snapshot.health = crate::snapshot::CacheHealth::Recovering; } @@ -1504,7 +1504,7 @@ fn start_workers( reclaim_engines, reclaim_control: ReclaimControl::new(), reclaim_io_timeout: options.reclaim_io_timeout, - recovery: BackgroundRecovery::new(options.io_recovery_timeout), + io_recovery: IoRecovery::new(options.io_recovery_timeout), managed_memory, metrics, memory, @@ -1646,7 +1646,7 @@ fn append_worker(state: Arc, shard_id: usize) { ); } state.regions.enter_miss_only(); - state.recovery.stop(); + state.io_recovery.stop(); control.fail(&error); // Wake engine admission in case another shard is blocked behind work that // can no longer make progress after this runtime entered miss-only. @@ -1713,7 +1713,7 @@ fn reclaim_worker( .lifecycle .store(LIFECYCLE_FAILED, Ordering::Release); state.regions.enter_miss_only(); - state.recovery.stop(); + state.io_recovery.stop(); log::error!( target: "cache2::health", event = "cache_reclaim_worker_failed", @@ -1750,7 +1750,7 @@ fn reclaim_worker_result( if state.reclaim_control.is_stopped()? { return Ok(()); } - let mut recovery = state.recovery.attempt(); + let mut attempt = state.io_recovery.attempt(); let Some(receipt) = state.regions.begin_reclaim()? else { break; }; @@ -1777,11 +1777,11 @@ fn reclaim_worker_result( engine.as_ref(), IoOperation::read(io_buffer, absolute), state.reclaim_io_timeout, - &mut recovery, + &mut attempt, ) .map_err(|error| error.into_lease().0)?; let completion = request - .wait_with_recovery(engine.as_ref(), &mut recovery) + .wait_with_io_recovery(engine.as_ref(), &mut attempt) .map_err(|error| error.into_lease().0)?; let (result, returned) = completion.into_lease(); let transferred = result?; @@ -1845,7 +1845,7 @@ fn reclaim_worker_result( state.append_controls[reinsert_shard].wait_for_drain(generation)?; } state.regions.complete_reclaim(receipt)?; - recovery.finish(); + attempt.finish(); drop(reinsert_operation); if state.activity_counters { state.metrics.record_reclaim(stats); @@ -1899,7 +1899,7 @@ fn append_worker_result( &state.staging, engine.as_ref(), shard_id, - &state.recovery, + &state.io_recovery, )?; deadline = None; } @@ -1934,7 +1934,7 @@ fn append_worker_result( &state.staging, engine.as_ref(), shard_id, - &state.recovery, + &state.io_recovery, )?; } Ok(None) => {} @@ -2070,7 +2070,7 @@ async fn drain_shards_async(state: &RuntimeState, stop: bool) -> io::Result<()> } fn stop_workers(mut workers: RuntimeWorkers) -> io::Result { - workers.state.recovery.stop(); + workers.state.io_recovery.stop(); let drain = drain_shards(&workers.state, true); let mut join_error = None; for worker in workers.append_workers.drain(..) { diff --git a/cache2/src/cache/runtime/shutdown_tests.rs b/cache2/src/cache/runtime/shutdown_tests.rs index 33aa90b..05c53ee 100644 --- a/cache2/src/cache/runtime/shutdown_tests.rs +++ b/cache2/src/cache/runtime/shutdown_tests.rs @@ -204,8 +204,8 @@ fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { let runtime = session.runtime().unwrap().clone(); runtime.put(b"existing", b"value").unwrap(); runtime.drain().unwrap(); - let mut first = runtime.state.recovery.attempt(); - let mut second = runtime.state.recovery.attempt(); + let mut first = runtime.state.io_recovery.attempt(); + let mut second = runtime.state.io_recovery.attempt(); assert!(first.next_deadline(Instant::now()).is_some()); assert!(second.next_deadline(Instant::now()).is_some()); assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Recovering); diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index 6dc77d4..50b01eb 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -81,9 +81,8 @@ use crate::managed_memory::BufferLease; use crate::snapshot::CacheIoDirectionSnapshot; mod posix; -pub mod recovery; -use self::recovery::RecoveryAttempt; +use crate::io::recovery::IoRecoveryAttempt; #[cfg(all( feature = "io-uring", @@ -770,10 +769,10 @@ impl BoundedIoRequest { /// Allows an issued background operation to finish without cancelling it /// at the normal deadline. The same request keeps its slot and buffer; /// callers must still validate completion before publishing or reusing it. - pub fn wait_with_recovery( + pub fn wait_with_io_recovery( mut self, engine: &IoEngine, - recovery: &mut RecoveryAttempt<'_>, + attempt: &mut IoRecoveryAttempt<'_>, ) -> Result { let original = self.deadline; loop { @@ -781,7 +780,7 @@ impl BoundedIoRequest { Ok(completion) => return Ok(completion), Err(request) => request, }; - match recovery.next_deadline(original) { + match attempt.next_deadline(original) { Some(deadline) => self.deadline = deadline, None => return self.wait(engine), } @@ -956,7 +955,7 @@ pub fn submit_background_io( engine: &IoEngine, mut operation: IoOperation, timeout: Duration, - recovery: &mut RecoveryAttempt<'_>, + attempt: &mut IoRecoveryAttempt<'_>, ) -> Result { let original = Instant::now() .checked_add(timeout) @@ -969,7 +968,7 @@ pub fn submit_background_io( return Ok(request); } Err(error) if error.error.kind() == io::ErrorKind::TimedOut => { - let Some(next) = recovery.next_deadline(original) else { + let Some(next) = attempt.next_deadline(original) else { return Err(error); }; deadline = next; diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index 05d37b2..b4d816b 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -15,13 +15,13 @@ use std::sync::mpsc; use std::time::Duration; -use super::recovery::BackgroundRecovery; use super::*; use crate::IoOutcome; use crate::IoRole; use crate::StatsOptions; use crate::fixtures::TestFile; use crate::io::file::PositionedIo; +use crate::io::recovery::IoRecovery; use crate::managed_memory::ManagedMemory; use crate::managed_memory::ManagedMemoryLimits; use crate::managed_memory::aligned_buffer_capacity; @@ -1128,9 +1128,9 @@ fn background_recovery_keeps_the_original_request_and_accepts_late_completion() let (tx, rx) = mpsc::channel(); let engine = &engine; scope.spawn(move || { - tx.send(request.wait_with_recovery( + tx.send(request.wait_with_io_recovery( engine, - &mut BackgroundRecovery::new(Some(Duration::from_secs(2))).attempt(), + &mut IoRecovery::new(Some(Duration::from_secs(2))).attempt(), )) .unwrap(); }); @@ -1175,9 +1175,9 @@ fn exhausted_background_recovery_still_fences_unfinished_writes() { assert!(io.wait_for_entered(1)); request.deadline = Instant::now(); request.cancel_grace = Duration::from_millis(10); - let result = request.wait_with_recovery( + let result = request.wait_with_io_recovery( &engine, - &mut BackgroundRecovery::new(Some(Duration::from_millis(20))).attempt(), + &mut IoRecovery::new(Some(Duration::from_millis(20))).attempt(), ); let pending = engine.writes_in_flight(); let rejected = engine.submit(IoOperation::write( @@ -1211,15 +1211,15 @@ fn background_admission_recovers_without_duplicate_submission() { let engine = &engine; let managed_memory = &managed_memory; scope.spawn(move || { - let recovery = BackgroundRecovery::new(Some(Duration::from_secs(2))); - let mut attempt = recovery.attempt(); + let io_recovery = IoRecovery::new(Some(Duration::from_secs(2))); + let mut attempt = io_recovery.attempt(); let result = submit_background_io( engine, IoOperation::read(read_buffer(managed_memory, 4096), 0), Duration::from_millis(10), &mut attempt, ) - .map(|request| request.wait_with_recovery(engine, &mut attempt)); + .map(|request| request.wait_with_io_recovery(engine, &mut attempt)); tx.send(result).unwrap(); }); let early = rx.recv_timeout(Duration::from_millis(50)); @@ -1245,7 +1245,7 @@ fn unlimited_recovery_keeps_admission_paused_until_validation() { let io = Arc::new(BlockingIo::default()); let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); - let recovery = BackgroundRecovery::new(None); + let io_recovery = IoRecovery::new(None); let mut request = submit_cache_io( &engine, IoOperation::write( @@ -1260,30 +1260,30 @@ fn unlimited_recovery_keeps_admission_paused_until_validation() { std::thread::scope(|scope| { let (completed_tx, completed_rx) = mpsc::channel(); let (validate_tx, validate_rx) = mpsc::channel(); - let recovery = &recovery; + let io_recovery = &io_recovery; let engine = &engine; scope.spawn(move || { - let mut attempt = recovery.attempt(); - let completion = request.wait_with_recovery(engine, &mut attempt).unwrap(); + let mut attempt = io_recovery.attempt(); + let completion = request.wait_with_io_recovery(engine, &mut attempt).unwrap(); completed_tx.send(completion).unwrap(); validate_rx.recv().unwrap(); attempt.finish(); }); // Cross more than one polling interval while preserving the same I/O. let early = completed_rx.recv_timeout(Duration::from_millis(1100)); - let paused = recovery.is_recovering(); + let paused = io_recovery.is_recovering(); io.release(); assert!(matches!(early, Err(mpsc::RecvTimeoutError::Timeout))); assert!(paused); let completion = completed_rx.recv_timeout(Duration::from_secs(2)).unwrap(); assert!( - recovery.is_recovering(), + io_recovery.is_recovering(), "completion delivery alone must not resume fills" ); assert!(completion.into_io_result().0.is_ok()); validate_tx.send(()).unwrap(); }); - assert!(!recovery.is_recovering()); + assert!(!io_recovery.is_recovering()); assert_eq!(lock_unpoisoned(&io.state).entered, 1); engine.shutdown().unwrap(); } @@ -1293,7 +1293,7 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { let io = Arc::new(BlockingIo::default()); let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let managed_memory = managed_memory(); - let recovery = BackgroundRecovery::new(None); + let io_recovery = IoRecovery::new(None); let mut request = submit_cache_io( &engine, IoOperation::write( @@ -1307,15 +1307,15 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { request.deadline = Instant::now(); std::thread::scope(|scope| { let (tx, rx) = mpsc::channel(); - let recovery = &recovery; + let io_recovery = &io_recovery; let engine = &engine; scope.spawn(move || { - let mut attempt = recovery.attempt(); - tx.send(request.wait_with_recovery(engine, &mut attempt)) + let mut attempt = io_recovery.attempt(); + tx.send(request.wait_with_io_recovery(engine, &mut attempt)) .unwrap(); }); let early = rx.recv_timeout(Duration::from_millis(30)); - recovery.stop(); + io_recovery.stop(); let stopped = rx.recv_timeout(Duration::from_secs(2)); let pending = engine.writes_in_flight(); let charged = managed_memory.snapshot().current_bytes; diff --git a/cache2/src/io/mod.rs b/cache2/src/io/mod.rs index 7364595..d250f76 100644 --- a/cache2/src/io/mod.rs +++ b/cache2/src/io/mod.rs @@ -17,3 +17,4 @@ pub mod engine; pub mod file; pub mod fs; +pub mod recovery; diff --git a/cache2/src/io/engine/recovery.rs b/cache2/src/io/recovery.rs similarity index 79% rename from cache2/src/io/engine/recovery.rs rename to cache2/src/io/recovery.rs index 23ea450..2c2b374 100644 --- a/cache2/src/io/engine/recovery.rs +++ b/cache2/src/io/recovery.rs @@ -12,7 +12,8 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Reversible background timeout recovery, separate from the health latch. +//! Recovery of timed-out background I/O while retaining issued requests. +//! This admission fence is separate from restart recovery and the terminal health latch. use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicUsize; @@ -21,13 +22,13 @@ use std::time::Duration; use std::time::Instant; /// Shared across background workers. Resource ownership remains with each worker. -pub struct BackgroundRecovery { +pub struct IoRecovery { timeout: Option, pending: AtomicUsize, stopped: AtomicBool, } -impl BackgroundRecovery { +impl IoRecovery { pub const fn new(timeout: Option) -> Self { Self { timeout, @@ -36,9 +37,9 @@ impl BackgroundRecovery { } } - pub fn attempt(&self) -> RecoveryAttempt<'_> { - RecoveryAttempt { - recovery: self, + pub fn attempt(&self) -> IoRecoveryAttempt<'_> { + IoRecoveryAttempt { + io_recovery: self, entered: false, } } @@ -54,21 +55,21 @@ impl BackgroundRecovery { /// Retains the admission fence until the caller validates and publishes success. /// Failed attempts deliberately leave the fence raised until instance teardown. -pub struct RecoveryAttempt<'a> { - recovery: &'a BackgroundRecovery, +pub struct IoRecoveryAttempt<'a> { + io_recovery: &'a IoRecovery, entered: bool, } -impl RecoveryAttempt<'_> { +impl IoRecoveryAttempt<'_> { /// Poll completion/admission at fixed one-second intervals, without extending /// a configured total budget. Shutdown also terminates unlimited recovery. pub fn next_deadline(&mut self, original: Instant) -> Option { - if self.recovery.stopped.load(Ordering::Acquire) { + if self.io_recovery.stopped.load(Ordering::Acquire) { return None; } let now = Instant::now(); let deadline = now.checked_add(Duration::from_secs(1))?; - let deadline = match self.recovery.timeout { + let deadline = match self.io_recovery.timeout { Some(timeout) => { let end = original.checked_add(timeout)?; if now >= end { @@ -80,7 +81,7 @@ impl RecoveryAttempt<'_> { }; if !self.entered { self.entered = true; - if self.recovery.pending.fetch_add(1, Ordering::AcqRel) == 0 { + if self.io_recovery.pending.fetch_add(1, Ordering::AcqRel) == 0 { log::warn!(target: "cache2::health", event = "cache_io_recovery_started"; "background I/O timed out; pausing cache fills while retaining owned requests"); } @@ -90,7 +91,7 @@ impl RecoveryAttempt<'_> { /// Called only after operation-result validation and publication succeed. pub fn finish(self) { - if self.entered && self.recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1 { + if self.entered && self.io_recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1 { log::info!(target: "cache2::health", event = "cache_io_recovery_completed"; "all timed-out background operations recovered and passed validation"); } diff --git a/cache2/src/region/appender.rs b/cache2/src/region/appender.rs index 4371167..e32d8d0 100644 --- a/cache2/src/region/appender.rs +++ b/cache2/src/region/appender.rs @@ -30,12 +30,12 @@ use crate::io::engine::IoEngine; use crate::io::engine::IoOperation; use crate::io::engine::OperationKind; use crate::io::engine::RequestId; -#[cfg(test)] -use crate::io::engine::recovery::BackgroundRecovery; -use crate::io::engine::recovery::RecoveryAttempt; use crate::io::engine::submit_background_io; use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::io::file::WritePoint; +#[cfg(test)] +use crate::io::recovery::IoRecovery; +use crate::io::recovery::IoRecoveryAttempt; use crate::region::manager::RegionWriteSpan; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; @@ -86,9 +86,9 @@ impl RegionSpanFlight { pub fn wait( self, engine: &IoEngine, - recovery: &mut RecoveryAttempt<'_>, + attempt: &mut IoRecoveryAttempt<'_>, ) -> RegionSpanCompletion { - let completion = match self.request.wait_with_recovery(engine, recovery) { + let completion = match self.request.wait_with_io_recovery(engine, attempt) { Ok(completion) => completion, Err(timeout) => { let (error, buffer) = timeout.into_buffer(); @@ -147,7 +147,7 @@ pub fn submit_span( span: RegionWriteSpan, buffer: IoBuffer, absolute: u64, - recovery: &mut RecoveryAttempt<'_>, + attempt: &mut IoRecoveryAttempt<'_>, ) -> Result { let (expected_len, expected_absolute) = match validate_span(geometry, span) { Ok(validated) => validated, @@ -187,7 +187,7 @@ pub fn submit_span( engine, IoOperation::write(WritePoint::Record, buffer, absolute), CACHE_IO_COMPLETION_TIMEOUT, - recovery, + attempt, ) { Ok(request) => request, Err(error) => { @@ -311,8 +311,8 @@ mod tests { let mut lease = BufferLease::try_fixed(4096).unwrap(); lease.prepare(4096).unwrap().fill(0x5a); let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; - let recovery = BackgroundRecovery::new(Some(Duration::from_secs(5))); - let mut attempt = recovery.attempt(); + let io_recovery = IoRecovery::new(Some(Duration::from_secs(5))); + let mut attempt = io_recovery.attempt(); let completion = submit_span( &engine, geometry(), @@ -325,7 +325,7 @@ mod tests { .wait(&engine, &mut attempt); assert!(completion.result.is_ok()); attempt.finish(); - assert!(!recovery.is_recovering()); + assert!(!io_recovery.is_recovering()); assert_eq!(completion.span, span()); assert_eq!( completion.buffer.unwrap().as_slice().unwrap(), @@ -345,8 +345,8 @@ mod tests { let buffer = IoBuffer::for_write(lease, 4096).unwrap(); let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; - let recovery = BackgroundRecovery::new(Some(Duration::ZERO)); - let mut attempt = recovery.attempt(); + let io_recovery = IoRecovery::new(Some(Duration::ZERO)); + let mut attempt = io_recovery.attempt(); let completion = submit_span(&engine, geometry(), span(), buffer, absolute, &mut attempt) .unwrap() .wait(&engine, &mut attempt); @@ -383,7 +383,7 @@ mod tests { invalid, buffer, 0, - &mut BackgroundRecovery::new(Some(Duration::ZERO)).attempt(), + &mut IoRecovery::new(Some(Duration::ZERO)).attempt(), ) { Err(error) => error, Ok(_) => panic!("unaligned span must not be submitted"), diff --git a/cache2/src/region/mod.rs b/cache2/src/region/mod.rs index 0265b13..1497f0c 100644 --- a/cache2/src/region/mod.rs +++ b/cache2/src/region/mod.rs @@ -1029,9 +1029,9 @@ impl RegionStore { staging: &AppendStaging, engine: &IoEngine, shard_id: usize, - recovery: &crate::io::engine::recovery::BackgroundRecovery, + io_recovery: &crate::io::recovery::IoRecovery, ) -> io::Result> { - let mut recovery = recovery.attempt(); + let mut attempt = io_recovery.attempt(); let shard_mutation = self.lock_shard_mutation(shard_id)?; let geometry_for = |manager: &RegionManager| { let region_count = u32::try_from(manager.regions().len()).map_err(|_| { @@ -1126,7 +1126,7 @@ impl RegionStore { absolute, records, } = job; - let flight = match submit_span(engine, geometry, span, buffer, absolute, &mut recovery) { + let flight = match submit_span(engine, geometry, span, buffer, absolute, &mut attempt) { Ok(flight) => flight, Err(error) => { let original = error.error; @@ -1134,7 +1134,7 @@ impl RegionStore { return Err(original); } }; - let completion = flight.wait(engine, &mut recovery); + let completion = flight.wait(engine, &mut attempt); let RegionSpanCompletion { span, result, @@ -1176,7 +1176,7 @@ impl RegionStore { self.health.enter_miss_only(); return Err(staging_io_error(error)); } - recovery.finish(); + attempt.finish(); Ok(Some(span)) } diff --git a/cache2/src/region/persistence/mod.rs b/cache2/src/region/persistence/mod.rs index 77289b9..7d34339 100644 --- a/cache2/src/region/persistence/mod.rs +++ b/cache2/src/region/persistence/mod.rs @@ -53,10 +53,10 @@ use crate::region::recovery::RECOVERY_IMAGE_INDEX_OFFSET; use crate::region::recovery::RECOVERY_PAGE_SIZE; use crate::region::recovery::RecoveryImageHeader; use crate::region::recovery::RecoveryImageHeaderProbe; -use crate::region::recovery::RecoveryState; use crate::region::recovery::STATE_FILE_SIZE; use crate::region::recovery::STATE_SLOT_COUNT; use crate::region::recovery::SelectedState; +use crate::region::recovery::SessionState; use crate::region::recovery::StateBinding; use crate::region::recovery::StatePageWrite; use crate::region::recovery::StateRecord; @@ -325,7 +325,7 @@ impl RegionPersistence { let Some(selected) = recovery_state else { return self.cold_recovery(state_rejection.unwrap_or("no_valid_state")); }; - if selected.record.state != RecoveryState::Clean { + if selected.record.state != SessionState::Clean { return self.cold_recovery("unclean_shutdown"); } if !selected.record.binding.matches_data(data) { @@ -667,7 +667,7 @@ impl RegionPersistence { })?; let clean_state = prepare_next_state( self.current_state, - RecoveryState::Clean, + SessionState::Clean, StateBinding::from_data(data, Some(header.image_binding())), ) .map_err(|_| io::Error::other("CLEAN generation cannot advance"))?; @@ -741,7 +741,7 @@ impl RegionPersistence { let data = self.data_superblock()?; let expected = prepare_next_state( self.current_state, - RecoveryState::Clean, + SessionState::Clean, prepared.state.record.binding, ) .map_err(|_| io::Error::other("CLEAN generation cannot advance"))?; diff --git a/cache2/src/region/persistence/tests.rs b/cache2/src/region/persistence/tests.rs index 57dc0db..476e2a6 100644 --- a/cache2/src/region/persistence/tests.rs +++ b/cache2/src/region/persistence/tests.rs @@ -924,7 +924,7 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { &staging, &engine, 0, - &crate::io::engine::recovery::BackgroundRecovery::new(Some(std::time::Duration::ZERO)), + &crate::io::recovery::IoRecovery::new(Some(std::time::Duration::ZERO)), ) .unwrap() .unwrap(); @@ -1077,7 +1077,7 @@ fn same_hash_candidate_requires_full_key() { &staging, &engine, 0, - &crate::io::engine::recovery::BackgroundRecovery::new(Some(std::time::Duration::ZERO)), + &crate::io::recovery::IoRecovery::new(Some(std::time::Duration::ZERO)), ) .unwrap() .expect("collision owner must publish"); @@ -1217,9 +1217,7 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { &staging, &engine, 0, - &crate::io::engine::recovery::BackgroundRecovery::new(Some( - std::time::Duration::ZERO - )) + &crate::io::recovery::IoRecovery::new(Some(std::time::Duration::ZERO)) ) .unwrap_err() .raw_os_error(), diff --git a/cache2/src/region/recovery/mod.rs b/cache2/src/region/recovery/mod.rs index af13fe7..94443d2 100644 --- a/cache2/src/region/recovery/mod.rs +++ b/cache2/src/region/recovery/mod.rs @@ -288,15 +288,16 @@ impl DataSuperblock { } } +/// Durable session status; CLEAN alone makes a recovery image eligible. #[repr(u8)] #[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub enum RecoveryState { +pub enum SessionState { Empty = 0, Running = 1, Clean = 2, } -impl RecoveryState { +impl SessionState { fn decode(value: u8) -> Option { match value { 0 => Some(Self::Empty), @@ -548,7 +549,7 @@ impl StateBinding { #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub struct StateRecord { pub generation: u64, - pub state: RecoveryState, + pub state: SessionState, pub binding: StateBinding, } @@ -635,7 +636,7 @@ impl StateRecord { return StateSlotProbe::Corrupt; } - let Some(state) = RecoveryState::decode(page[STATE_KIND_OFFSET]) else { + let Some(state) = SessionState::decode(page[STATE_KIND_OFFSET]) else { return StateSlotProbe::Corrupt; }; let Some(cache_uuid) = get_id(page, STATE_CACHE_UUID_OFFSET) else { @@ -689,7 +690,7 @@ impl StateRecord { } fn matches_clean(self, data: DataSuperblock, image: ImageBinding) -> bool { - self.state == RecoveryState::Clean + self.state == SessionState::Clean && self.binding.matches_data(data) && self.binding.image == Some(image) } @@ -697,7 +698,7 @@ impl StateRecord { fn is_valid(self) -> bool { self.generation != 0 && self.binding.is_valid() - && (self.state != RecoveryState::Clean || self.binding.image.is_some()) + && (self.state != SessionState::Clean || self.binding.image.is_some()) } } @@ -817,7 +818,7 @@ pub enum PrepareStateError { /// write followed by `fdatasync` before the caller acts on the transition. pub fn prepare_next_state( current: Option, - state: RecoveryState, + state: SessionState, binding: StateBinding, ) -> Result { let (slot, generation) = match current { @@ -858,12 +859,12 @@ pub fn prepare_running_barrier( current: Option, binding: StateBinding, ) -> Result { - let first = prepare_next_state(current, RecoveryState::Running, binding)?; + let first = prepare_next_state(current, SessionState::Running, binding)?; let first_selected = SelectedState { slot: first.slot, record: first.record, }; - let second = prepare_next_state(Some(first_selected), RecoveryState::Running, binding)?; + let second = prepare_next_state(Some(first_selected), SessionState::Running, binding)?; Ok(RunningBarrierWrite { first, second }) } @@ -955,13 +956,13 @@ mod tests { } } - fn record(generation: u64, state: RecoveryState) -> StateRecord { + fn record(generation: u64, state: SessionState) -> StateRecord { StateRecord { generation, state, binding: StateBinding::from_data( data_superblock(), - (state == RecoveryState::Clean).then_some(image()), + (state == SessionState::Clean).then_some(image()), ), } } @@ -981,7 +982,7 @@ mod tests { #[test] fn clean_state_matches_committed_golden_bytes() { - let clean = record(19, RecoveryState::Clean); + let clean = record(19, SessionState::Clean); let clean_golden = assert_golden( &clean.encode().unwrap(), include_str!("format_v1/clean_state.golden"), @@ -1071,9 +1072,9 @@ mod tests { #[test] fn state_records_round_trip_all_states() { for state in [ - RecoveryState::Empty, - RecoveryState::Running, - RecoveryState::Clean, + SessionState::Empty, + SessionState::Running, + SessionState::Clean, ] { let expected = record(19, state); let encoded = expected.encode().unwrap(); @@ -1136,15 +1137,15 @@ mod tests { #[test] fn clean_requires_an_image_binding() { - let mut invalid = record(1, RecoveryState::Running); - invalid.state = RecoveryState::Clean; + let mut invalid = record(1, SessionState::Running); + invalid.state = SessionState::Clean; assert_eq!(invalid.encode(), Err(CodecError::StateRecord)); } #[test] fn latest_valid_ignores_a_torn_or_corrupt_newer_slot() { - let older = record(8, RecoveryState::Clean).encode().unwrap(); - let newer = record(9, RecoveryState::Running).encode().unwrap(); + let older = record(8, SessionState::Clean).encode().unwrap(); + let newer = record(9, SessionState::Running).encode().unwrap(); let mut corrupt = newer; corrupt[72] ^= 1; @@ -1152,14 +1153,14 @@ mod tests { let selected = latest_state([&older, invalid_newer]).unwrap().unwrap(); assert_eq!(selected.slot, 0); assert_eq!(selected.record.generation, 8); - assert_eq!(selected.record.state, RecoveryState::Clean); + assert_eq!(selected.record.state, SessionState::Clean); } } #[test] fn latest_valid_uses_monotonic_generation_not_slot_number() { - let newer = record(42, RecoveryState::Running).encode().unwrap(); - let older = record(41, RecoveryState::Clean).encode().unwrap(); + let newer = record(42, SessionState::Running).encode().unwrap(); + let older = record(41, SessionState::Clean).encode().unwrap(); let selected = latest_state([&newer, &older]).unwrap().unwrap(); assert_eq!(selected.slot, 0); @@ -1168,8 +1169,8 @@ mod tests { #[test] fn equal_generation_with_different_records_is_ambiguous() { - let first = record(5, RecoveryState::Running).encode().unwrap(); - let second = record(5, RecoveryState::Clean).encode().unwrap(); + let first = record(5, SessionState::Running).encode().unwrap(); + let second = record(5, SessionState::Clean).encode().unwrap(); assert_eq!( latest_state([&first, &second]), @@ -1179,8 +1180,8 @@ mod tests { #[test] fn unsupported_state_version_never_falls_back_to_old_clean() { - let old_clean = record(7, RecoveryState::Clean).encode().unwrap(); - let mut unsupported = record(8, RecoveryState::Running).encode().unwrap(); + let old_clean = record(7, SessionState::Clean).encode().unwrap(); + let mut unsupported = record(8, SessionState::Running).encode().unwrap(); put_u16(&mut unsupported, STATE_VERSION_OFFSET, 99); write_page_crc(&mut unsupported); @@ -1216,7 +1217,7 @@ mod tests { RecoveryImageHeaderProbe::Unsupported(99) ); - let mut state = record(8, RecoveryState::Running).encode().unwrap(); + let mut state = record(8, SessionState::Running).encode().unwrap(); put_u16(&mut state, STATE_VERSION_OFFSET, 99); assert_eq!(StateRecord::probe(&state), StateSlotProbe::Corrupt); write_page_crc(&mut state); @@ -1226,7 +1227,7 @@ mod tests { #[test] fn prepare_next_state_alternates_slots_and_increments_generation() { let binding = StateBinding::from_data(data_superblock(), None); - let first = prepare_next_state(None, RecoveryState::Running, binding).unwrap(); + let first = prepare_next_state(None, SessionState::Running, binding).unwrap(); assert_eq!(first.slot, 0); assert_eq!(first.offset(), 0); assert_eq!(first.record.generation, 1); @@ -1235,7 +1236,7 @@ mod tests { slot: first.slot, record: first.record, }; - let second = prepare_next_state(Some(selected), RecoveryState::Empty, binding).unwrap(); + let second = prepare_next_state(Some(selected), SessionState::Empty, binding).unwrap(); assert_eq!(second.slot, 1); assert_eq!(second.offset(), RECOVERY_PAGE_SIZE as u64); assert_eq!(second.record.generation, 2); @@ -1246,22 +1247,18 @@ mod tests { fn prepare_next_state_rejects_an_invalid_selected_slot() { let current = SelectedState { slot: 2, - record: record(10, RecoveryState::Running), + record: record(10, SessionState::Running), }; assert_eq!( - prepare_next_state( - Some(current), - RecoveryState::Running, - current.record.binding - ), + prepare_next_state(Some(current), SessionState::Running, current.record.binding), Err(PrepareStateError::InvalidSlot(2)) ); } #[test] fn running_barrier_replaces_both_slots_before_open() { - let old_clean = record(9, RecoveryState::Clean); - let old_running = record(8, RecoveryState::Running); + let old_clean = record(9, SessionState::Clean); + let old_running = record(8, SessionState::Running); let current = SelectedState { slot: 0, record: old_clean, @@ -1273,19 +1270,19 @@ mod tests { assert_eq!(barrier.first.record.generation, 10); assert_eq!(barrier.second.slot, 0); assert_eq!(barrier.second.record.generation, 11); - assert_eq!(barrier.first.record.state, RecoveryState::Running); - assert_eq!(barrier.second.record.state, RecoveryState::Running); + assert_eq!(barrier.first.record.state, SessionState::Running); + assert_eq!(barrier.second.record.state, SessionState::Running); let mut slot0 = barrier.second.page; let mut slot1 = barrier.first.page; slot0[200] ^= 1; let selected = latest_state([&slot0, &slot1]).unwrap().unwrap(); - assert_eq!(selected.record.state, RecoveryState::Running); + assert_eq!(selected.record.state, SessionState::Running); slot0 = barrier.second.page; slot1[201] ^= 1; let selected = latest_state([&slot0, &slot1]).unwrap().unwrap(); - assert_eq!(selected.record.state, RecoveryState::Running); + assert_eq!(selected.record.state, SessionState::Running); // Before either new page reaches storage, the old CLEAN remains safe: // this session has not yet admitted mutations. @@ -1297,7 +1294,7 @@ mod tests { .unwrap() .record .state, - RecoveryState::Clean + SessionState::Clean ); } @@ -1305,12 +1302,12 @@ mod tests { fn generation_overflow_is_rejected() { let selected = SelectedState { slot: 1, - record: record(u64::MAX, RecoveryState::Running), + record: record(u64::MAX, SessionState::Running), }; assert_eq!( prepare_next_state( Some(selected), - RecoveryState::Running, + SessionState::Running, selected.record.binding ), Err(PrepareStateError::GenerationExhausted) @@ -1321,7 +1318,7 @@ mod tests { fn clean_match_binds_every_identity_and_generation() { let data = data_superblock(); let image = image(); - let clean = record(23, RecoveryState::Clean); + let clean = record(23, SessionState::Clean); assert!(clean.matches_clean(data, image)); let mut wrong_data = data; @@ -1340,7 +1337,7 @@ mod tests { #[test] fn clean_image_match_checks_the_complete_identity_triad() { let data = data_superblock(); - let clean = record(23, RecoveryState::Clean); + let clean = record(23, SessionState::Clean); let header = image_header(); assert!(clean_image_matches( clean, @@ -1367,7 +1364,7 @@ mod tests { TEST_INDEX_LEN )); assert!(!clean_image_matches( - record(24, RecoveryState::Running), + record(24, SessionState::Running), data, header, TEST_IMAGE_FILE_LEN, From aadb377b2a42d981a5e222e29ff02a7946d17575 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:24:12 +0800 Subject: [PATCH 07/12] refactor: align append state and write completion vocabulary Use AppendShardState for each shard's reservations, staged span, and rotation state. Name submitted writes PendingWrite and pair them with WriteCompletion and WriteSubmitError, matching the existing read path. Update fields, locals, submission calls, and fixtures without changing buffer ownership, completion validation, or index publication ordering. --- cache2/src/region/appender.rs | 46 ++++++++++++-------------- cache2/src/region/manager.rs | 61 +++++++++++++++++------------------ cache2/src/region/mod.rs | 12 +++---- cache2/src/region/staging.rs | 1 - 4 files changed, 56 insertions(+), 64 deletions(-) diff --git a/cache2/src/region/appender.rs b/cache2/src/region/appender.rs index e32d8d0..41ce56c 100644 --- a/cache2/src/region/appender.rs +++ b/cache2/src/region/appender.rs @@ -40,16 +40,16 @@ use crate::region::manager::RegionWriteSpan; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; -pub struct RegionSpanSubmitError { +pub struct WriteSubmitError { pub error: io::Error, pub span: RegionWriteSpan, pub buffer: Option, } -impl fmt::Debug for RegionSpanSubmitError { +impl fmt::Debug for WriteSubmitError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { formatter - .debug_struct("RegionSpanSubmitError") + .debug_struct("WriteSubmitError") .field("error", &self.error) .field("span", &self.span) .field("buffer_returned", &self.buffer.is_some()) @@ -57,42 +57,38 @@ impl fmt::Debug for RegionSpanSubmitError { } } -impl fmt::Display for RegionSpanSubmitError { +impl fmt::Display for WriteSubmitError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { self.error.fmt(formatter) } } -impl std::error::Error for RegionSpanSubmitError { +impl std::error::Error for WriteSubmitError { fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { Some(&self.error) } } -pub struct RegionSpanFlight { +pub struct PendingWrite { span: RegionWriteSpan, expected_len: usize, request_id: RequestId, request: BoundedIoRequest, } -pub struct RegionSpanCompletion { +pub struct WriteCompletion { pub span: RegionWriteSpan, pub result: io::Result<()>, pub buffer: Option, } -impl RegionSpanFlight { - pub fn wait( - self, - engine: &IoEngine, - attempt: &mut IoRecoveryAttempt<'_>, - ) -> RegionSpanCompletion { +impl PendingWrite { + pub fn wait(self, engine: &IoEngine, attempt: &mut IoRecoveryAttempt<'_>) -> WriteCompletion { let completion = match self.request.wait_with_io_recovery(engine, attempt) { Ok(completion) => completion, Err(timeout) => { let (error, buffer) = timeout.into_buffer(); - return RegionSpanCompletion { + return WriteCompletion { span: self.span, result: Err(error), buffer, @@ -130,7 +126,7 @@ impl RegionSpanFlight { } Ok(()) }); - RegionSpanCompletion { + WriteCompletion { span: self.span, result, buffer, @@ -141,18 +137,18 @@ impl RegionSpanFlight { // The error returns the owned aligned buffer without another fallible // allocation; boxing it would violate that overload-path property. #[allow(clippy::result_large_err)] -pub fn submit_span( +pub fn submit_write( engine: &IoEngine, geometry: DataGeometry, span: RegionWriteSpan, buffer: IoBuffer, absolute: u64, attempt: &mut IoRecoveryAttempt<'_>, -) -> Result { +) -> Result { let (expected_len, expected_absolute) = match validate_span(geometry, span) { Ok(validated) => validated, Err(error) => { - return Err(RegionSpanSubmitError { + return Err(WriteSubmitError { error, span, buffer: Some(buffer), @@ -160,7 +156,7 @@ pub fn submit_span( } }; if buffer.len() != expected_len || absolute != expected_absolute { - return Err(RegionSpanSubmitError { + return Err(WriteSubmitError { error: io::Error::new( io::ErrorKind::InvalidInput, "staging job does not match its Region span", @@ -174,7 +170,7 @@ pub fn submit_span( && bytes.len() % DIRECT_IO_ALIGNMENT == 0 }); if !buffer_is_direct_aligned { - return Err(RegionSpanSubmitError { + return Err(WriteSubmitError { error: io::Error::new( io::ErrorKind::InvalidInput, "Region span buffer is not direct-I/O aligned", @@ -192,14 +188,14 @@ pub fn submit_span( Ok(request) => request, Err(error) => { let (error, buffer) = error.into_buffer(); - return Err(RegionSpanSubmitError { + return Err(WriteSubmitError { error, span, buffer, }); } }; - Ok(RegionSpanFlight { + Ok(PendingWrite { span, expected_len, request_id: request.id(), @@ -313,7 +309,7 @@ mod tests { let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; let io_recovery = IoRecovery::new(Some(Duration::from_secs(5))); let mut attempt = io_recovery.attempt(); - let completion = submit_span( + let completion = submit_write( &engine, geometry(), span(), @@ -347,7 +343,7 @@ mod tests { let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; let io_recovery = IoRecovery::new(Some(Duration::ZERO)); let mut attempt = io_recovery.attempt(); - let completion = submit_span(&engine, geometry(), span(), buffer, absolute, &mut attempt) + let completion = submit_write(&engine, geometry(), span(), buffer, absolute, &mut attempt) .unwrap() .wait(&engine, &mut attempt); assert!(completion.result.is_ok()); @@ -377,7 +373,7 @@ mod tests { let mut invalid = span(); invalid.end_offset += 1; let buffer = IoBuffer::for_write(BufferLease::try_fixed(4096).unwrap(), 4096).unwrap(); - let error = match submit_span( + let error = match submit_write( &engine, geometry(), invalid, diff --git a/cache2/src/region/manager.rs b/cache2/src/region/manager.rs index b08a112..9f1f566 100644 --- a/cache2/src/region/manager.rs +++ b/cache2/src/region/manager.rs @@ -164,7 +164,7 @@ struct OpenWriteSpan { } #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] -struct ShardMutation { +struct AppendShardState { tail: Option, open_span: Option, pending_padding: Option, @@ -173,7 +173,7 @@ struct ShardMutation { rotation: Option, } -impl ShardMutation { +impl AppendShardState { const fn is_quiescent(self) -> bool { self.tail.is_none() && self.open_span.is_none() @@ -203,7 +203,7 @@ pub struct RegionManager { next_seqno: u64, regions: Vec, active_regions: Vec, - shard_mutations: Vec, + append_states: Vec, free_regions: VecDeque, sealed_regions: VecDeque, reclaiming: Vec, @@ -240,8 +240,8 @@ impl RegionManager { let mut regions = try_vec(region_count)?; let mut active_regions = try_unassigned_vec(active_count)?; - let mut shard_mutations = try_vec(active_count)?; - shard_mutations.resize(active_count, ShardMutation::default()); + let mut append_states = try_vec(active_count)?; + append_states.resize(active_count, AppendShardState::default()); let mut free_regions = try_unassigned_queue(free_count, free_count.max(active_count))?; let mut sealed_regions = try_unassigned_queue(sealed_count, sealed_capacity)?; let reclaiming = try_vec(1)?; @@ -292,7 +292,7 @@ impl RegionManager { next_seqno, regions, active_regions, - shard_mutations, + append_states, free_regions, sealed_regions, reclaiming, @@ -412,7 +412,7 @@ impl RegionManager { return Err(RegionMutationError::InvalidRecordLength); } let shard = self - .shard_mutations + .append_states .get(shard_id) .ok_or(RegionMutationError::InvalidShard)?; if shard.tail.is_some() || shard.pending_padding.is_some() || shard.rotation.is_some() { @@ -449,7 +449,7 @@ impl RegionManager { RegionMutationError::RegionFull }; if error == RegionMutationError::RegionFull { - self.shard_mutations[shard_id].rotation_requested = true; + self.append_states[shard_id].rotation_requested = true; } return Err(error); } @@ -465,7 +465,7 @@ impl RegionManager { seqno, }; self.regions[region_index].reserved_used = end; - self.shard_mutations[shard_id].tail = Some(receipt); + self.append_states[shard_id].tail = Some(receipt); Ok(receipt) } @@ -481,7 +481,7 @@ impl RegionManager { receipt: RegionAppendReservation, ) -> Result<(), RegionMutationError> { let shard = self - .shard_mutations + .append_states .get(receipt.shard_id) .ok_or(RegionMutationError::InvalidShard)?; if shard.tail != Some(receipt) { @@ -519,8 +519,8 @@ impl RegionManager { )); } }; - self.shard_mutations[receipt.shard_id].open_span = Some(next_span); - self.shard_mutations[receipt.shard_id].tail = None; + self.append_states[receipt.shard_id].open_span = Some(next_span); + self.append_states[receipt.shard_id].tail = None; Ok(()) } @@ -533,7 +533,7 @@ impl RegionManager { shard_id: usize, ) -> Result, RegionMutationError> { let shard = self - .shard_mutations + .append_states .get(shard_id) .copied() .ok_or(RegionMutationError::InvalidShard)?; @@ -604,7 +604,7 @@ impl RegionManager { max_seqno: open.max_seqno, }; self.regions[region_index].reserved_used = padded_end_offset; - self.shard_mutations[shard_id].pending_padding = Some(receipt); + self.append_states[shard_id].pending_padding = Some(receipt); Ok(Some(receipt)) } @@ -617,7 +617,7 @@ impl RegionManager { receipt: RegionAppendReservation, ) -> Result<(), RegionMutationError> { let shard = self - .shard_mutations + .append_states .get(receipt.shard_id) .ok_or(RegionMutationError::InvalidShard)?; if shard.tail != Some(receipt) { @@ -640,7 +640,7 @@ impl RegionManager { return Err(RegionMutationError::StaleReceipt); } self.regions[region_index].reserved_used = u64::from(receipt.offset); - self.shard_mutations[receipt.shard_id].tail = None; + self.append_states[receipt.shard_id].tail = None; Ok(()) } @@ -652,7 +652,7 @@ impl RegionManager { shard_id: usize, ) -> Result { let shard = self - .shard_mutations + .append_states .get(shard_id) .ok_or(RegionMutationError::InvalidShard)?; if shard.tail.is_some() @@ -676,7 +676,7 @@ impl RegionManager { padding: RegionPaddingReceipt, ) -> Result { let shard = self - .shard_mutations + .append_states .get(padding.shard_id) .copied() .ok_or(RegionMutationError::InvalidShard)?; @@ -724,7 +724,7 @@ impl RegionManager { end_offset: u64, ) -> Result { let shard = self - .shard_mutations + .append_states .get(shard_id) .copied() .ok_or(RegionMutationError::InvalidShard)?; @@ -747,7 +747,7 @@ impl RegionManager { record_count: open.record_count, max_seqno: open.max_seqno, }; - let shard = &mut self.shard_mutations[shard_id]; + let shard = &mut self.append_states[shard_id]; shard.open_span = None; shard.pending_padding = None; shard.submitted_span = Some(receipt); @@ -762,7 +762,7 @@ impl RegionManager { receipt: RegionWriteSpan, ) -> Result<(), RegionMutationError> { let shard = self - .shard_mutations + .append_states .get(receipt.shard_id) .ok_or(RegionMutationError::InvalidShard)?; if shard.submitted_span != Some(receipt) { @@ -792,7 +792,7 @@ impl RegionManager { let region = &mut self.regions[region_index]; region.completed_used = receipt.end_offset; region.physical_record_count = physical_record_count; - self.shard_mutations[receipt.shard_id].submitted_span = None; + self.append_states[receipt.shard_id].submitted_span = None; Ok(()) } @@ -813,7 +813,7 @@ impl RegionManager { shard_id: usize, ) -> Result<(), RegionMutationError> { let shard = self - .shard_mutations + .append_states .get_mut(shard_id) .ok_or(RegionMutationError::InvalidShard)?; shard.rotation_requested = true; @@ -866,8 +866,8 @@ impl RegionManager { activated_region_id: victim_region_id, activated_created_seqno: created_seqno, }; - self.shard_mutations[shard_id].rotation_requested = false; - self.shard_mutations[shard_id].rotation = Some(receipt); + self.append_states[shard_id].rotation_requested = false; + self.append_states[shard_id].rotation = Some(receipt); Ok(receipt) } @@ -876,7 +876,7 @@ impl RegionManager { shard_id: usize, ) -> Result { let shard = self - .shard_mutations + .append_states .get(shard_id) .ok_or(RegionMutationError::InvalidShard)?; if !shard.rotation_requested @@ -943,7 +943,7 @@ impl RegionManager { receipt: RegionRotationReceipt, ) -> Result<(), RegionMutationError> { let shard = self - .shard_mutations + .append_states .get(receipt.shard_id) .ok_or(RegionMutationError::InvalidShard)?; if shard.rotation != Some(receipt) { @@ -980,7 +980,7 @@ impl RegionManager { } self.sealed_regions.push_back(receipt.sealed_region_id); self.rotations = self.rotations.saturating_add(1); - self.shard_mutations[receipt.shard_id].rotation = None; + self.append_states[receipt.shard_id].rotation = None; Ok(()) } @@ -1078,10 +1078,7 @@ impl RegionManager { partitions: Box<[PartitionMetadataRecord]>, ) -> Result { if !self.reclaiming.is_empty() - || self - .shard_mutations - .iter() - .any(|shard| !shard.is_quiescent()) + || self.append_states.iter().any(|shard| !shard.is_quiescent()) { return Err(RegionMetadataError::InvalidField("live_region_authority")); } diff --git a/cache2/src/region/mod.rs b/cache2/src/region/mod.rs index 1497f0c..1f8bff0 100644 --- a/cache2/src/region/mod.rs +++ b/cache2/src/region/mod.rs @@ -25,7 +25,7 @@ use std::sync::atomic::AtomicU8; use std::sync::atomic::AtomicU64; use std::sync::atomic::Ordering; -use self::appender::submit_span; +use self::appender::submit_write; use self::index::ReclaimIndexAction; use self::index::RegionIndex; use self::index::heat_memory_bytes; @@ -59,7 +59,7 @@ use crate::io::engine::IoEngine; use crate::io::engine::ReadSlot; use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::managed_memory::BufferLease; -use crate::region::appender::RegionSpanCompletion; +use crate::region::appender::WriteCompletion; #[cfg(test)] use crate::region::index::packed::IndexEntry; use crate::region::index::packed::PackedLocation; @@ -1126,16 +1126,16 @@ impl RegionStore { absolute, records, } = job; - let flight = match submit_span(engine, geometry, span, buffer, absolute, &mut attempt) { - Ok(flight) => flight, + let pending = match submit_write(engine, geometry, span, buffer, absolute, &mut attempt) { + Ok(pending) => pending, Err(error) => { let original = error.error; self.fail_staged_span(staging, span, error.buffer, records); return Err(original); } }; - let completion = flight.wait(engine, &mut attempt); - let RegionSpanCompletion { + let completion = pending.wait(engine, &mut attempt); + let WriteCompletion { span, result, buffer, diff --git a/cache2/src/region/staging.rs b/cache2/src/region/staging.rs index 9057531..bc229e6 100644 --- a/cache2/src/region/staging.rs +++ b/cache2/src/region/staging.rs @@ -62,7 +62,6 @@ pub struct ShardFillSnapshot { /// containing device span completes. The descriptor stays owned by the /// completion path; staging never calls into the index while holding a shard /// lock. -/// Compact transient completion descriptor. #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub struct StagedRecord { hash: u64, From dcdbb81404cf4e8429b15702b701e428e575661a Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:25:37 +0800 Subject: [PATCH 08/12] refactor: finish I/O module boundary cleanup Remove the duplicate Unix import guard left by folding the positioned-I/O wrapper into DataFileHandles and group the shared recovery import with the other I/O capabilities. Retain the kernel's registered-files wording in the SQPOLL diagnostic. --- cache2/src/io/engine/mod.rs | 3 +-- cache2/src/io/engine/posix.rs | 1 - cache2/src/io/engine/uring.rs | 2 +- 3 files changed, 2 insertions(+), 4 deletions(-) diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index 50b01eb..672648d 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -77,13 +77,12 @@ use crate::io::file::FileIoPath; use crate::io::file::FileIoStats; use crate::io::file::FileIoStatsHandle; use crate::io::file::WritePoint; +use crate::io::recovery::IoRecoveryAttempt; use crate::managed_memory::BufferLease; use crate::snapshot::CacheIoDirectionSnapshot; mod posix; -use crate::io::recovery::IoRecoveryAttempt; - #[cfg(all( feature = "io-uring", target_os = "linux", diff --git a/cache2/src/io/engine/posix.rs b/cache2/src/io/engine/posix.rs index afa109e..9b7339f 100644 --- a/cache2/src/io/engine/posix.rs +++ b/cache2/src/io/engine/posix.rs @@ -27,7 +27,6 @@ use crate::io::engine::IoEngine; use crate::io::engine::IoOperation; use crate::io::engine::lock_unpoisoned; #[cfg(unix)] -#[cfg(unix)] use crate::io::file::DataFileHandles; use crate::io::file::FileIoStatsHandle; use crate::io::file::PositionedIo; diff --git a/cache2/src/io/engine/uring.rs b/cache2/src/io/engine/uring.rs index 8364356..b037fcb 100644 --- a/cache2/src/io/engine/uring.rs +++ b/cache2/src/io/engine/uring.rs @@ -153,7 +153,7 @@ pub fn start( if config.sq_poll.is_some() && !ring.params().is_feature_sqpoll_nonfixed() { return Err(io::Error::new( io::ErrorKind::Unsupported, - "kernel io_uring SQPOLL requires registered handles", + "kernel io_uring SQPOLL requires registered files", )); } let mut probe = Probe::new(); From 91efac52278e0edd75af3cbc8d9b8ca8b25cb771 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 00:26:29 +0800 Subject: [PATCH 09/12] docs: define cache execution and region ownership roles Document the session/runtime/store/persistence boundaries and distinguish Regions, append shards, and index partitions. Explain the remaining I/O capabilities, resource ownership, and separate restart and timeout states. Update the close-path description and use Desc in the contribution naming conventions so future changes follow the same domain vocabulary. --- ARCHITECTURE.md | 26 ++++++++++++++++++++++++-- CONTRIBUTING.md | 2 +- 2 files changed, 25 insertions(+), 3 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 910c28b..fe2c063 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -28,6 +28,28 @@ Cache The complete raw key is hashed once with seeded XXH3-64 at the public operation boundary. L1 compares the full key, and an L2 hit validates the key stored in the record. Returned values always match the complete key. +## Implementation roles + +The private implementation separates cache execution, live L2 state, and persistence. A Region is a fixed extent in the data file, an append shard is a write-admission and batching lane, and an index partition is a lookup and locking boundary. These are independent divisions; none is a storage backend or a separate cache instance. + +| Type | Responsibility | +| ------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------- | +| `CacheSession` | Owns one open/close lifecycle, ordering recovery, durable RUNNING, runtime startup, shutdown, and CLEAN publication. | +| `CacheRuntime` | Executes public requests across L1 and L2; owns bounded I/O engines, append staging, and append/reclaim workers. | +| `RegionStore` | Holds the live L2 index and Region manager, append gates, generation checks, and terminal health latch. It does not own cache files or start workers. | +| `RegionManager` | Allocates append positions, rotates Active Regions, maintains the Free/Sealed queues, and tracks outstanding append/reclaim work. | +| `RegionDesc` | Describes one Region's state, generation, reserved/completed prefix, and physical record count. | +| `AppendStaging` | Owns the fixed per-shard buffers and the records awaiting index publication. | +| `ReadDesc` | Describes the exact candidate and byte range for one bounded L2 read. | +| `PendingRead` / `PendingWrite` | Retain the submitted operation and its buffer until I/O completes; completion validation precedes returning a value or publishing an index entry. | +| `RegionPersistence` | Owns file locks, validates restart artifacts, constructs cold or recovered Region state, and durably publishes recovery images. | + +`cache/session.rs` contains the lifecycle ordering, `cache/runtime/` contains request execution and workers, and `region/` contains L2 state and operations. `region/persistence/` coordinates files; `region/recovery/` defines the persistent formats and their validation. The session passes a quiescent `FrozenRegionStore` to persistence, which produces a `PreparedClean` only after data and image durability; publishing CLEAN is a separate final step. + +The I/O layer names capabilities and resources directly. `PositionedIo` is synchronous positioned read/write access; `StorageFile` adds file identity, extent management, locking, sync, and handle duplication. `CacheFile` is the OS file resource, while `DataFileHandles` holds its buffered/direct descriptors for engine use. `FileSystem` covers path operations such as create, rename, and parent-directory sync; `OsFileSystem` supplies those operations. These narrow interfaces also support fault injection. Region lifecycle and storage have one concrete implementation and no backend trait. + +Three states serve different purposes: `RegionState` is Free/Active/Sealed, persisted `SessionState` is EMPTY/RUNNING/CLEAN, and `IoRecovery` tracks reversible background I/O timeout recovery. The Region health latch is a separate terminal failure boundary. Recovery of an I/O request cannot restore an unhealthy Region store or make a restart image eligible. + ## Consistency model Each accepted mutation receives a logical sequence number. L1 uses that number to avoid replacing a newer exact-key resident value with an older publication or promotion, and a sequenced delete does not remove a newer L1 value. Reclaim preserves the original logical sequence when it rewrites a live record. @@ -162,7 +184,7 @@ The index image is mapped writable and private. Runtime mutations therefore use ### Warm close publication -The public handle separates a shared data-plane view from the cache-internal lifecycle owner. `close_warm(&self)` therefore works while other `Arc` handles remain alive. One acquire-load rejects calls after close starts. L1 hits and all reads pay no operation-gate read-modify-write. A permanent mutation fence and the write engines provide the quiescence required by recovery. Read I/O owns no persistent state and cannot pin the snapshot. Calls admitted before close are in flight and may finish; queued but unsubmitted L2 work also cannot pin close. Retained public handles keep only bounded in-memory data-plane resources alive until drop. +The public handle shares request execution through `CacheRuntime`, while `CacheSession` owns the open/close lifecycle. `close_warm(&self)` therefore works while other `Arc` handles remain alive. One acquire-load rejects calls after close starts. L1 hits and all reads pay no operation-gate read-modify-write. A permanent mutation fence and the write engines provide the quiescence required by recovery. Read I/O owns no persistent state and cannot pin the snapshot. Calls admitted before close are in flight and may finish; queued but unsubmitted L2 work also cannot pin close. Retained public handles keep only bounded in-memory runtime resources alive until drop. `close_warm` establishes one recoverable snapshot in this order: @@ -189,6 +211,6 @@ Stale data, overload, and cache loss are valid outcomes. Every returned value pa ### Background I/O timeout recovery -A background write or reclaim admission/completion timeout enters a reversible recovery state separate from the terminal Region health latch. `RuntimeOptions::io_recovery_timeout` defaults to `None` (until completion or close); `Some(duration)` bounds the additional wait and `Some(Duration::ZERO)` disables recovery. Each affected worker holds a recovery attempt through validation and publication. A shared pending count fences new foreground fills until all attempts succeed; reads and deletes remain independent, and previously accepted work may finish. The snapshot reports `CacheHealth::Recovering` unless a draining or terminal state takes precedence. Workers use fixed one-second admission/completion checks anchored to the original deadline for finite budgets. Only unsubmitted operations returned by admission can be retried. Issued requests retain their worker, slot, buffer, staging span, and Region until completion. No new queue or worker is introduced, and memory remains bounded regardless of recovery duration. +A background write or reclaim admission/completion timeout enters a reversible recovery state separate from the terminal Region health latch. `RuntimeOptions::io_recovery_timeout` defaults to `None` (until completion or close); `Some(duration)` bounds the additional wait and `Some(Duration::ZERO)` disables recovery. Each affected worker holds an `IoRecoveryAttempt` through validation and publication. A shared pending count fences new foreground fills until all attempts succeed; reads and deletes remain independent, and previously accepted work may finish. The snapshot reports `CacheHealth::Recovering` unless a draining or terminal state takes precedence. Workers use fixed one-second admission/completion checks anchored to the original deadline for finite budgets. Only unsubmitted operations returned by admission can be retried. Issued requests retain their worker, slot, buffer, staging span, and Region until completion. No new queue or worker is introduced, and memory remains bounded regardless of recovery duration. `cache_io_recovery_started` and `cache_io_recovery_completed` log transitions into recovery and out of it after all affected operations pass validation and publication. Failed attempts never lower the admission fence. Actual I/O errors, invalid completions, and finite-budget exhaustion retain the terminal failure policy; this implementation does not retry completed I/O failures or reopen a failed instance. Closing or a background worker failure stops recovery, observed within one polling interval; normal cancellation then preserves the unfenced-write boundary and prevents unsafe CLEAN publication. Close starts this interruption before waiting for mutation quiescence, including reclaim reinsertion. Drain remains a completion fence and may wait indefinitely in unlimited recovery mode. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index b007432..7643719 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -31,7 +31,7 @@ Cover observable behavior changes with tests. See [BENCHMARK.md](BENCHMARK.md) f Follow the surrounding code and the design constraints in [ARCHITECTURE.md](ARCHITECTURE.md), including bounded resource use, best-effort consistency, request-path priorities, and recovery guarantees. -Name types by their concrete domain role. Use `Options` for editable inputs awaiting validation and `Config` for validated or resolved configuration. Keep names such as `Layout` for geometry and use specific domain nouns such as `Descriptor`, `Candidate`, `Victims`, or `Inspection` for runtime data. Avoid `Plan` as a type-name suffix; computing data before using it does not by itself make that data a plan. Keep related fields, local variables, and functions consistent with the type's role. +Name types by their concrete domain role. Use `Options` for editable inputs awaiting validation and `Config` for validated or resolved configuration. Keep names such as `Layout` for geometry and use specific domain nouns such as `Desc`, `Candidate`, `Victims`, or `Inspection` for runtime data. Avoid `Plan` as a type-name suffix; computing data before using it does not by itself make that data a plan. Keep related fields, local variables, and functions consistent with the type's role. Declare restricted visibility at module boundaries and use `pub` for items in those modules' APIs. Keep items private when only their defining module and its descendants need them. For items reachable through public modules or re-exported public types, reserve `pub` for intentional public API and use narrower visibility for internal callers. From 8e42da8b4cdbe3698c2b4368e87ac3c846153687 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 08:37:30 +0800 Subject: [PATCH 10/12] refactor: name fill budgets by records and byte units Distinguish record credits from completed I/O operations throughout fill control. Name budget acquisition and refund explicitly, and distinguish foreground fill admission from background flush pacing. Use limits for retained FillLimits and max_flush_bytes for the burst floor supplied by Region geometry. Preserve the packed-credit representation, refill arithmetic, oversized-flush handling, and public configuration. --- cache2/src/cache/runtime/mod.rs | 13 +- cache2/src/io/engine/tests.rs | 6 +- cache2/src/io/fill_control.rs | 295 ++++++++++++++++++-------------- 3 files changed, 177 insertions(+), 137 deletions(-) diff --git a/cache2/src/cache/runtime/mod.rs b/cache2/src/cache/runtime/mod.rs index b72d62d..8097b03 100644 --- a/cache2/src/cache/runtime/mod.rs +++ b/cache2/src/cache/runtime/mod.rs @@ -848,7 +848,7 @@ impl CacheRuntime { } }; if let Some(fill) = &state.io_recovery.fill - && !fill.try_admit() + && !fill.try_admit_fill() { if state.activity_counters { state.metrics.record_write_rejection(); @@ -1922,8 +1922,13 @@ fn append_worker_result( let bytes = fill.bytes as u64; let records = u32::try_from(fill.records).unwrap_or(u32::MAX); let charge = match &state.io_recovery.fill { - None => Some(FlushCharge { ops: 0, units: 0 }), - Some(control) => control.try_flush(bytes, records, essential), + None => Some(FlushCharge { + records: 0, + byte_units: 0, + }), + Some(control) => { + control.try_acquire_flush_budget(bytes, records, essential) + } }; if let Some(charge) = charge { let engine = state.write_engine_for(shard_id as u64); @@ -1936,7 +1941,7 @@ fn append_worker_result( Some(_) => deadline = None, None => { if let Some(control) = &state.io_recovery.fill { - control.refund_flush(charge); + control.refund_flush_budget(charge); } deadline = Some(Instant::now() + STAGING_RETRY_DELAY); } diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index d7d5e2d..8916ad6 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -1375,7 +1375,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp } let paused = control.snapshot(); let recovered = io_recovery.is_recovering(); - let admission = control.try_admit(); + let admission = control.try_admit_fill(); io.release(); let completion = returned_rx.recv_timeout(Duration::from_secs(2)).unwrap(); assert_eq!(paused.pressure, FillPressure::Paused); @@ -1384,7 +1384,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp assert_eq!(paused.outstanding_bytes, 4096); assert!(completion.into_io_result().0.is_ok()); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); validate_tx.send(()).unwrap(); }); let deadline = Instant::now() + Duration::from_secs(2); @@ -1392,7 +1392,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp std::thread::sleep(Duration::from_millis(10)); } assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); assert_eq!(lock_unpoisoned(&io.state).entered, 1); engine.shutdown().unwrap(); } diff --git a/cache2/src/io/fill_control.rs b/cache2/src/io/fill_control.rs index dde7541..fa537da 100644 --- a/cache2/src/io/fill_control.rs +++ b/cache2/src/io/fill_control.rs @@ -29,7 +29,7 @@ use crate::FillControlSnapshot; use crate::FillLimits; use crate::FillPressure; -const UNIT: u64 = 64; +const CREDIT_UNIT_BYTES: u64 = 64; const MAX_CAS_ATTEMPTS: usize = 4; const MIN_BYTES_PER_SECOND: u64 = 640; const MAX_BYTES_PER_SECOND: u64 = 1 << 40; @@ -44,18 +44,18 @@ pub const FLUSH_RETRY: Duration = Duration::from_millis(100); /// Credit taken by one non-essential flush. Zero for Observe and essential work. #[derive(Clone, Copy)] pub struct FlushCharge { - pub(crate) ops: u64, - pub(crate) units: u64, + pub records: u64, + pub byte_units: u64, } -fn packed_ops(value: u64) -> u64 { +fn packed_records(value: u64) -> u64 { value >> 32 } -fn packed_units(value: u64) -> u64 { +fn packed_byte_units(value: u64) -> u64 { value & u64::from(u32::MAX) } -fn pack_credit(ops: u64, units: u64) -> u64 { - (ops << 32) | units +fn pack_credit(records: u64, byte_units: u64) -> u64 { + (records << 32) | byte_units } fn encode_pressure(pressure: FillPressure) -> u8 { @@ -84,10 +84,10 @@ struct Slot { } pub struct FillController { - options: FillLimits, + limits: FillLimits, enforcing: bool, - max_units: u64, - max_ops: u64, + max_byte_units: u64, + max_records: u64, origin: Instant, credit: AtomicU64, last_refill_ns: AtomicU64, @@ -99,21 +99,19 @@ pub struct FillController { would_reject: AtomicU64, dropped_observations: AtomicU64, completed_bytes: AtomicU64, - completed_ops: AtomicU64, + completed_io_operations: AtomicU64, slots: Box<[Slot]>, } impl FillController { pub fn allocation_bytes(options: FillControlOptions, slots: usize) -> io::Result { - let settings = match options { + let limits = match options { FillControlOptions::Disabled => return Ok(0), - FillControlOptions::Observe(settings) | FillControlOptions::Adaptive(settings) => { - settings - } + FillControlOptions::Observe(limits) | FillControlOptions::Adaptive(limits) => limits, }; - if !(MIN_BYTES_PER_SECOND..=MAX_BYTES_PER_SECOND).contains(&settings.max_bytes_per_second) + if !(MIN_BYTES_PER_SECOND..=MAX_BYTES_PER_SECOND).contains(&limits.max_bytes_per_second) || !(MIN_RECORDS_PER_SECOND..=MAX_RECORDS_PER_SECOND) - .contains(&settings.max_records_per_second) + .contains(&limits.max_records_per_second) { return Err(io::Error::new( io::ErrorKind::InvalidInput, @@ -134,24 +132,24 @@ impl FillController { pub fn new( options: FillControlOptions, slots: usize, - max_record: u64, + max_flush_bytes: u64, ) -> io::Result>> { Self::allocation_bytes(options, slots)?; - let (settings, enforcing) = match options { + let (limits, enforcing) = match options { FillControlOptions::Disabled => return Ok(None), - FillControlOptions::Observe(settings) => (settings, false), - FillControlOptions::Adaptive(settings) => (settings, true), + FillControlOptions::Observe(limits) => (limits, false), + FillControlOptions::Adaptive(limits) => (limits, true), }; - let max_units = (settings.max_bytes_per_second / TICKS_PER_SECOND) - .max(max_record) - .div_ceil(UNIT); - if max_units > u64::from(u32::MAX) { + let max_byte_units = (limits.max_bytes_per_second / TICKS_PER_SECOND) + .max(max_flush_bytes) + .div_ceil(CREDIT_UNIT_BYTES); + if max_byte_units > u64::from(u32::MAX) { return Err(io::Error::new( io::ErrorKind::InvalidInput, "fill burst exceeds budget representation", )); } - let max_ops = u64::from(settings.max_records_per_second / TICKS_PER_SECOND as u32).max(1); + let max_records = u64::from(limits.max_records_per_second / TICKS_PER_SECOND as u32).max(1); let mut table = Vec::new(); table.try_reserve_exact(slots).map_err(|_| { io::Error::new( @@ -164,12 +162,12 @@ impl FillController { bytes: AtomicU64::new(0), }); Ok(Some(Arc::new(Self { - options: settings, + limits, enforcing, - max_units, - max_ops, + max_byte_units, + max_records, origin: Instant::now(), - credit: AtomicU64::new(pack_credit(max_ops, max_units)), + credit: AtomicU64::new(pack_credit(max_records, max_byte_units)), last_refill_ns: AtomicU64::new(0), pressure: AtomicU8::new(encode_pressure(FillPressure::Healthy)), stopped: AtomicBool::new(false), @@ -179,7 +177,7 @@ impl FillController { would_reject: AtomicU64::new(0), dropped_observations: AtomicU64::new(0), completed_bytes: AtomicU64::new(0), - completed_ops: AtomicU64::new(0), + completed_io_operations: AtomicU64::new(0), slots: table.into_boxed_slice(), }))) } @@ -284,8 +282,8 @@ impl FillController { (0, 0) } else { ( - self.options.max_bytes_per_second, - self.options.max_records_per_second, + self.limits.max_bytes_per_second, + self.limits.max_records_per_second, ) }; FillControlSnapshot { @@ -304,7 +302,7 @@ impl FillController { } /// Foreground admission: Adaptive rejects only while paused. - pub fn try_admit(&self) -> bool { + pub fn try_admit_fill(&self) -> bool { if !self.enforcing { if self.paused() { self.would_reject.fetch_add(1, Ordering::Relaxed); @@ -320,41 +318,60 @@ impl FillController { } /// Background flush pacing. Essential flushes always proceed. - pub fn try_flush(&self, bytes: u64, records: u32, essential: bool) -> Option { + pub fn try_acquire_flush_budget( + &self, + bytes: u64, + records: u32, + essential: bool, + ) -> Option { if essential || !self.enforcing { - return Some(FlushCharge { ops: 0, units: 0 }); + return Some(FlushCharge { + records: 0, + byte_units: 0, + }); } self.refill(); - let raw_units = bytes.div_ceil(UNIT).max(1); - let raw_ops = u64::from(records.max(1)); - let units = raw_units.min(self.max_units); - let ops = raw_ops.min(self.max_ops); - let byte_oversized = raw_units >= self.max_units; - let record_oversized = raw_ops >= self.max_ops; + let raw_byte_units = bytes.div_ceil(CREDIT_UNIT_BYTES).max(1); + let raw_records = u64::from(records.max(1)); + let byte_units = raw_byte_units.min(self.max_byte_units); + let records = raw_records.min(self.max_records); + let byte_oversized = raw_byte_units >= self.max_byte_units; + let record_oversized = raw_records >= self.max_records; let mut value = self.credit.load(Ordering::Relaxed); for _ in 0..MAX_CAS_ATTEMPTS { - let have_units = packed_units(value); - let have_ops = packed_ops(value); - if have_units == 0 || have_ops == 0 { + let have_byte_units = packed_byte_units(value); + let have_records = packed_records(value); + if have_byte_units == 0 || have_records == 0 { return None; } - if !byte_oversized && have_units < units { + if !byte_oversized && have_byte_units < byte_units { return None; } - if !record_oversized && have_ops < ops { + if !record_oversized && have_records < records { return None; } - let take_units = if byte_oversized { have_units } else { units }; - let take_ops = if record_oversized { have_ops } else { ops }; - let next = pack_credit(have_ops - take_ops, have_units - take_units); + let take_byte_units = if byte_oversized { + have_byte_units + } else { + byte_units + }; + let take_records = if record_oversized { + have_records + } else { + records + }; + let next = pack_credit( + have_records - take_records, + have_byte_units - take_byte_units, + ); match self .credit .compare_exchange(value, next, Ordering::AcqRel, Ordering::Relaxed) { Ok(_) => { return Some(FlushCharge { - ops: take_ops, - units: take_units, + records: take_records, + byte_units: take_byte_units, }); } Err(current) => value = current, @@ -363,11 +380,11 @@ impl FillController { None } - pub fn refund_flush(&self, charge: FlushCharge) { - if !self.enforcing || (charge.ops == 0 && charge.units == 0) { + pub fn refund_flush_budget(&self, charge: FlushCharge) { + if !self.enforcing || (charge.records == 0 && charge.byte_units == 0) { return; } - self.add_credit(charge.ops, charge.units); + self.add_credit(charge.records, charge.byte_units); } fn refill(&self) { @@ -376,18 +393,18 @@ impl FillController { let Some(dt) = now.checked_sub(last).filter(|dt| *dt != 0) else { return; }; - let add_units = self - .options + let add_byte_units = self + .limits .max_bytes_per_second .saturating_mul(dt) .checked_div(1_000_000_000) .unwrap_or(0) - / UNIT; - let add_ops = u64::from(self.options.max_records_per_second) + / CREDIT_UNIT_BYTES; + let add_records = u64::from(self.limits.max_records_per_second) .saturating_mul(dt) .checked_div(1_000_000_000) .unwrap_or(0); - if add_units == 0 && add_ops == 0 { + if add_byte_units == 0 && add_records == 0 { return; } if self @@ -397,18 +414,18 @@ impl FillController { { return; } - self.add_credit(add_ops, add_units); + self.add_credit(add_records, add_byte_units); } - fn add_credit(&self, ops: u64, units: u64) { - if ops == 0 && units == 0 { + fn add_credit(&self, records: u64, byte_units: u64) { + if records == 0 && byte_units == 0 { return; } let mut value = self.credit.load(Ordering::Relaxed); loop { let next = pack_credit( - (packed_ops(value) + ops).min(self.max_ops), - (packed_units(value) + units).min(self.max_units), + (packed_records(value) + records).min(self.max_records), + (packed_byte_units(value) + byte_units).min(self.max_byte_units), ); match self .credit @@ -476,7 +493,9 @@ impl Drop for Observation<'_> { let bytes = slot.bytes.load(Ordering::Relaxed); slot.start_ns.store(0, Ordering::Release); if self.succeeded { - self.control.completed_ops.fetch_add(1, Ordering::Relaxed); + self.control + .completed_io_operations + .fetch_add(1, Ordering::Relaxed); self.control .completed_bytes .fetch_add(bytes, Ordering::Relaxed); @@ -492,12 +511,12 @@ mod tests { use super::*; fn control(enforcing: bool) -> Arc { - let settings = FillLimits::new(64_000, 100); + let limits = FillLimits::new(64_000, 100); FillController::new( if enforcing { - FillControlOptions::Adaptive(settings) + FillControlOptions::Adaptive(limits) } else { - FillControlOptions::Observe(settings) + FillControlOptions::Observe(limits) }, 4, 4096, @@ -508,7 +527,7 @@ mod tests { fn restore_burst(control: &FillController) { control.credit.store( - pack_credit(control.max_ops, control.max_units), + pack_credit(control.max_records, control.max_byte_units), Ordering::Release, ); } @@ -523,32 +542,32 @@ mod tests { fn foreground_admit_ignores_flush_budget() { let control = control(true); for _ in 0..10 { - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); } - assert!(control.try_flush(64, 1, false).is_none()); - assert!(control.try_admit()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_none()); + assert!(control.try_admit_fill()); } #[test] fn record_budget_paces_nonessential_flush() { let control = control(true); for _ in 0..10 { - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); } - assert!(control.try_flush(64, 1, false).is_none()); - assert!(control.try_flush(64, 1, true).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_none()); + assert!(control.try_acquire_flush_budget(64, 1, true).is_some()); restore_burst(&control); - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); } #[test] fn byte_budget_paces_nonessential_flush() { let control = control(true); - assert!(control.try_flush(4096, 1, false).is_some()); - assert!(control.try_flush(4096, 1, false).is_none()); - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(4096, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(4096, 1, false).is_none()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); restore_burst(&control); - assert!(control.try_flush(4096, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(4096, 1, false).is_some()); } #[test] @@ -559,10 +578,10 @@ mod tests { fast.finish(); old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); drop(old); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); } #[test] @@ -572,7 +591,7 @@ mod tests { old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); for _ in 0..20 { - assert!(control.try_admit()); + assert!(control.try_admit_fill()); } assert_eq!(control.snapshot().would_reject, 20); assert_eq!(control.snapshot().rejections, 0); @@ -583,14 +602,14 @@ mod tests { fn observe_would_reject_counts_pause_not_budget() { let control = control(false); for _ in 0..20 { - assert!(control.try_flush(4096, 1, false).is_some()); - assert!(control.try_admit()); + assert!(control.try_acquire_flush_budget(4096, 1, false).is_some()); + assert!(control.try_admit_fill()); } assert_eq!(control.snapshot().would_reject, 0); let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); assert_eq!(control.snapshot().would_reject, 1); assert_eq!(control.snapshot().rejections, 0); } @@ -601,8 +620,8 @@ mod tests { control.last_refill_ns.store(0, Ordering::Relaxed); control.refill(); let credit = control.credit.load(Ordering::Relaxed); - assert_eq!(packed_ops(credit), control.max_ops); - assert_eq!(packed_units(credit), control.max_units); + assert_eq!(packed_records(credit), control.max_records); + assert_eq!(packed_byte_units(credit), control.max_byte_units); } #[test] @@ -614,11 +633,11 @@ mod tests { ) .unwrap() .unwrap(); - assert!(control.try_flush(4096, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(4096, 1, false).is_some()); restore_burst(&control); - assert!(control.try_flush(4096, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(4096, 1, false).is_some()); control.stop(); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); } #[test] @@ -626,10 +645,10 @@ mod tests { let control = control(true); control.set_recovering(true); assert_eq!(control.snapshot().pressure, FillPressure::Paused); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); control.set_recovering(false); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); } #[test] @@ -637,13 +656,13 @@ mod tests { let control = control(true); control.set_recovering(true); control.set_recovering(true); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); control.set_recovering(false); assert_eq!(control.snapshot().pressure, FillPressure::Paused); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); control.set_recovering(false); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); } #[test] @@ -661,7 +680,7 @@ mod tests { drop(pending); let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); drop(pending); - assert_eq!(control.completed_ops.load(Ordering::Relaxed), 0); + assert_eq!(control.completed_io_operations.load(Ordering::Relaxed), 0); assert_eq!(control.completed_bytes.load(Ordering::Relaxed), 0); assert_eq!(control.slots[0].start_ns.load(Ordering::Acquire), 0); } @@ -677,7 +696,7 @@ mod tests { let accepted = &accepted; scope.spawn(move || { for _ in 0..100 { - if control.try_flush(640, 1, false).is_some() { + if control.try_acquire_flush_budget(640, 1, false).is_some() { accepted.fetch_add(1, Ordering::Relaxed); } } @@ -692,10 +711,10 @@ mod tests { let control = control(true); let mut pending = control.observe(4096, Duration::from_secs(2)).unwrap(); pending.note_slow(); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); pending.clear_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); pending.finish(); } @@ -708,14 +727,14 @@ mod tests { scope.spawn(move || { let mut pending = control.observe(64, Duration::from_secs(2)).unwrap(); pending.note_slow(); - assert!(!control.try_admit()); + assert!(!control.try_admit_fill()); drop(pending); }); } }); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); assert_eq!(control.pause_holders.load(Ordering::Acquire), 0); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); } #[test] @@ -730,7 +749,7 @@ mod tests { drop(second); assert_eq!(control.pause_holders.load(Ordering::Acquire), 0); assert_eq!(control.snapshot().pressure, FillPressure::Healthy); - assert!(control.try_admit()); + assert!(control.try_admit_fill()); } } @@ -738,25 +757,36 @@ mod tests { fn burst_sized_flush_proceeds_on_remaining_credit() { let control = control(true); freeze_refill(&control); - assert!(control.try_flush(64, 1, false).is_some()); - assert!(control.try_flush(1_000_000, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); + assert!( + control + .try_acquire_flush_budget(1_000_000, 1, false) + .is_some() + ); restore_burst(&control); freeze_refill(&control); - assert!(control.try_flush(64, 1, false).is_some()); - assert!(control.try_flush(64, 1_000_000, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); + assert!( + control + .try_acquire_flush_budget(64, 1_000_000, false) + .is_some() + ); } #[test] fn refund_restores_nonessential_budget() { let control = control(true); freeze_refill(&control); - let mut last = FlushCharge { ops: 0, units: 0 }; + let mut last = FlushCharge { + records: 0, + byte_units: 0, + }; for _ in 0..10 { - last = control.try_flush(64, 1, false).unwrap(); + last = control.try_acquire_flush_budget(64, 1, false).unwrap(); } - assert!(control.try_flush(64, 1, false).is_none()); - control.refund_flush(last); - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_none()); + control.refund_flush_budget(last); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); } #[test] @@ -769,7 +799,7 @@ mod tests { .unwrap() .unwrap(); freeze_refill(&control); - while control.try_flush(64, 1, false).is_some() {} + while control.try_acquire_flush_budget(64, 1, false).is_some() {} let one_ms_ago = control.elapsed_ns().saturating_sub(1_000_000); control.last_refill_ns.store(one_ms_ago, Ordering::Relaxed); for _ in 0..32 { @@ -778,7 +808,7 @@ mod tests { assert_eq!(control.last_refill_ns.load(Ordering::Relaxed), one_ms_ago); std::thread::sleep(Duration::from_millis(110)); control.refill(); - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); } #[test] @@ -791,24 +821,27 @@ mod tests { .unwrap() .unwrap(); freeze_refill(&control); - while control.try_flush(64, 1, false).is_some() {} + while control.try_acquire_flush_budget(64, 1, false).is_some() {} let before = control.credit.load(Ordering::Relaxed); std::thread::scope(|scope| { for _ in 0..32 { let control = &control; scope.spawn(move || { - control.refund_flush(FlushCharge { ops: 1, units: 1 }); + control.refund_flush_budget(FlushCharge { + records: 1, + byte_units: 1, + }); }); } }); let credit = control.credit.load(Ordering::Relaxed); assert_eq!( - packed_ops(credit), - (packed_ops(before) + 32).min(control.max_ops) + packed_records(credit), + (packed_records(before) + 32).min(control.max_records) ); assert_eq!( - packed_units(credit), - (packed_units(before) + 32).min(control.max_units) + packed_byte_units(credit), + (packed_byte_units(before) + 32).min(control.max_byte_units) ); } @@ -817,16 +850,18 @@ mod tests { let control = control(true); freeze_refill(&control); for _ in 0..9 { - assert!(control.try_flush(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); } - let charge = control.try_flush(1_000_000, 1, false).unwrap(); - assert_eq!(charge.ops, 1); - assert_eq!(charge.units, control.max_units - 9); - control.refund_flush(charge); + let charge = control + .try_acquire_flush_budget(1_000_000, 1, false) + .unwrap(); + assert_eq!(charge.records, 1); + assert_eq!(charge.byte_units, control.max_byte_units - 9); + control.refund_flush_budget(charge); let credit = control.credit.load(Ordering::Relaxed); - assert_eq!(packed_ops(credit), 1); - assert_eq!(packed_units(credit), charge.units); - assert!(control.try_flush(64, 1, false).is_some()); - assert!(control.try_flush(64, 1, false).is_none()); + assert_eq!(packed_records(credit), 1); + assert_eq!(packed_byte_units(credit), charge.byte_units); + assert!(control.try_acquire_flush_budget(64, 1, false).is_some()); + assert!(control.try_acquire_flush_budget(64, 1, false).is_none()); } } From 33a0c77a55d6407aae0d3a017e3dbfe178da383f Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 08:41:28 +0800 Subject: [PATCH 11/12] refactor: separate fill control from background I/O recovery Let CacheRuntime own FillController and IoRecovery independently. Replace the broad recovery attempt with BackgroundIoAttempt, which borrows both policies for admission, slow-I/O observation, timeout recovery, and final publication. Move this coordination to io/background.rs and pass each attempt into Region writes instead of exposing the runtime's policies. Keep timeout and cancellation behavior, recovery fences, and completion ordering intact. Centralize stopping both policies in the runtime, remove the unused observation timeout argument, and name I/O observations and background waits by their role. Exercise recovery in Disabled, Observe, and Adaptive modes; verify that post-close completion cannot reopen fills and failed publication keeps recovery fenced after its I/O observation is released. --- cache2/src/cache/runtime/mod.rs | 47 ++++-- cache2/src/cache/runtime/shutdown_tests.rs | 148 ++++++++++------- cache2/src/io/{recovery.rs => background.rs} | 157 ++++++++++++------- cache2/src/io/engine/mod.rs | 18 +-- cache2/src/io/engine/tests.rs | 33 ++-- cache2/src/io/fill_control.rs | 37 ++--- cache2/src/io/mod.rs | 2 +- cache2/src/region/appender.rs | 18 +-- cache2/src/region/mod.rs | 4 +- cache2/src/region/persistence/tests.rs | 15 +- 10 files changed, 288 insertions(+), 191 deletions(-) rename cache2/src/io/{recovery.rs => background.rs} (55%) diff --git a/cache2/src/cache/runtime/mod.rs b/cache2/src/cache/runtime/mod.rs index 8097b03..ad068a0 100644 --- a/cache2/src/cache/runtime/mod.rs +++ b/cache2/src/cache/runtime/mod.rs @@ -51,6 +51,8 @@ use crate::config::runtime::read_io_wait_capacity; use crate::config::runtime::read_io_wait_timeout; use crate::config::storage_geometry; use crate::hashing::route_hash; +use crate::io::background::BackgroundIoAttempt; +use crate::io::background::IoRecovery; use crate::io::engine::IoBuffer; use crate::io::engine::IoEngine; use crate::io::engine::IoOperation; @@ -62,7 +64,6 @@ use crate::io::file::DataFileHandles; use crate::io::fill_control::FLUSH_RETRY; use crate::io::fill_control::FillController; use crate::io::fill_control::FlushCharge; -use crate::io::recovery::IoRecovery; use crate::managed_memory::BufferLease; use crate::managed_memory::CACHE_THREAD_STACK_BYTES; use crate::managed_memory::ManagedMemory; @@ -441,6 +442,7 @@ struct RuntimeState { reclaim_control: ReclaimControl, reclaim_io_timeout: Duration, io_recovery: IoRecovery, + fill_control: Option>, managed_memory: Arc, metrics: Arc, memory: Arc, @@ -512,6 +514,17 @@ impl ReclaimControl { } impl RuntimeState { + fn background_io_attempt(&self) -> BackgroundIoAttempt<'_> { + BackgroundIoAttempt::new(&self.io_recovery, self.fill_control.as_deref()) + } + + fn stop_fill_and_recovery(&self) { + self.io_recovery.stop(); + if let Some(fill) = &self.fill_control { + fill.stop(); + } + } + fn write_engine_for(&self, route: u64) -> &Arc { &self.write_engines[route_hash(route, self.write_engines.len())] } @@ -798,7 +811,7 @@ impl CacheRuntime { } pub fn start_close(&self) { - self.state.io_recovery.stop(); + self.state.stop_fill_and_recovery(); self.operations.start_close(); } @@ -847,7 +860,7 @@ impl CacheRuntime { return Err(write_overload_error()); } }; - if let Some(fill) = &state.io_recovery.fill + if let Some(fill) = &state.fill_control && !fill.try_admit_fill() { if state.activity_counters { @@ -1299,7 +1312,7 @@ impl CacheRuntime { { snapshot.health = crate::snapshot::CacheHealth::Recovering; } - if let Some(fill) = &state.io_recovery.fill { + if let Some(fill) = &state.fill_control { snapshot.fill_control = fill.snapshot(); } snapshot.io = aggregate_io_stats( @@ -1509,7 +1522,7 @@ fn start_workers( io::Error::new(io::ErrorKind::OutOfMemory, "cannot allocate shard controls") })?; append_controls.resize_with(shard_count, || Arc::new(AppendWorkerControl::new())); - let fill = FillController::new( + let fill_control = FillController::new( options.fill_control, shard_count + reclaim_worker_count, data.geometry.region_size, @@ -1523,7 +1536,8 @@ fn start_workers( reclaim_engines, reclaim_control: ReclaimControl::new(), reclaim_io_timeout: options.reclaim_io_timeout, - io_recovery: IoRecovery::with_fill(options.io_recovery_timeout, fill), + io_recovery: IoRecovery::new(options.io_recovery_timeout), + fill_control, managed_memory, metrics, memory, @@ -1665,7 +1679,7 @@ fn append_worker(state: Arc, shard_id: usize) { ); } state.regions.enter_miss_only(); - state.io_recovery.stop(); + state.stop_fill_and_recovery(); control.fail(&error); // Wake engine admission in case another shard is blocked behind work that // can no longer make progress after this runtime entered miss-only. @@ -1732,7 +1746,7 @@ fn reclaim_worker( .lifecycle .store(LIFECYCLE_FAILED, Ordering::Release); state.regions.enter_miss_only(); - state.io_recovery.stop(); + state.stop_fill_and_recovery(); log::error!( target: "cache2::health", event = "cache_reclaim_worker_failed", @@ -1769,7 +1783,7 @@ fn reclaim_worker_result( if state.reclaim_control.is_stopped()? { return Ok(()); } - let mut attempt = state.io_recovery.attempt(); + let mut attempt = state.background_io_attempt(); let Some(receipt) = state.regions.begin_reclaim()? else { break; }; @@ -1800,7 +1814,7 @@ fn reclaim_worker_result( ) .map_err(|error| error.into_lease().0)?; let completion = request - .wait_with_io_recovery(engine.as_ref(), &mut attempt) + .wait_background(engine.as_ref(), &mut attempt) .map_err(|error| error.into_lease().0)?; let (result, returned) = completion.into_lease(); let transferred = result?; @@ -1835,8 +1849,7 @@ fn reclaim_worker_result( let reinsert_shard = reinsert_shards.take(); let preserve_hot = state.regions.reclaim_can_reinsert()? && !state - .io_recovery - .fill + .fill_control .as_ref() .is_some_and(|fill| fill.suppress_reinsertion()); let reinsert_operation = if preserve_hot { @@ -1921,7 +1934,7 @@ fn append_worker_result( let essential = flags & (WAKE_URGENT | WAKE_ROTATE) != 0 || draining; let bytes = fill.bytes as u64; let records = u32::try_from(fill.records).unwrap_or(u32::MAX); - let charge = match &state.io_recovery.fill { + let charge = match &state.fill_control { None => Some(FlushCharge { records: 0, byte_units: 0, @@ -1936,11 +1949,11 @@ fn append_worker_result( &state.staging, engine.as_ref(), shard_id, - &state.io_recovery, + state.background_io_attempt(), )? { Some(_) => deadline = None, None => { - if let Some(control) = &state.io_recovery.fill { + if let Some(control) = &state.fill_control { control.refund_flush_budget(charge); } deadline = Some(Instant::now() + STAGING_RETRY_DELAY); @@ -1981,7 +1994,7 @@ fn append_worker_result( &state.staging, engine.as_ref(), shard_id, - &state.io_recovery, + state.background_io_attempt(), )?; } Ok(None) => {} @@ -2117,7 +2130,7 @@ async fn drain_shards_async(state: &RuntimeState, stop: bool) -> io::Result<()> } fn stop_workers(mut workers: RuntimeWorkers) -> io::Result { - workers.state.io_recovery.stop(); + workers.state.stop_fill_and_recovery(); let drain = drain_shards(&workers.state, true); let mut join_error = None; for worker in workers.append_workers.drain(..) { diff --git a/cache2/src/cache/runtime/shutdown_tests.rs b/cache2/src/cache/runtime/shutdown_tests.rs index 1d3c94c..8136b9a 100644 --- a/cache2/src/cache/runtime/shutdown_tests.rs +++ b/cache2/src/cache/runtime/shutdown_tests.rs @@ -176,69 +176,103 @@ fn assert_close_does_not_wait_for_read(submit_before_close: bool) { #[test] fn recovery_rejects_fills_preserves_reads_and_waits_for_all_workers() { + use crate::FillControlOptions; + use crate::FillLimits; + use crate::FillPressure; use crate::cache::session::CacheSession; use crate::region::persistence::RegionPaths; use crate::region::recovery::PersistentId; use crate::snapshot::CacheHealth; - let root = env::temp_dir().join(format!("cache2-recovery-admission-{}", std::process::id())); - std::fs::create_dir_all(&root).unwrap(); - let paths = RegionPaths::new(root.join("data"), root.join("state"), root.join("image")); - let data = DataSuperblock { - generation: 1, - cache_uuid: PersistentId::from_bytes([1; 16]).unwrap(), - data_identity: PersistentId::from_bytes([2; 16]).unwrap(), - geometry: DataGeometry { - data_file_len: DataGeometry::expected_file_len(4096, 4).unwrap(), - region_size: 4096, - region_count: 4, - }, - hash_seed: 3, - storage_fingerprint: 4, - }; - let config = RuntimeOptions { - append_shards: 1, - l1_capacity_bytes: 0, - ..RuntimeOptions::default() - }; - let mut session = CacheSession::for_test_with_options(paths, data, 8, config).unwrap(); - let runtime = session.runtime().unwrap().clone(); - runtime.put(b"existing", b"value").unwrap(); - runtime.drain().unwrap(); - let mut first = runtime.state.io_recovery.attempt(); - let mut second = runtime.state.io_recovery.attempt(); - assert!(first.next_deadline(Instant::now()).is_some()); - assert!(second.next_deadline(Instant::now()).is_some()); - assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Recovering); - assert_eq!( - runtime.put(b"new", b"value").unwrap_err().kind(), - io::ErrorKind::WouldBlock - ); - assert_eq!( - runtime.put_l2(b"new", b"value").unwrap_err().kind(), - io::ErrorKind::WouldBlock - ); - assert_eq!(runtime.get(b"existing").unwrap().unwrap().value(), b"value"); - runtime.delete(b"existing").unwrap(); - assert!(runtime.get(b"existing").unwrap().is_none()); - first.finish(); - assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Recovering); - second.finish(); - assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Running); - let deadline = Instant::now() + Duration::from_secs(2); - loop { - match runtime.put(b"new", b"value") { - Ok(_) => break, - Err(error) => { - assert_eq!(error.kind(), io::ErrorKind::WouldBlock); - assert!(Instant::now() < deadline, "fills did not resume"); - std::thread::sleep(Duration::from_millis(1)); + for fill_control in [ + FillControlOptions::Disabled, + FillControlOptions::Observe(FillLimits::new(1_048_576, 1000)), + FillControlOptions::Adaptive(FillLimits::new(1_048_576, 1000)), + ] { + let root = + env::temp_dir().join(format!("cache2-recovery-admission-{}", std::process::id())); + std::fs::create_dir_all(&root).unwrap(); + let paths = RegionPaths::new(root.join("data"), root.join("state"), root.join("image")); + let data = DataSuperblock { + generation: 1, + cache_uuid: PersistentId::from_bytes([1; 16]).unwrap(), + data_identity: PersistentId::from_bytes([2; 16]).unwrap(), + geometry: DataGeometry { + data_file_len: DataGeometry::expected_file_len(4096, 4).unwrap(), + region_size: 4096, + region_count: 4, + }, + hash_seed: 3, + storage_fingerprint: 4, + }; + let config = RuntimeOptions { + fill_control, + append_shards: 1, + l1_capacity_bytes: 0, + ..RuntimeOptions::default() + }; + let mut session = CacheSession::for_test_with_options(paths, data, 8, config).unwrap(); + let runtime = session.runtime().unwrap().clone(); + runtime.put(b"existing", b"value").unwrap(); + runtime.drain().unwrap(); + let mut first = runtime.state.background_io_attempt(); + let mut second = runtime.state.background_io_attempt(); + assert!(first.next_deadline(Instant::now()).is_some()); + assert!(second.next_deadline(Instant::now()).is_some()); + let snapshot = runtime.snapshot().unwrap(); + assert_eq!(snapshot.health, CacheHealth::Recovering); + if fill_control != FillControlOptions::Disabled { + assert_eq!(snapshot.fill_control.pressure, FillPressure::Paused); + } + assert_eq!( + runtime.put(b"new", b"value").unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!( + runtime.put_l2(b"new", b"value").unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(runtime.get(b"existing").unwrap().unwrap().value(), b"value"); + runtime.delete(b"existing").unwrap(); + assert!(runtime.get(b"existing").unwrap().is_none()); + first.finish(); + let snapshot = runtime.snapshot().unwrap(); + assert_eq!(snapshot.health, CacheHealth::Recovering); + if fill_control != FillControlOptions::Disabled { + assert_eq!(snapshot.fill_control.pressure, FillPressure::Paused); + } + second.finish(); + assert_eq!(runtime.snapshot().unwrap().health, CacheHealth::Running); + let deadline = Instant::now() + Duration::from_secs(2); + loop { + match runtime.put(b"new", b"value") { + Ok(_) => break, + Err(error) => { + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + assert!(Instant::now() < deadline, "fills did not resume"); + std::thread::sleep(Duration::from_millis(1)); + } } } + runtime.drain().unwrap(); + assert_eq!(runtime.get(b"new").unwrap().unwrap().value(), b"value"); + // A completion after close starts must release recovery without reopening fills. + let mut pending = runtime.state.background_io_attempt(); + assert!(pending.next_deadline(Instant::now()).is_some()); + runtime.start_close(); + pending.finish(); + assert!( + runtime + .state + .background_io_attempt() + .next_deadline(Instant::now()) + .is_none() + ); + if let Some(fill) = &runtime.state.fill_control { + assert_eq!(fill.snapshot().pressure, FillPressure::Paused); + } + session.close_fast().unwrap(); + std::fs::remove_dir_all(root).unwrap(); } - runtime.drain().unwrap(); - assert_eq!(runtime.get(b"new").unwrap().unwrap().value(), b"value"); - session.close_fast().unwrap(); - std::fs::remove_dir_all(root).unwrap(); } #[test] @@ -272,7 +306,7 @@ fn adaptive_pressure_preserves_reads_and_deletes_and_resumes_fills() { let runtime = session.runtime().unwrap().clone(); runtime.put(b"existing", b"value").unwrap(); runtime.drain().unwrap(); - let fill = runtime.state.io_recovery.fill.as_ref().unwrap(); + let fill = runtime.state.fill_control.as_ref().unwrap(); fill.set_recovering(true); let deadline = Instant::now() + Duration::from_secs(2); while fill.snapshot().pressure != crate::FillPressure::Paused { diff --git a/cache2/src/io/recovery.rs b/cache2/src/io/background.rs similarity index 55% rename from cache2/src/io/recovery.rs rename to cache2/src/io/background.rs index 0607e64..fb1c288 100644 --- a/cache2/src/io/recovery.rs +++ b/cache2/src/io/background.rs @@ -12,10 +12,9 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Recovery of timed-out background I/O while retaining issued requests. -//! This admission fence is separate from restart recovery and the terminal health latch. +//! Background I/O observation, timeout recovery, and publication boundaries. +//! Fill control and recovery are independent runtime-owned policies borrowed by each attempt. -use std::sync::Arc; use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicUsize; use std::sync::atomic::Ordering; @@ -23,14 +22,16 @@ use std::time::Duration; use std::time::Instant; use crate::io::fill_control::FillController; -use crate::io::fill_control::Observation; +use crate::io::fill_control::IoObservation; -/// Shared across background workers. Resource ownership remains with each worker. +const SLOW_IO_CHECK_CAP: Duration = Duration::from_millis(500); + +/// Shared timeout budget and admission fence for background I/O recovery. +/// Restart recovery and terminal Region health are separate concerns. pub struct IoRecovery { timeout: Option, pending: AtomicUsize, stopped: AtomicBool, - pub fill: Option>, } impl IoRecovery { @@ -39,25 +40,6 @@ impl IoRecovery { timeout, pending: AtomicUsize::new(0), stopped: AtomicBool::new(false), - fill: None, - } - } - - pub fn with_fill(timeout: Option, fill: Option>) -> Self { - Self { - fill, - ..Self::new(timeout) - } - } - - pub fn attempt(&self) -> IoRecoveryAttempt<'_> { - IoRecoveryAttempt { - io_recovery: self, - entered: false, - slow: false, - started: None, - io_timeout: Duration::ZERO, - observation: None, } } @@ -67,41 +49,74 @@ impl IoRecovery { pub fn stop(&self) { self.stopped.store(true, Ordering::Release); - if let Some(fill) = &self.fill { - fill.stop(); + } + + fn next_deadline(&self, original: Instant) -> Option { + if self.stopped.load(Ordering::Acquire) { + return None; } + let now = Instant::now(); + let deadline = now.checked_add(Duration::from_secs(1))?; + let deadline = match self.timeout { + Some(timeout) => { + let end = original.checked_add(timeout)?; + if now >= end { + return None; + } + deadline.min(end) + } + None => deadline, + }; + Some(deadline) } } -/// Retains the admission fence until the caller validates and publishes success. -/// Failed attempts deliberately leave the fence raised until instance teardown. -pub struct IoRecoveryAttempt<'a> { +/// Tracks one background operation from admission through validation and publication. +/// Slow I/O pauses fills until completion; timeout recovery keeps its fence until +/// successful publication. Failed recovery deliberately leaves the fence raised. +pub struct BackgroundIoAttempt<'a> { io_recovery: &'a IoRecovery, - entered: bool, + fill_control: Option<&'a FillController>, + recovering: bool, slow: bool, started: Option, io_timeout: Duration, - observation: Option>, + observation: Option>, } -impl IoRecoveryAttempt<'_> { +impl<'a> BackgroundIoAttempt<'a> { + pub const fn new( + io_recovery: &'a IoRecovery, + fill_control: Option<&'a FillController>, + ) -> Self { + Self { + io_recovery, + fill_control, + recovering: false, + slow: false, + started: None, + io_timeout: Duration::ZERO, + observation: None, + } + } + pub fn start(&mut self, bytes: u64, timeout: Duration) { self.started = Some(Instant::now()); self.io_timeout = timeout; - if let Some(fill) = &self.io_recovery.fill { - self.observation = fill.observe(bytes, timeout); + if let Some(fill) = self.fill_control { + self.observation = fill.observe_io(bytes); } } /// First wait bound: a fill checkpoint before the normal I/O timeout. - pub fn wait_cap(&self, original: Instant) -> Instant { - if self.io_recovery.fill.is_none() || self.slow { + pub fn wait_deadline(&self, original: Instant) -> Instant { + if self.fill_control.is_none() || self.slow { return original; } let Some(start) = self.started else { return original; }; - FillController::checkpoint(start, self.io_timeout).min(original) + (start + (self.io_timeout / 4).min(SLOW_IO_CHECK_CAP)).min(original) } pub fn note_slow(&mut self) { @@ -127,24 +142,10 @@ impl IoRecoveryAttempt<'_> { /// Poll completion/admission at fixed one-second intervals, without extending /// a configured total budget. Shutdown also terminates unlimited recovery. pub fn next_deadline(&mut self, original: Instant) -> Option { - if self.io_recovery.stopped.load(Ordering::Acquire) { - return None; - } - let now = Instant::now(); - let deadline = now.checked_add(Duration::from_secs(1))?; - let deadline = match self.io_recovery.timeout { - Some(timeout) => { - let end = original.checked_add(timeout)?; - if now >= end { - return None; - } - deadline.min(end) - } - None => deadline, - }; - if !self.entered { - self.entered = true; - if let Some(fill) = &self.io_recovery.fill { + let deadline = self.io_recovery.next_deadline(original)?; + if !self.recovering { + self.recovering = true; + if let Some(fill) = self.fill_control { fill.set_recovering(true); } if self.io_recovery.pending.fetch_add(1, Ordering::AcqRel) == 0 { @@ -160,9 +161,9 @@ impl IoRecoveryAttempt<'_> { if let Some(observation) = self.observation.take() { observation.finish(); } - if self.entered { + if self.recovering { let last = self.io_recovery.pending.fetch_sub(1, Ordering::AcqRel) == 1; - if let Some(fill) = &self.io_recovery.fill { + if let Some(fill) = self.fill_control { fill.set_recovering(false); } if last { @@ -172,3 +173,43 @@ impl IoRecoveryAttempt<'_> { } } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::FillControlOptions; + use crate::FillLimits; + use crate::FillPressure; + + #[test] + fn failed_publication_keeps_recovery_fenced_after_io_observation_ends() { + let fill = FillController::new( + FillControlOptions::Adaptive(FillLimits::new(64_000, 100)), + 2, + 4096, + ) + .unwrap() + .unwrap(); + let recovery = IoRecovery::new(None); + let mut failed = BackgroundIoAttempt::new(&recovery, Some(&fill)); + failed.start(4096, Duration::from_secs(1)); + failed.note_slow(); + assert_eq!(fill.snapshot().pressure, FillPressure::Paused); + assert!(!recovery.is_recovering()); + + assert!(failed.next_deadline(Instant::now()).is_some()); + failed.clear_slow(); + // I/O completed, but validation/publication failed before finish. + drop(failed); + assert_eq!(fill.snapshot().outstanding_operations, 0); + assert!(recovery.is_recovering()); + assert!(!fill.try_admit_fill()); + + let mut successful = BackgroundIoAttempt::new(&recovery, Some(&fill)); + successful.start(4096, Duration::from_secs(1)); + assert!(successful.next_deadline(Instant::now()).is_some()); + successful.finish(); + assert!(recovery.is_recovering()); + assert!(!fill.try_admit_fill()); + } +} diff --git a/cache2/src/io/engine/mod.rs b/cache2/src/io/engine/mod.rs index 13a64cc..eb4fb5a 100644 --- a/cache2/src/io/engine/mod.rs +++ b/cache2/src/io/engine/mod.rs @@ -48,6 +48,7 @@ use asyncband::semaphore::Semaphore; #[cfg(unix)] use crate::config::runtime::IoEngineConfig; +use crate::io::background::BackgroundIoAttempt; #[cfg(unix)] use crate::io::file::DataFileHandles; #[cfg(all( @@ -77,7 +78,6 @@ use crate::io::file::FileIoPath; use crate::io::file::FileIoStats; use crate::io::file::FileIoStatsHandle; use crate::io::file::WritePoint; -use crate::io::recovery::IoRecoveryAttempt; use crate::managed_memory::BufferLease; use crate::snapshot::CacheIoDirectionSnapshot; @@ -765,18 +765,18 @@ impl BoundedIoRequest { self.request.id() } - /// Allows an issued background operation to finish without cancelling it - /// at the normal deadline. The same request keeps its slot and buffer; - /// callers must still validate completion before publishing or reusing it. - pub fn wait_with_io_recovery( + /// Observes slow background I/O before its normal deadline, then applies + /// timeout recovery. The same request keeps its slot and buffer; callers + /// must still validate completion before publishing or reusing it. + pub fn wait_background( mut self, engine: &IoEngine, - attempt: &mut IoRecoveryAttempt<'_>, + attempt: &mut BackgroundIoAttempt<'_>, ) -> Result { let original = self.deadline; loop { let cap = if Instant::now() < original { - attempt.wait_cap(original) + attempt.wait_deadline(original) } else { self.deadline }; @@ -979,7 +979,7 @@ pub fn submit_background_io( engine: &IoEngine, mut operation: IoOperation, timeout: Duration, - attempt: &mut IoRecoveryAttempt<'_>, + attempt: &mut BackgroundIoAttempt<'_>, ) -> Result { let bytes = match &operation { IoOperation::Read { buffer, .. } | IoOperation::Write { buffer, .. } => buffer.len() as u64, @@ -988,7 +988,7 @@ pub fn submit_background_io( let original = Instant::now() .checked_add(timeout) .unwrap_or_else(Instant::now); - let mut deadline = attempt.wait_cap(original); + let mut deadline = attempt.wait_deadline(original); loop { match submit_cache_io_until(engine, operation, deadline, CACHE_IO_CANCEL_GRACE) { Ok(mut request) => { diff --git a/cache2/src/io/engine/tests.rs b/cache2/src/io/engine/tests.rs index 8916ad6..cfce4f8 100644 --- a/cache2/src/io/engine/tests.rs +++ b/cache2/src/io/engine/tests.rs @@ -20,8 +20,9 @@ use crate::IoOutcome; use crate::IoRole; use crate::StatsOptions; use crate::fixtures::TestFile; +use crate::io::background::BackgroundIoAttempt; +use crate::io::background::IoRecovery; use crate::io::file::PositionedIo; -use crate::io::recovery::IoRecovery; use crate::managed_memory::ManagedMemory; use crate::managed_memory::ManagedMemoryLimits; use crate::managed_memory::aligned_buffer_capacity; @@ -1128,9 +1129,12 @@ fn background_recovery_keeps_the_original_request_and_accepts_late_completion() let (tx, rx) = mpsc::channel(); let engine = &engine; scope.spawn(move || { - tx.send(request.wait_with_io_recovery( + tx.send(request.wait_background( engine, - &mut IoRecovery::new(Some(Duration::from_secs(2))).attempt(), + &mut BackgroundIoAttempt::new( + &IoRecovery::new(Some(Duration::from_secs(2))), + None, + ), )) .unwrap(); }); @@ -1175,9 +1179,9 @@ fn exhausted_background_recovery_still_fences_unfinished_writes() { assert!(io.wait_for_entered(1)); request.deadline = Instant::now(); request.cancel_grace = Duration::from_millis(10); - let result = request.wait_with_io_recovery( + let result = request.wait_background( &engine, - &mut IoRecovery::new(Some(Duration::from_millis(20))).attempt(), + &mut BackgroundIoAttempt::new(&IoRecovery::new(Some(Duration::from_millis(20))), None), ); let pending = engine.writes_in_flight(); let rejected = engine.submit(IoOperation::write( @@ -1212,14 +1216,14 @@ fn background_admission_recovers_without_duplicate_submission() { let managed_memory = &managed_memory; scope.spawn(move || { let io_recovery = IoRecovery::new(Some(Duration::from_secs(2))); - let mut attempt = io_recovery.attempt(); + let mut attempt = BackgroundIoAttempt::new(&io_recovery, None); let result = submit_background_io( engine, IoOperation::read(read_buffer(managed_memory, 4096), 0), Duration::from_millis(10), &mut attempt, ) - .map(|request| request.wait_with_io_recovery(engine, &mut attempt)); + .map(|request| request.wait_background(engine, &mut attempt)); tx.send(result).unwrap(); }); let early = rx.recv_timeout(Duration::from_millis(50)); @@ -1263,8 +1267,8 @@ fn unlimited_recovery_keeps_admission_paused_until_validation() { let io_recovery = &io_recovery; let engine = &engine; scope.spawn(move || { - let mut attempt = io_recovery.attempt(); - let completion = request.wait_with_io_recovery(engine, &mut attempt).unwrap(); + let mut attempt = BackgroundIoAttempt::new(io_recovery, None); + let completion = request.wait_background(engine, &mut attempt).unwrap(); completed_tx.send(completion).unwrap(); validate_rx.recv().unwrap(); attempt.finish(); @@ -1310,8 +1314,8 @@ fn shutdown_interrupts_unlimited_recovery_without_releasing_pending_write() { let io_recovery = &io_recovery; let engine = &engine; scope.spawn(move || { - let mut attempt = io_recovery.attempt(); - tx.send(request.wait_with_io_recovery(engine, &mut attempt)) + let mut attempt = BackgroundIoAttempt::new(io_recovery, None); + tx.send(request.wait_background(engine, &mut attempt)) .unwrap(); }); let early = rx.recv_timeout(Duration::from_millis(30)); @@ -1344,7 +1348,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp FillControlOptions::Observe(settings) }; let control = FillController::new(mode, 1, 4096).unwrap().unwrap(); - let io_recovery = IoRecovery::with_fill(None, Some(Arc::clone(&control))); + let io_recovery = IoRecovery::new(None); let io = Arc::new(BlockingIo::default()); let engine = IoEngine::for_test(io.clone(), 1).unwrap(); let memory = managed_memory(); @@ -1352,10 +1356,11 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp let (returned_tx, returned_rx) = mpsc::channel(); let (validate_tx, validate_rx) = mpsc::channel(); let io_recovery = &io_recovery; + let control = &control; let engine = &engine; let memory = &memory; scope.spawn(move || { - let mut attempt = io_recovery.attempt(); + let mut attempt = BackgroundIoAttempt::new(io_recovery, Some(control)); let request = submit_background_io( engine, IoOperation::write(WritePoint::Record, write_buffer(memory, &[5; 4096]), 0), @@ -1363,7 +1368,7 @@ fn adaptive_pressure_pauses_before_real_engine_timeout_and_resumes_after_io_comp &mut attempt, ) .unwrap(); - let completion = request.wait_with_io_recovery(engine, &mut attempt).unwrap(); + let completion = request.wait_background(engine, &mut attempt).unwrap(); returned_tx.send(completion).unwrap(); validate_rx.recv().unwrap(); attempt.finish(); diff --git a/cache2/src/io/fill_control.rs b/cache2/src/io/fill_control.rs index fa537da..2117695 100644 --- a/cache2/src/io/fill_control.rs +++ b/cache2/src/io/fill_control.rs @@ -35,7 +35,6 @@ const MIN_BYTES_PER_SECOND: u64 = 640; const MAX_BYTES_PER_SECOND: u64 = 1 << 40; const MIN_RECORDS_PER_SECOND: u32 = 10; const MAX_RECORDS_PER_SECOND: u32 = u32::MAX; -const STALL_CAP: Duration = Duration::from_millis(500); const TICKS_PER_SECOND: u64 = 10; /// Wake shard workers this often when a non-essential flush is waiting for budget. @@ -182,10 +181,6 @@ impl FillController { }))) } - pub fn checkpoint(start: Instant, timeout: Duration) -> Instant { - start + (timeout / 4).min(STALL_CAP) - } - fn elapsed_ns(&self) -> u64 { nanos(self.origin.elapsed()) } @@ -437,7 +432,7 @@ impl FillController { } } - pub fn observe(&self, bytes: u64, _timeout: Duration) -> Option> { + pub fn observe_io(&self, bytes: u64) -> Option> { let start = self.elapsed_ns().max(1); for (index, slot) in self.slots.iter().enumerate() { if slot @@ -446,7 +441,7 @@ impl FillController { .is_ok() { slot.bytes.store(bytes, Ordering::Relaxed); - return Some(Observation { + return Some(IoObservation { control: self, index, succeeded: false, @@ -460,13 +455,13 @@ impl FillController { } /// One worker-owned observation. Failed work is removed, never marked successful. -pub struct Observation<'a> { +pub struct IoObservation<'a> { control: &'a FillController, index: usize, succeeded: bool, slow: bool, } -impl Observation<'_> { +impl IoObservation<'_> { pub fn note_slow(&mut self) { if self.slow { return; @@ -487,7 +482,7 @@ impl Observation<'_> { self.succeeded = true; } } -impl Drop for Observation<'_> { +impl Drop for IoObservation<'_> { fn drop(&mut self) { let slot = &self.control.slots[self.index]; let bytes = slot.bytes.load(Ordering::Relaxed); @@ -573,8 +568,8 @@ mod tests { #[test] fn old_request_pauses_while_other_requests_progress() { let control = control(true); - let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); - let fast = control.observe(64, Duration::from_secs(2)).unwrap(); + let mut old = control.observe_io(4096).unwrap(); + let fast = control.observe_io(64).unwrap(); fast.finish(); old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); @@ -587,7 +582,7 @@ mod tests { #[test] fn observe_never_rejects_or_suppresses_reinsertion() { let control = control(false); - let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); + let mut old = control.observe_io(4096).unwrap(); old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); for _ in 0..20 { @@ -606,7 +601,7 @@ mod tests { assert!(control.try_admit_fill()); } assert_eq!(control.snapshot().would_reject, 0); - let mut old = control.observe(4096, Duration::from_secs(2)).unwrap(); + let mut old = control.observe_io(4096).unwrap(); old.note_slow(); assert_eq!(control.snapshot().pressure, FillPressure::Paused); assert!(control.try_admit_fill()); @@ -674,11 +669,11 @@ mod tests { ) .unwrap() .unwrap(); - let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); - assert!(control.observe(64, Duration::from_secs(5)).is_none()); + let pending = control.observe_io(4096).unwrap(); + assert!(control.observe_io(64).is_none()); assert_eq!(control.snapshot().dropped_observations, 1); drop(pending); - let pending = control.observe(4096, Duration::from_secs(5)).unwrap(); + let pending = control.observe_io(4096).unwrap(); drop(pending); assert_eq!(control.completed_io_operations.load(Ordering::Relaxed), 0); assert_eq!(control.completed_bytes.load(Ordering::Relaxed), 0); @@ -709,7 +704,7 @@ mod tests { #[test] fn io_completion_releases_pause_before_observation_drop() { let control = control(true); - let mut pending = control.observe(4096, Duration::from_secs(2)).unwrap(); + let mut pending = control.observe_io(4096).unwrap(); pending.note_slow(); assert!(!control.try_admit_fill()); pending.clear_slow(); @@ -725,7 +720,7 @@ mod tests { for _ in 0..8 { let control = &control; scope.spawn(move || { - let mut pending = control.observe(64, Duration::from_secs(2)).unwrap(); + let mut pending = control.observe_io(64).unwrap(); pending.note_slow(); assert!(!control.try_admit_fill()); drop(pending); @@ -741,8 +736,8 @@ mod tests { fn overlapping_hold_and_release_cannot_stick_paused() { let control = control(true); for _ in 0..1_000 { - let mut first = control.observe(64, Duration::from_secs(2)).unwrap(); - let mut second = control.observe(64, Duration::from_secs(2)).unwrap(); + let mut first = control.observe_io(64).unwrap(); + let mut second = control.observe_io(64).unwrap(); first.note_slow(); second.note_slow(); drop(first); diff --git a/cache2/src/io/mod.rs b/cache2/src/io/mod.rs index 183c345..3e3c512 100644 --- a/cache2/src/io/mod.rs +++ b/cache2/src/io/mod.rs @@ -14,8 +14,8 @@ //! Positioned file access and bounded owned-buffer execution. +pub mod background; pub mod engine; pub mod file; pub mod fill_control; pub mod fs; -pub mod recovery; diff --git a/cache2/src/region/appender.rs b/cache2/src/region/appender.rs index 41ce56c..53f9091 100644 --- a/cache2/src/region/appender.rs +++ b/cache2/src/region/appender.rs @@ -23,6 +23,9 @@ use std::io; #[cfg(test)] use std::time::Duration; +use crate::io::background::BackgroundIoAttempt; +#[cfg(test)] +use crate::io::background::IoRecovery; use crate::io::engine::BoundedIoRequest; use crate::io::engine::CACHE_IO_COMPLETION_TIMEOUT; use crate::io::engine::IoBuffer; @@ -33,9 +36,6 @@ use crate::io::engine::RequestId; use crate::io::engine::submit_background_io; use crate::io::file::DIRECT_IO_ALIGNMENT; use crate::io::file::WritePoint; -#[cfg(test)] -use crate::io::recovery::IoRecovery; -use crate::io::recovery::IoRecoveryAttempt; use crate::region::manager::RegionWriteSpan; use crate::region::recovery::DATA_REGION_AREA_OFFSET; use crate::region::recovery::DataGeometry; @@ -83,8 +83,8 @@ pub struct WriteCompletion { } impl PendingWrite { - pub fn wait(self, engine: &IoEngine, attempt: &mut IoRecoveryAttempt<'_>) -> WriteCompletion { - let completion = match self.request.wait_with_io_recovery(engine, attempt) { + pub fn wait(self, engine: &IoEngine, attempt: &mut BackgroundIoAttempt<'_>) -> WriteCompletion { + let completion = match self.request.wait_background(engine, attempt) { Ok(completion) => completion, Err(timeout) => { let (error, buffer) = timeout.into_buffer(); @@ -143,7 +143,7 @@ pub fn submit_write( span: RegionWriteSpan, buffer: IoBuffer, absolute: u64, - attempt: &mut IoRecoveryAttempt<'_>, + attempt: &mut BackgroundIoAttempt<'_>, ) -> Result { let (expected_len, expected_absolute) = match validate_span(geometry, span) { Ok(validated) => validated, @@ -308,7 +308,7 @@ mod tests { lease.prepare(4096).unwrap().fill(0x5a); let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; let io_recovery = IoRecovery::new(Some(Duration::from_secs(5))); - let mut attempt = io_recovery.attempt(); + let mut attempt = BackgroundIoAttempt::new(&io_recovery, None); let completion = submit_write( &engine, geometry(), @@ -342,7 +342,7 @@ mod tests { let buffer = IoBuffer::for_write(lease, 4096).unwrap(); let absolute = DATA_REGION_AREA_OFFSET + geometry().region_size; let io_recovery = IoRecovery::new(Some(Duration::ZERO)); - let mut attempt = io_recovery.attempt(); + let mut attempt = BackgroundIoAttempt::new(&io_recovery, None); let completion = submit_write(&engine, geometry(), span(), buffer, absolute, &mut attempt) .unwrap() .wait(&engine, &mut attempt); @@ -379,7 +379,7 @@ mod tests { invalid, buffer, 0, - &mut IoRecovery::new(Some(Duration::ZERO)).attempt(), + &mut BackgroundIoAttempt::new(&IoRecovery::new(Some(Duration::ZERO)), None), ) { Err(error) => error, Ok(_) => panic!("unaligned span must not be submitted"), diff --git a/cache2/src/region/mod.rs b/cache2/src/region/mod.rs index 1f8bff0..2bb9c1c 100644 --- a/cache2/src/region/mod.rs +++ b/cache2/src/region/mod.rs @@ -54,6 +54,7 @@ use self::staging::StagingEncodeError; use self::staging::StagingError; use crate::checksum::crc32c; use crate::hashing::route_hash; +use crate::io::background::BackgroundIoAttempt; use crate::io::engine::IoBuffer; use crate::io::engine::IoEngine; use crate::io::engine::ReadSlot; @@ -1029,9 +1030,8 @@ impl RegionStore { staging: &AppendStaging, engine: &IoEngine, shard_id: usize, - io_recovery: &crate::io::recovery::IoRecovery, + mut attempt: BackgroundIoAttempt<'_>, ) -> io::Result> { - let mut attempt = io_recovery.attempt(); let shard_mutation = self.lock_shard_mutation(shard_id)?; let geometry_for = |manager: &RegionManager| { let region_count = u32::try_from(manager.regions().len()).map_err(|_| { diff --git a/cache2/src/region/persistence/tests.rs b/cache2/src/region/persistence/tests.rs index 476e2a6..3836ad6 100644 --- a/cache2/src/region/persistence/tests.rs +++ b/cache2/src/region/persistence/tests.rs @@ -924,7 +924,10 @@ fn completed_owned_span_publishes_index_without_a_steady_state_sync() { &staging, &engine, 0, - &crate::io::recovery::IoRecovery::new(Some(std::time::Duration::ZERO)), + crate::io::background::BackgroundIoAttempt::new( + &crate::io::background::IoRecovery::new(Some(std::time::Duration::ZERO)), + None, + ), ) .unwrap() .unwrap(); @@ -1077,7 +1080,10 @@ fn same_hash_candidate_requires_full_key() { &staging, &engine, 0, - &crate::io::recovery::IoRecovery::new(Some(std::time::Duration::ZERO)), + crate::io::background::BackgroundIoAttempt::new( + &crate::io::background::IoRecovery::new(Some(std::time::Duration::ZERO)), + None, + ), ) .unwrap() .expect("collision owner must publish"); @@ -1217,7 +1223,10 @@ fn failed_span_write_never_publishes_and_latches_miss_only() { &staging, &engine, 0, - &crate::io::recovery::IoRecovery::new(Some(std::time::Duration::ZERO)) + crate::io::background::BackgroundIoAttempt::new( + &crate::io::background::IoRecovery::new(Some(std::time::Duration::ZERO)), + None, + ) ) .unwrap_err() .raw_os_error(), From 68ffadbed5abe6ab027179531c68c7d99e0ddeb6 Mon Sep 17 00:00:00 2001 From: tison Date: Mon, 21 Sep 2026 08:41:37 +0800 Subject: [PATCH 12/12] docs: clarify background I/O policy ownership Describe independent fill control and timeout recovery, the per-operation BackgroundIoAttempt, and the runtime's coordinated stop boundary. Distinguish record and byte flush budgets from observed I/O request counts. --- ARCHITECTURE.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 19add0b..d7bcc06 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -48,6 +48,8 @@ The private implementation separates cache execution, live L2 state, and persist The I/O layer names capabilities and resources directly. `PositionedIo` is synchronous positioned read/write access; `StorageFile` adds file identity, extent management, locking, sync, and handle duplication. `CacheFile` is the OS file resource, while `DataFileHandles` holds its buffered/direct descriptors for engine use. `FileSystem` covers path operations such as create, rename, and parent-directory sync; `OsFileSystem` supplies those operations. These narrow interfaces also support fault injection. Region lifecycle and storage have one concrete implementation and no backend trait. +`CacheRuntime` owns `IoRecovery` and the optional `FillController` independently. `IoRecovery` tracks timeout recovery and its admission fence; `FillController` observes I/O pressure, gates new fills, and paces non-essential flushes. A `BackgroundIoAttempt` in `io/background.rs` borrows both for one operation, from admission through completion validation and publication. The runtime stops both policies before waiting for mutation quiescence. + Three states serve different purposes: `RegionState` is Free/Active/Sealed, persisted `SessionState` is EMPTY/RUNNING/CLEAN, and `IoRecovery` tracks reversible background I/O timeout recovery. The Region health latch is a separate terminal failure boundary. Recovery of an I/O request cannot restore an unhealthy Region store or make a restart image eligible. ## Consistency model @@ -142,6 +144,8 @@ Each lane uses one concrete `IoEngine` for admission, submission, cancellation, Adaptive foreground `put`/`put_l2` load pause-holder state and otherwise only compete for staging. Byte and record ceilings are instance-wide, shared by every shard worker; they pace non-essential background flush. Urgent, drain, and rotation flushes always proceed. A flush that cannot take a span refunds the consumed budget and retries. `Observe` does not delay flush or reject fills and counts pause refusals as `would_reject`; it still checkpoints waits so pause is observable. Adaptive rejects new fills immediately while paused and skips optional reinsertion. Pause is released when the slow I/O completes, not after later publication. Reads, deletes, and essential reclaim bypass fill budgets. Close stops admission independently of outstanding I/O. +Flush budgets count encoded bytes and records, while pressure observations count I/O requests. One batched write may contain many records. `try_acquire_flush_budget` returns a `FlushCharge` in record credits and byte units; acquiring this budget does not submit I/O. `refund_flush_budget` returns only the credits that were consumed. + See [adaptive fill admission](CONFIGURATION.md#adaptive-fill-admission) for rate ceilings, pause conditions, bounded bursts, and tuning limits. ### Memory @@ -219,6 +223,6 @@ Stale data, overload, and cache loss are valid outcomes. Every returned value pa ### Background I/O timeout recovery -A background write or reclaim admission/completion timeout enters a reversible recovery state separate from the terminal Region health latch. `RuntimeOptions::io_recovery_timeout` defaults to `None` (until completion or close); `Some(duration)` bounds the additional wait and `Some(Duration::ZERO)` disables recovery. Each affected worker holds an `IoRecoveryAttempt` through validation and publication. A shared pending count fences new foreground fills until all attempts succeed; reads and deletes remain independent, and previously accepted work may finish. The snapshot reports `CacheHealth::Recovering` unless a draining or terminal state takes precedence. Workers use fixed one-second admission/completion checks anchored to the original deadline for finite budgets. Only unsubmitted operations returned by admission can be retried. Issued requests retain their worker, slot, buffer, staging span, and Region until completion. No new queue or worker is introduced, and memory remains bounded regardless of recovery duration. +A background write or reclaim admission/completion timeout enters a reversible recovery state separate from the terminal Region health latch. `RuntimeOptions::io_recovery_timeout` defaults to `None` (until completion or close); `Some(duration)` bounds the additional wait and `Some(Duration::ZERO)` disables recovery. Each affected worker holds a `BackgroundIoAttempt` through validation and publication. A shared pending count fences new foreground fills until all attempts succeed; reads and deletes remain independent, and previously accepted work may finish. The snapshot reports `CacheHealth::Recovering` unless a draining or terminal state takes precedence. Workers use fixed one-second admission/completion checks anchored to the original deadline for finite budgets. Only unsubmitted operations returned by admission can be retried. Issued requests retain their worker, slot, buffer, staging span, and Region until completion. No new queue or worker is introduced, and memory remains bounded regardless of recovery duration. `cache_io_recovery_started` and `cache_io_recovery_completed` log transitions into recovery and out of it after all affected operations pass validation and publication. Failed attempts never lower the admission fence. Actual I/O errors, invalid completions, and finite-budget exhaustion retain the terminal failure policy; this implementation does not retry completed I/O failures or reopen a failed instance. Closing or a background worker failure stops recovery, observed within one polling interval; normal cancellation then preserves the unfenced-write boundary and prevents unsafe CLEAN publication. Close starts this interruption before waiting for mutation quiescence, including reclaim reinsertion. Drain remains a completion fence and may wait indefinitely in unlimited recovery mode.